1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This provides a class for OpenMP runtime code generation. 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "CGOpenMPRuntime.h" 14 #include "CGCXXABI.h" 15 #include "CGCleanup.h" 16 #include "CGRecordLayout.h" 17 #include "CodeGenFunction.h" 18 #include "TargetInfo.h" 19 #include "clang/AST/APValue.h" 20 #include "clang/AST/Attr.h" 21 #include "clang/AST/Decl.h" 22 #include "clang/AST/OpenMPClause.h" 23 #include "clang/AST/StmtOpenMP.h" 24 #include "clang/AST/StmtVisitor.h" 25 #include "clang/Basic/BitmaskEnum.h" 26 #include "clang/Basic/FileManager.h" 27 #include "clang/Basic/OpenMPKinds.h" 28 #include "clang/Basic/SourceManager.h" 29 #include "clang/CodeGen/ConstantInitBuilder.h" 30 #include "llvm/ADT/ArrayRef.h" 31 #include "llvm/ADT/SetOperations.h" 32 #include "llvm/ADT/SmallBitVector.h" 33 #include "llvm/ADT/StringExtras.h" 34 #include "llvm/Bitcode/BitcodeReader.h" 35 #include "llvm/IR/Constants.h" 36 #include "llvm/IR/DerivedTypes.h" 37 #include "llvm/IR/GlobalValue.h" 38 #include "llvm/IR/InstrTypes.h" 39 #include "llvm/IR/Value.h" 40 #include "llvm/Support/AtomicOrdering.h" 41 #include "llvm/Support/Format.h" 42 #include "llvm/Support/raw_ostream.h" 43 #include <cassert> 44 #include <numeric> 45 46 using namespace clang; 47 using namespace CodeGen; 48 using namespace llvm::omp; 49 50 namespace { 51 /// Base class for handling code generation inside OpenMP regions. 52 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo { 53 public: 54 /// Kinds of OpenMP regions used in codegen. 55 enum CGOpenMPRegionKind { 56 /// Region with outlined function for standalone 'parallel' 57 /// directive. 58 ParallelOutlinedRegion, 59 /// Region with outlined function for standalone 'task' directive. 60 TaskOutlinedRegion, 61 /// Region for constructs that do not require function outlining, 62 /// like 'for', 'sections', 'atomic' etc. directives. 63 InlinedRegion, 64 /// Region with outlined function for standalone 'target' directive. 65 TargetRegion, 66 }; 67 68 CGOpenMPRegionInfo(const CapturedStmt &CS, 69 const CGOpenMPRegionKind RegionKind, 70 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 71 bool HasCancel) 72 : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind), 73 CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {} 74 75 CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind, 76 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 77 bool HasCancel) 78 : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen), 79 Kind(Kind), HasCancel(HasCancel) {} 80 81 /// Get a variable or parameter for storing global thread id 82 /// inside OpenMP construct. 83 virtual const VarDecl *getThreadIDVariable() const = 0; 84 85 /// Emit the captured statement body. 86 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override; 87 88 /// Get an LValue for the current ThreadID variable. 89 /// \return LValue for thread id variable. This LValue always has type int32*. 90 virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF); 91 92 virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {} 93 94 CGOpenMPRegionKind getRegionKind() const { return RegionKind; } 95 96 OpenMPDirectiveKind getDirectiveKind() const { return Kind; } 97 98 bool hasCancel() const { return HasCancel; } 99 100 static bool classof(const CGCapturedStmtInfo *Info) { 101 return Info->getKind() == CR_OpenMP; 102 } 103 104 ~CGOpenMPRegionInfo() override = default; 105 106 protected: 107 CGOpenMPRegionKind RegionKind; 108 RegionCodeGenTy CodeGen; 109 OpenMPDirectiveKind Kind; 110 bool HasCancel; 111 }; 112 113 /// API for captured statement code generation in OpenMP constructs. 114 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo { 115 public: 116 CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar, 117 const RegionCodeGenTy &CodeGen, 118 OpenMPDirectiveKind Kind, bool HasCancel, 119 StringRef HelperName) 120 : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind, 121 HasCancel), 122 ThreadIDVar(ThreadIDVar), HelperName(HelperName) { 123 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 124 } 125 126 /// Get a variable or parameter for storing global thread id 127 /// inside OpenMP construct. 128 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 129 130 /// Get the name of the capture helper. 131 StringRef getHelperName() const override { return HelperName; } 132 133 static bool classof(const CGCapturedStmtInfo *Info) { 134 return CGOpenMPRegionInfo::classof(Info) && 135 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 136 ParallelOutlinedRegion; 137 } 138 139 private: 140 /// A variable or parameter storing global thread id for OpenMP 141 /// constructs. 142 const VarDecl *ThreadIDVar; 143 StringRef HelperName; 144 }; 145 146 /// API for captured statement code generation in OpenMP constructs. 147 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo { 148 public: 149 class UntiedTaskActionTy final : public PrePostActionTy { 150 bool Untied; 151 const VarDecl *PartIDVar; 152 const RegionCodeGenTy UntiedCodeGen; 153 llvm::SwitchInst *UntiedSwitch = nullptr; 154 155 public: 156 UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar, 157 const RegionCodeGenTy &UntiedCodeGen) 158 : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {} 159 void Enter(CodeGenFunction &CGF) override { 160 if (Untied) { 161 // Emit task switching point. 162 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 163 CGF.GetAddrOfLocalVar(PartIDVar), 164 PartIDVar->getType()->castAs<PointerType>()); 165 llvm::Value *Res = 166 CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation()); 167 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done."); 168 UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB); 169 CGF.EmitBlock(DoneBB); 170 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 171 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 172 UntiedSwitch->addCase(CGF.Builder.getInt32(0), 173 CGF.Builder.GetInsertBlock()); 174 emitUntiedSwitch(CGF); 175 } 176 } 177 void emitUntiedSwitch(CodeGenFunction &CGF) const { 178 if (Untied) { 179 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 180 CGF.GetAddrOfLocalVar(PartIDVar), 181 PartIDVar->getType()->castAs<PointerType>()); 182 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 183 PartIdLVal); 184 UntiedCodeGen(CGF); 185 CodeGenFunction::JumpDest CurPoint = 186 CGF.getJumpDestInCurrentScope(".untied.next."); 187 CGF.EmitBranch(CGF.ReturnBlock.getBlock()); 188 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 189 UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 190 CGF.Builder.GetInsertBlock()); 191 CGF.EmitBranchThroughCleanup(CurPoint); 192 CGF.EmitBlock(CurPoint.getBlock()); 193 } 194 } 195 unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); } 196 }; 197 CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS, 198 const VarDecl *ThreadIDVar, 199 const RegionCodeGenTy &CodeGen, 200 OpenMPDirectiveKind Kind, bool HasCancel, 201 const UntiedTaskActionTy &Action) 202 : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel), 203 ThreadIDVar(ThreadIDVar), Action(Action) { 204 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 205 } 206 207 /// Get a variable or parameter for storing global thread id 208 /// inside OpenMP construct. 209 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 210 211 /// Get an LValue for the current ThreadID variable. 212 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override; 213 214 /// Get the name of the capture helper. 215 StringRef getHelperName() const override { return ".omp_outlined."; } 216 217 void emitUntiedSwitch(CodeGenFunction &CGF) override { 218 Action.emitUntiedSwitch(CGF); 219 } 220 221 static bool classof(const CGCapturedStmtInfo *Info) { 222 return CGOpenMPRegionInfo::classof(Info) && 223 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 224 TaskOutlinedRegion; 225 } 226 227 private: 228 /// A variable or parameter storing global thread id for OpenMP 229 /// constructs. 230 const VarDecl *ThreadIDVar; 231 /// Action for emitting code for untied tasks. 232 const UntiedTaskActionTy &Action; 233 }; 234 235 /// API for inlined captured statement code generation in OpenMP 236 /// constructs. 237 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo { 238 public: 239 CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI, 240 const RegionCodeGenTy &CodeGen, 241 OpenMPDirectiveKind Kind, bool HasCancel) 242 : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel), 243 OldCSI(OldCSI), 244 OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {} 245 246 // Retrieve the value of the context parameter. 247 llvm::Value *getContextValue() const override { 248 if (OuterRegionInfo) 249 return OuterRegionInfo->getContextValue(); 250 llvm_unreachable("No context value for inlined OpenMP region"); 251 } 252 253 void setContextValue(llvm::Value *V) override { 254 if (OuterRegionInfo) { 255 OuterRegionInfo->setContextValue(V); 256 return; 257 } 258 llvm_unreachable("No context value for inlined OpenMP region"); 259 } 260 261 /// Lookup the captured field decl for a variable. 262 const FieldDecl *lookup(const VarDecl *VD) const override { 263 if (OuterRegionInfo) 264 return OuterRegionInfo->lookup(VD); 265 // If there is no outer outlined region,no need to lookup in a list of 266 // captured variables, we can use the original one. 267 return nullptr; 268 } 269 270 FieldDecl *getThisFieldDecl() const override { 271 if (OuterRegionInfo) 272 return OuterRegionInfo->getThisFieldDecl(); 273 return nullptr; 274 } 275 276 /// Get a variable or parameter for storing global thread id 277 /// inside OpenMP construct. 278 const VarDecl *getThreadIDVariable() const override { 279 if (OuterRegionInfo) 280 return OuterRegionInfo->getThreadIDVariable(); 281 return nullptr; 282 } 283 284 /// Get an LValue for the current ThreadID variable. 285 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override { 286 if (OuterRegionInfo) 287 return OuterRegionInfo->getThreadIDVariableLValue(CGF); 288 llvm_unreachable("No LValue for inlined OpenMP construct"); 289 } 290 291 /// Get the name of the capture helper. 292 StringRef getHelperName() const override { 293 if (auto *OuterRegionInfo = getOldCSI()) 294 return OuterRegionInfo->getHelperName(); 295 llvm_unreachable("No helper name for inlined OpenMP construct"); 296 } 297 298 void emitUntiedSwitch(CodeGenFunction &CGF) override { 299 if (OuterRegionInfo) 300 OuterRegionInfo->emitUntiedSwitch(CGF); 301 } 302 303 CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; } 304 305 static bool classof(const CGCapturedStmtInfo *Info) { 306 return CGOpenMPRegionInfo::classof(Info) && 307 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion; 308 } 309 310 ~CGOpenMPInlinedRegionInfo() override = default; 311 312 private: 313 /// CodeGen info about outer OpenMP region. 314 CodeGenFunction::CGCapturedStmtInfo *OldCSI; 315 CGOpenMPRegionInfo *OuterRegionInfo; 316 }; 317 318 /// API for captured statement code generation in OpenMP target 319 /// constructs. For this captures, implicit parameters are used instead of the 320 /// captured fields. The name of the target region has to be unique in a given 321 /// application so it is provided by the client, because only the client has 322 /// the information to generate that. 323 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo { 324 public: 325 CGOpenMPTargetRegionInfo(const CapturedStmt &CS, 326 const RegionCodeGenTy &CodeGen, StringRef HelperName) 327 : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target, 328 /*HasCancel=*/false), 329 HelperName(HelperName) {} 330 331 /// This is unused for target regions because each starts executing 332 /// with a single thread. 333 const VarDecl *getThreadIDVariable() const override { return nullptr; } 334 335 /// Get the name of the capture helper. 336 StringRef getHelperName() const override { return HelperName; } 337 338 static bool classof(const CGCapturedStmtInfo *Info) { 339 return CGOpenMPRegionInfo::classof(Info) && 340 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion; 341 } 342 343 private: 344 StringRef HelperName; 345 }; 346 347 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) { 348 llvm_unreachable("No codegen for expressions"); 349 } 350 /// API for generation of expressions captured in a innermost OpenMP 351 /// region. 352 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo { 353 public: 354 CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS) 355 : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen, 356 OMPD_unknown, 357 /*HasCancel=*/false), 358 PrivScope(CGF) { 359 // Make sure the globals captured in the provided statement are local by 360 // using the privatization logic. We assume the same variable is not 361 // captured more than once. 362 for (const auto &C : CS.captures()) { 363 if (!C.capturesVariable() && !C.capturesVariableByCopy()) 364 continue; 365 366 const VarDecl *VD = C.getCapturedVar(); 367 if (VD->isLocalVarDeclOrParm()) 368 continue; 369 370 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD), 371 /*RefersToEnclosingVariableOrCapture=*/false, 372 VD->getType().getNonReferenceType(), VK_LValue, 373 C.getLocation()); 374 PrivScope.addPrivate(VD, CGF.EmitLValue(&DRE).getAddress(CGF)); 375 } 376 (void)PrivScope.Privatize(); 377 } 378 379 /// Lookup the captured field decl for a variable. 380 const FieldDecl *lookup(const VarDecl *VD) const override { 381 if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD)) 382 return FD; 383 return nullptr; 384 } 385 386 /// Emit the captured statement body. 387 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override { 388 llvm_unreachable("No body for expressions"); 389 } 390 391 /// Get a variable or parameter for storing global thread id 392 /// inside OpenMP construct. 393 const VarDecl *getThreadIDVariable() const override { 394 llvm_unreachable("No thread id for expressions"); 395 } 396 397 /// Get the name of the capture helper. 398 StringRef getHelperName() const override { 399 llvm_unreachable("No helper name for expressions"); 400 } 401 402 static bool classof(const CGCapturedStmtInfo *Info) { return false; } 403 404 private: 405 /// Private scope to capture global variables. 406 CodeGenFunction::OMPPrivateScope PrivScope; 407 }; 408 409 /// RAII for emitting code of OpenMP constructs. 410 class InlinedOpenMPRegionRAII { 411 CodeGenFunction &CGF; 412 llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields; 413 FieldDecl *LambdaThisCaptureField = nullptr; 414 const CodeGen::CGBlockInfo *BlockInfo = nullptr; 415 bool NoInheritance = false; 416 417 public: 418 /// Constructs region for combined constructs. 419 /// \param CodeGen Code generation sequence for combined directives. Includes 420 /// a list of functions used for code generation of implicitly inlined 421 /// regions. 422 InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen, 423 OpenMPDirectiveKind Kind, bool HasCancel, 424 bool NoInheritance = true) 425 : CGF(CGF), NoInheritance(NoInheritance) { 426 // Start emission for the construct. 427 CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo( 428 CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel); 429 if (NoInheritance) { 430 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 431 LambdaThisCaptureField = CGF.LambdaThisCaptureField; 432 CGF.LambdaThisCaptureField = nullptr; 433 BlockInfo = CGF.BlockInfo; 434 CGF.BlockInfo = nullptr; 435 } 436 } 437 438 ~InlinedOpenMPRegionRAII() { 439 // Restore original CapturedStmtInfo only if we're done with code emission. 440 auto *OldCSI = 441 cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI(); 442 delete CGF.CapturedStmtInfo; 443 CGF.CapturedStmtInfo = OldCSI; 444 if (NoInheritance) { 445 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 446 CGF.LambdaThisCaptureField = LambdaThisCaptureField; 447 CGF.BlockInfo = BlockInfo; 448 } 449 } 450 }; 451 452 /// Values for bit flags used in the ident_t to describe the fields. 453 /// All enumeric elements are named and described in accordance with the code 454 /// from https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h 455 enum OpenMPLocationFlags : unsigned { 456 /// Use trampoline for internal microtask. 457 OMP_IDENT_IMD = 0x01, 458 /// Use c-style ident structure. 459 OMP_IDENT_KMPC = 0x02, 460 /// Atomic reduction option for kmpc_reduce. 461 OMP_ATOMIC_REDUCE = 0x10, 462 /// Explicit 'barrier' directive. 463 OMP_IDENT_BARRIER_EXPL = 0x20, 464 /// Implicit barrier in code. 465 OMP_IDENT_BARRIER_IMPL = 0x40, 466 /// Implicit barrier in 'for' directive. 467 OMP_IDENT_BARRIER_IMPL_FOR = 0x40, 468 /// Implicit barrier in 'sections' directive. 469 OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0, 470 /// Implicit barrier in 'single' directive. 471 OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140, 472 /// Call of __kmp_for_static_init for static loop. 473 OMP_IDENT_WORK_LOOP = 0x200, 474 /// Call of __kmp_for_static_init for sections. 475 OMP_IDENT_WORK_SECTIONS = 0x400, 476 /// Call of __kmp_for_static_init for distribute. 477 OMP_IDENT_WORK_DISTRIBUTE = 0x800, 478 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE) 479 }; 480 481 namespace { 482 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 483 /// Values for bit flags for marking which requires clauses have been used. 484 enum OpenMPOffloadingRequiresDirFlags : int64_t { 485 /// flag undefined. 486 OMP_REQ_UNDEFINED = 0x000, 487 /// no requires clause present. 488 OMP_REQ_NONE = 0x001, 489 /// reverse_offload clause. 490 OMP_REQ_REVERSE_OFFLOAD = 0x002, 491 /// unified_address clause. 492 OMP_REQ_UNIFIED_ADDRESS = 0x004, 493 /// unified_shared_memory clause. 494 OMP_REQ_UNIFIED_SHARED_MEMORY = 0x008, 495 /// dynamic_allocators clause. 496 OMP_REQ_DYNAMIC_ALLOCATORS = 0x010, 497 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS) 498 }; 499 500 enum OpenMPOffloadingReservedDeviceIDs { 501 /// Device ID if the device was not defined, runtime should get it 502 /// from environment variables in the spec. 503 OMP_DEVICEID_UNDEF = -1, 504 }; 505 } // anonymous namespace 506 507 /// Describes ident structure that describes a source location. 508 /// All descriptions are taken from 509 /// https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h 510 /// Original structure: 511 /// typedef struct ident { 512 /// kmp_int32 reserved_1; /**< might be used in Fortran; 513 /// see above */ 514 /// kmp_int32 flags; /**< also f.flags; KMP_IDENT_xxx flags; 515 /// KMP_IDENT_KMPC identifies this union 516 /// member */ 517 /// kmp_int32 reserved_2; /**< not really used in Fortran any more; 518 /// see above */ 519 ///#if USE_ITT_BUILD 520 /// /* but currently used for storing 521 /// region-specific ITT */ 522 /// /* contextual information. */ 523 ///#endif /* USE_ITT_BUILD */ 524 /// kmp_int32 reserved_3; /**< source[4] in Fortran, do not use for 525 /// C++ */ 526 /// char const *psource; /**< String describing the source location. 527 /// The string is composed of semi-colon separated 528 // fields which describe the source file, 529 /// the function and a pair of line numbers that 530 /// delimit the construct. 531 /// */ 532 /// } ident_t; 533 enum IdentFieldIndex { 534 /// might be used in Fortran 535 IdentField_Reserved_1, 536 /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member. 537 IdentField_Flags, 538 /// Not really used in Fortran any more 539 IdentField_Reserved_2, 540 /// Source[4] in Fortran, do not use for C++ 541 IdentField_Reserved_3, 542 /// String describing the source location. The string is composed of 543 /// semi-colon separated fields which describe the source file, the function 544 /// and a pair of line numbers that delimit the construct. 545 IdentField_PSource 546 }; 547 548 /// Schedule types for 'omp for' loops (these enumerators are taken from 549 /// the enum sched_type in kmp.h). 550 enum OpenMPSchedType { 551 /// Lower bound for default (unordered) versions. 552 OMP_sch_lower = 32, 553 OMP_sch_static_chunked = 33, 554 OMP_sch_static = 34, 555 OMP_sch_dynamic_chunked = 35, 556 OMP_sch_guided_chunked = 36, 557 OMP_sch_runtime = 37, 558 OMP_sch_auto = 38, 559 /// static with chunk adjustment (e.g., simd) 560 OMP_sch_static_balanced_chunked = 45, 561 /// Lower bound for 'ordered' versions. 562 OMP_ord_lower = 64, 563 OMP_ord_static_chunked = 65, 564 OMP_ord_static = 66, 565 OMP_ord_dynamic_chunked = 67, 566 OMP_ord_guided_chunked = 68, 567 OMP_ord_runtime = 69, 568 OMP_ord_auto = 70, 569 OMP_sch_default = OMP_sch_static, 570 /// dist_schedule types 571 OMP_dist_sch_static_chunked = 91, 572 OMP_dist_sch_static = 92, 573 /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers. 574 /// Set if the monotonic schedule modifier was present. 575 OMP_sch_modifier_monotonic = (1 << 29), 576 /// Set if the nonmonotonic schedule modifier was present. 577 OMP_sch_modifier_nonmonotonic = (1 << 30), 578 }; 579 580 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP 581 /// region. 582 class CleanupTy final : public EHScopeStack::Cleanup { 583 PrePostActionTy *Action; 584 585 public: 586 explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {} 587 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 588 if (!CGF.HaveInsertPoint()) 589 return; 590 Action->Exit(CGF); 591 } 592 }; 593 594 } // anonymous namespace 595 596 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const { 597 CodeGenFunction::RunCleanupsScope Scope(CGF); 598 if (PrePostAction) { 599 CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction); 600 Callback(CodeGen, CGF, *PrePostAction); 601 } else { 602 PrePostActionTy Action; 603 Callback(CodeGen, CGF, Action); 604 } 605 } 606 607 /// Check if the combiner is a call to UDR combiner and if it is so return the 608 /// UDR decl used for reduction. 609 static const OMPDeclareReductionDecl * 610 getReductionInit(const Expr *ReductionOp) { 611 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 612 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 613 if (const auto *DRE = 614 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 615 if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) 616 return DRD; 617 return nullptr; 618 } 619 620 static void emitInitWithReductionInitializer(CodeGenFunction &CGF, 621 const OMPDeclareReductionDecl *DRD, 622 const Expr *InitOp, 623 Address Private, Address Original, 624 QualType Ty) { 625 if (DRD->getInitializer()) { 626 std::pair<llvm::Function *, llvm::Function *> Reduction = 627 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 628 const auto *CE = cast<CallExpr>(InitOp); 629 const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee()); 630 const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts(); 631 const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts(); 632 const auto *LHSDRE = 633 cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr()); 634 const auto *RHSDRE = 635 cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr()); 636 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 637 PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()), Private); 638 PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()), Original); 639 (void)PrivateScope.Privatize(); 640 RValue Func = RValue::get(Reduction.second); 641 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 642 CGF.EmitIgnoredExpr(InitOp); 643 } else { 644 llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty); 645 std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"}); 646 auto *GV = new llvm::GlobalVariable( 647 CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true, 648 llvm::GlobalValue::PrivateLinkage, Init, Name); 649 LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty); 650 RValue InitRVal; 651 switch (CGF.getEvaluationKind(Ty)) { 652 case TEK_Scalar: 653 InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation()); 654 break; 655 case TEK_Complex: 656 InitRVal = 657 RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation())); 658 break; 659 case TEK_Aggregate: { 660 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_LValue); 661 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, LV); 662 CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(), 663 /*IsInitializer=*/false); 664 return; 665 } 666 } 667 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_PRValue); 668 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal); 669 CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(), 670 /*IsInitializer=*/false); 671 } 672 } 673 674 /// Emit initialization of arrays of complex types. 675 /// \param DestAddr Address of the array. 676 /// \param Type Type of array. 677 /// \param Init Initial expression of array. 678 /// \param SrcAddr Address of the original array. 679 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr, 680 QualType Type, bool EmitDeclareReductionInit, 681 const Expr *Init, 682 const OMPDeclareReductionDecl *DRD, 683 Address SrcAddr = Address::invalid()) { 684 // Perform element-by-element initialization. 685 QualType ElementTy; 686 687 // Drill down to the base element type on both arrays. 688 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 689 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr); 690 if (DRD) 691 SrcAddr = 692 CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType()); 693 694 llvm::Value *SrcBegin = nullptr; 695 if (DRD) 696 SrcBegin = SrcAddr.getPointer(); 697 llvm::Value *DestBegin = DestAddr.getPointer(); 698 // Cast from pointer to array type to pointer to single element. 699 llvm::Value *DestEnd = 700 CGF.Builder.CreateGEP(DestAddr.getElementType(), DestBegin, NumElements); 701 // The basic structure here is a while-do loop. 702 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body"); 703 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done"); 704 llvm::Value *IsEmpty = 705 CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty"); 706 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 707 708 // Enter the loop body, making that address the current address. 709 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 710 CGF.EmitBlock(BodyBB); 711 712 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 713 714 llvm::PHINode *SrcElementPHI = nullptr; 715 Address SrcElementCurrent = Address::invalid(); 716 if (DRD) { 717 SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2, 718 "omp.arraycpy.srcElementPast"); 719 SrcElementPHI->addIncoming(SrcBegin, EntryBB); 720 SrcElementCurrent = 721 Address(SrcElementPHI, SrcAddr.getElementType(), 722 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 723 } 724 llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI( 725 DestBegin->getType(), 2, "omp.arraycpy.destElementPast"); 726 DestElementPHI->addIncoming(DestBegin, EntryBB); 727 Address DestElementCurrent = 728 Address(DestElementPHI, DestAddr.getElementType(), 729 DestAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 730 731 // Emit copy. 732 { 733 CodeGenFunction::RunCleanupsScope InitScope(CGF); 734 if (EmitDeclareReductionInit) { 735 emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent, 736 SrcElementCurrent, ElementTy); 737 } else 738 CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(), 739 /*IsInitializer=*/false); 740 } 741 742 if (DRD) { 743 // Shift the address forward by one element. 744 llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32( 745 SrcAddr.getElementType(), SrcElementPHI, /*Idx0=*/1, 746 "omp.arraycpy.dest.element"); 747 SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock()); 748 } 749 750 // Shift the address forward by one element. 751 llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32( 752 DestAddr.getElementType(), DestElementPHI, /*Idx0=*/1, 753 "omp.arraycpy.dest.element"); 754 // Check whether we've reached the end. 755 llvm::Value *Done = 756 CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done"); 757 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 758 DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock()); 759 760 // Done. 761 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 762 } 763 764 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) { 765 return CGF.EmitOMPSharedLValue(E); 766 } 767 768 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF, 769 const Expr *E) { 770 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E)) 771 return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false); 772 return LValue(); 773 } 774 775 void ReductionCodeGen::emitAggregateInitialization( 776 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, Address SharedAddr, 777 const OMPDeclareReductionDecl *DRD) { 778 // Emit VarDecl with copy init for arrays. 779 // Get the address of the original variable captured in current 780 // captured region. 781 const auto *PrivateVD = 782 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 783 bool EmitDeclareReductionInit = 784 DRD && (DRD->getInitializer() || !PrivateVD->hasInit()); 785 EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(), 786 EmitDeclareReductionInit, 787 EmitDeclareReductionInit ? ClausesData[N].ReductionOp 788 : PrivateVD->getInit(), 789 DRD, SharedAddr); 790 } 791 792 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds, 793 ArrayRef<const Expr *> Origs, 794 ArrayRef<const Expr *> Privates, 795 ArrayRef<const Expr *> ReductionOps) { 796 ClausesData.reserve(Shareds.size()); 797 SharedAddresses.reserve(Shareds.size()); 798 Sizes.reserve(Shareds.size()); 799 BaseDecls.reserve(Shareds.size()); 800 const auto *IOrig = Origs.begin(); 801 const auto *IPriv = Privates.begin(); 802 const auto *IRed = ReductionOps.begin(); 803 for (const Expr *Ref : Shareds) { 804 ClausesData.emplace_back(Ref, *IOrig, *IPriv, *IRed); 805 std::advance(IOrig, 1); 806 std::advance(IPriv, 1); 807 std::advance(IRed, 1); 808 } 809 } 810 811 void ReductionCodeGen::emitSharedOrigLValue(CodeGenFunction &CGF, unsigned N) { 812 assert(SharedAddresses.size() == N && OrigAddresses.size() == N && 813 "Number of generated lvalues must be exactly N."); 814 LValue First = emitSharedLValue(CGF, ClausesData[N].Shared); 815 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Shared); 816 SharedAddresses.emplace_back(First, Second); 817 if (ClausesData[N].Shared == ClausesData[N].Ref) { 818 OrigAddresses.emplace_back(First, Second); 819 } else { 820 LValue First = emitSharedLValue(CGF, ClausesData[N].Ref); 821 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref); 822 OrigAddresses.emplace_back(First, Second); 823 } 824 } 825 826 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) { 827 QualType PrivateType = getPrivateType(N); 828 bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref); 829 if (!PrivateType->isVariablyModifiedType()) { 830 Sizes.emplace_back( 831 CGF.getTypeSize(OrigAddresses[N].first.getType().getNonReferenceType()), 832 nullptr); 833 return; 834 } 835 llvm::Value *Size; 836 llvm::Value *SizeInChars; 837 auto *ElemType = OrigAddresses[N].first.getAddress(CGF).getElementType(); 838 auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType); 839 if (AsArraySection) { 840 Size = CGF.Builder.CreatePtrDiff(ElemType, 841 OrigAddresses[N].second.getPointer(CGF), 842 OrigAddresses[N].first.getPointer(CGF)); 843 Size = CGF.Builder.CreateNUWAdd( 844 Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1)); 845 SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf); 846 } else { 847 SizeInChars = 848 CGF.getTypeSize(OrigAddresses[N].first.getType().getNonReferenceType()); 849 Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf); 850 } 851 Sizes.emplace_back(SizeInChars, Size); 852 CodeGenFunction::OpaqueValueMapping OpaqueMap( 853 CGF, 854 cast<OpaqueValueExpr>( 855 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 856 RValue::get(Size)); 857 CGF.EmitVariablyModifiedType(PrivateType); 858 } 859 860 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N, 861 llvm::Value *Size) { 862 QualType PrivateType = getPrivateType(N); 863 if (!PrivateType->isVariablyModifiedType()) { 864 assert(!Size && !Sizes[N].second && 865 "Size should be nullptr for non-variably modified reduction " 866 "items."); 867 return; 868 } 869 CodeGenFunction::OpaqueValueMapping OpaqueMap( 870 CGF, 871 cast<OpaqueValueExpr>( 872 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 873 RValue::get(Size)); 874 CGF.EmitVariablyModifiedType(PrivateType); 875 } 876 877 void ReductionCodeGen::emitInitialization( 878 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, Address SharedAddr, 879 llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) { 880 assert(SharedAddresses.size() > N && "No variable was generated"); 881 const auto *PrivateVD = 882 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 883 const OMPDeclareReductionDecl *DRD = 884 getReductionInit(ClausesData[N].ReductionOp); 885 if (CGF.getContext().getAsArrayType(PrivateVD->getType())) { 886 if (DRD && DRD->getInitializer()) 887 (void)DefaultInit(CGF); 888 emitAggregateInitialization(CGF, N, PrivateAddr, SharedAddr, DRD); 889 } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) { 890 (void)DefaultInit(CGF); 891 QualType SharedType = SharedAddresses[N].first.getType(); 892 emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp, 893 PrivateAddr, SharedAddr, SharedType); 894 } else if (!DefaultInit(CGF) && PrivateVD->hasInit() && 895 !CGF.isTrivialInitializer(PrivateVD->getInit())) { 896 CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr, 897 PrivateVD->getType().getQualifiers(), 898 /*IsInitializer=*/false); 899 } 900 } 901 902 bool ReductionCodeGen::needCleanups(unsigned N) { 903 QualType PrivateType = getPrivateType(N); 904 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 905 return DTorKind != QualType::DK_none; 906 } 907 908 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N, 909 Address PrivateAddr) { 910 QualType PrivateType = getPrivateType(N); 911 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 912 if (needCleanups(N)) { 913 PrivateAddr = CGF.Builder.CreateElementBitCast( 914 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 915 CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType); 916 } 917 } 918 919 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 920 LValue BaseLV) { 921 BaseTy = BaseTy.getNonReferenceType(); 922 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 923 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 924 if (const auto *PtrTy = BaseTy->getAs<PointerType>()) { 925 BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(CGF), PtrTy); 926 } else { 927 LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(CGF), BaseTy); 928 BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal); 929 } 930 BaseTy = BaseTy->getPointeeType(); 931 } 932 return CGF.MakeAddrLValue( 933 CGF.Builder.CreateElementBitCast(BaseLV.getAddress(CGF), 934 CGF.ConvertTypeForMem(ElTy)), 935 BaseLV.getType(), BaseLV.getBaseInfo(), 936 CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType())); 937 } 938 939 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 940 llvm::Type *BaseLVType, CharUnits BaseLVAlignment, 941 llvm::Value *Addr) { 942 Address Tmp = Address::invalid(); 943 Address TopTmp = Address::invalid(); 944 Address MostTopTmp = Address::invalid(); 945 BaseTy = BaseTy.getNonReferenceType(); 946 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 947 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 948 Tmp = CGF.CreateMemTemp(BaseTy); 949 if (TopTmp.isValid()) 950 CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp); 951 else 952 MostTopTmp = Tmp; 953 TopTmp = Tmp; 954 BaseTy = BaseTy->getPointeeType(); 955 } 956 llvm::Type *Ty = BaseLVType; 957 if (Tmp.isValid()) 958 Ty = Tmp.getElementType(); 959 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty); 960 if (Tmp.isValid()) { 961 CGF.Builder.CreateStore(Addr, Tmp); 962 return MostTopTmp; 963 } 964 return Address::deprecated(Addr, BaseLVAlignment); 965 } 966 967 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) { 968 const VarDecl *OrigVD = nullptr; 969 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) { 970 const Expr *Base = OASE->getBase()->IgnoreParenImpCasts(); 971 while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base)) 972 Base = TempOASE->getBase()->IgnoreParenImpCasts(); 973 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 974 Base = TempASE->getBase()->IgnoreParenImpCasts(); 975 DE = cast<DeclRefExpr>(Base); 976 OrigVD = cast<VarDecl>(DE->getDecl()); 977 } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) { 978 const Expr *Base = ASE->getBase()->IgnoreParenImpCasts(); 979 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 980 Base = TempASE->getBase()->IgnoreParenImpCasts(); 981 DE = cast<DeclRefExpr>(Base); 982 OrigVD = cast<VarDecl>(DE->getDecl()); 983 } 984 return OrigVD; 985 } 986 987 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N, 988 Address PrivateAddr) { 989 const DeclRefExpr *DE; 990 if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) { 991 BaseDecls.emplace_back(OrigVD); 992 LValue OriginalBaseLValue = CGF.EmitLValue(DE); 993 LValue BaseLValue = 994 loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(), 995 OriginalBaseLValue); 996 Address SharedAddr = SharedAddresses[N].first.getAddress(CGF); 997 llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff( 998 SharedAddr.getElementType(), BaseLValue.getPointer(CGF), 999 SharedAddr.getPointer()); 1000 llvm::Value *PrivatePointer = 1001 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 1002 PrivateAddr.getPointer(), SharedAddr.getType()); 1003 llvm::Value *Ptr = CGF.Builder.CreateGEP( 1004 SharedAddr.getElementType(), PrivatePointer, Adjustment); 1005 return castToBase(CGF, OrigVD->getType(), 1006 SharedAddresses[N].first.getType(), 1007 OriginalBaseLValue.getAddress(CGF).getType(), 1008 OriginalBaseLValue.getAlignment(), Ptr); 1009 } 1010 BaseDecls.emplace_back( 1011 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl())); 1012 return PrivateAddr; 1013 } 1014 1015 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const { 1016 const OMPDeclareReductionDecl *DRD = 1017 getReductionInit(ClausesData[N].ReductionOp); 1018 return DRD && DRD->getInitializer(); 1019 } 1020 1021 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) { 1022 return CGF.EmitLoadOfPointerLValue( 1023 CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1024 getThreadIDVariable()->getType()->castAs<PointerType>()); 1025 } 1026 1027 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt *S) { 1028 if (!CGF.HaveInsertPoint()) 1029 return; 1030 // 1.2.2 OpenMP Language Terminology 1031 // Structured block - An executable statement with a single entry at the 1032 // top and a single exit at the bottom. 1033 // The point of exit cannot be a branch out of the structured block. 1034 // longjmp() and throw() must not violate the entry/exit criteria. 1035 CGF.EHStack.pushTerminate(); 1036 if (S) 1037 CGF.incrementProfileCounter(S); 1038 CodeGen(CGF); 1039 CGF.EHStack.popTerminate(); 1040 } 1041 1042 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue( 1043 CodeGenFunction &CGF) { 1044 return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1045 getThreadIDVariable()->getType(), 1046 AlignmentSource::Decl); 1047 } 1048 1049 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC, 1050 QualType FieldTy) { 1051 auto *Field = FieldDecl::Create( 1052 C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy, 1053 C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()), 1054 /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit); 1055 Field->setAccess(AS_public); 1056 DC->addDecl(Field); 1057 return Field; 1058 } 1059 1060 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator, 1061 StringRef Separator) 1062 : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator), 1063 OMPBuilder(CGM.getModule()), OffloadEntriesInfoManager(CGM) { 1064 KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8); 1065 1066 // Initialize Types used in OpenMPIRBuilder from OMPKinds.def 1067 OMPBuilder.initialize(); 1068 loadOffloadInfoMetadata(); 1069 } 1070 1071 void CGOpenMPRuntime::clear() { 1072 InternalVars.clear(); 1073 // Clean non-target variable declarations possibly used only in debug info. 1074 for (const auto &Data : EmittedNonTargetVariables) { 1075 if (!Data.getValue().pointsToAliveValue()) 1076 continue; 1077 auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue()); 1078 if (!GV) 1079 continue; 1080 if (!GV->isDeclaration() || GV->getNumUses() > 0) 1081 continue; 1082 GV->eraseFromParent(); 1083 } 1084 } 1085 1086 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const { 1087 SmallString<128> Buffer; 1088 llvm::raw_svector_ostream OS(Buffer); 1089 StringRef Sep = FirstSeparator; 1090 for (StringRef Part : Parts) { 1091 OS << Sep << Part; 1092 Sep = Separator; 1093 } 1094 return std::string(OS.str()); 1095 } 1096 1097 static llvm::Function * 1098 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty, 1099 const Expr *CombinerInitializer, const VarDecl *In, 1100 const VarDecl *Out, bool IsCombiner) { 1101 // void .omp_combiner.(Ty *in, Ty *out); 1102 ASTContext &C = CGM.getContext(); 1103 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 1104 FunctionArgList Args; 1105 ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(), 1106 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1107 ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(), 1108 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1109 Args.push_back(&OmpOutParm); 1110 Args.push_back(&OmpInParm); 1111 const CGFunctionInfo &FnInfo = 1112 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 1113 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 1114 std::string Name = CGM.getOpenMPRuntime().getName( 1115 {IsCombiner ? "omp_combiner" : "omp_initializer", ""}); 1116 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 1117 Name, &CGM.getModule()); 1118 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 1119 if (CGM.getLangOpts().Optimize) { 1120 Fn->removeFnAttr(llvm::Attribute::NoInline); 1121 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 1122 Fn->addFnAttr(llvm::Attribute::AlwaysInline); 1123 } 1124 CodeGenFunction CGF(CGM); 1125 // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions. 1126 // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions. 1127 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(), 1128 Out->getLocation()); 1129 CodeGenFunction::OMPPrivateScope Scope(CGF); 1130 Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm); 1131 Scope.addPrivate( 1132 In, CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>()) 1133 .getAddress(CGF)); 1134 Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm); 1135 Scope.addPrivate( 1136 Out, CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>()) 1137 .getAddress(CGF)); 1138 (void)Scope.Privatize(); 1139 if (!IsCombiner && Out->hasInit() && 1140 !CGF.isTrivialInitializer(Out->getInit())) { 1141 CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out), 1142 Out->getType().getQualifiers(), 1143 /*IsInitializer=*/true); 1144 } 1145 if (CombinerInitializer) 1146 CGF.EmitIgnoredExpr(CombinerInitializer); 1147 Scope.ForceCleanup(); 1148 CGF.FinishFunction(); 1149 return Fn; 1150 } 1151 1152 void CGOpenMPRuntime::emitUserDefinedReduction( 1153 CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) { 1154 if (UDRMap.count(D) > 0) 1155 return; 1156 llvm::Function *Combiner = emitCombinerOrInitializer( 1157 CGM, D->getType(), D->getCombiner(), 1158 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()), 1159 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()), 1160 /*IsCombiner=*/true); 1161 llvm::Function *Initializer = nullptr; 1162 if (const Expr *Init = D->getInitializer()) { 1163 Initializer = emitCombinerOrInitializer( 1164 CGM, D->getType(), 1165 D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init 1166 : nullptr, 1167 cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()), 1168 cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()), 1169 /*IsCombiner=*/false); 1170 } 1171 UDRMap.try_emplace(D, Combiner, Initializer); 1172 if (CGF) { 1173 auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn); 1174 Decls.second.push_back(D); 1175 } 1176 } 1177 1178 std::pair<llvm::Function *, llvm::Function *> 1179 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) { 1180 auto I = UDRMap.find(D); 1181 if (I != UDRMap.end()) 1182 return I->second; 1183 emitUserDefinedReduction(/*CGF=*/nullptr, D); 1184 return UDRMap.lookup(D); 1185 } 1186 1187 namespace { 1188 // Temporary RAII solution to perform a push/pop stack event on the OpenMP IR 1189 // Builder if one is present. 1190 struct PushAndPopStackRAII { 1191 PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF, 1192 bool HasCancel, llvm::omp::Directive Kind) 1193 : OMPBuilder(OMPBuilder) { 1194 if (!OMPBuilder) 1195 return; 1196 1197 // The following callback is the crucial part of clangs cleanup process. 1198 // 1199 // NOTE: 1200 // Once the OpenMPIRBuilder is used to create parallel regions (and 1201 // similar), the cancellation destination (Dest below) is determined via 1202 // IP. That means if we have variables to finalize we split the block at IP, 1203 // use the new block (=BB) as destination to build a JumpDest (via 1204 // getJumpDestInCurrentScope(BB)) which then is fed to 1205 // EmitBranchThroughCleanup. Furthermore, there will not be the need 1206 // to push & pop an FinalizationInfo object. 1207 // The FiniCB will still be needed but at the point where the 1208 // OpenMPIRBuilder is asked to construct a parallel (or similar) construct. 1209 auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) { 1210 assert(IP.getBlock()->end() == IP.getPoint() && 1211 "Clang CG should cause non-terminated block!"); 1212 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1213 CGF.Builder.restoreIP(IP); 1214 CodeGenFunction::JumpDest Dest = 1215 CGF.getOMPCancelDestination(OMPD_parallel); 1216 CGF.EmitBranchThroughCleanup(Dest); 1217 }; 1218 1219 // TODO: Remove this once we emit parallel regions through the 1220 // OpenMPIRBuilder as it can do this setup internally. 1221 llvm::OpenMPIRBuilder::FinalizationInfo FI({FiniCB, Kind, HasCancel}); 1222 OMPBuilder->pushFinalizationCB(std::move(FI)); 1223 } 1224 ~PushAndPopStackRAII() { 1225 if (OMPBuilder) 1226 OMPBuilder->popFinalizationCB(); 1227 } 1228 llvm::OpenMPIRBuilder *OMPBuilder; 1229 }; 1230 } // namespace 1231 1232 static llvm::Function *emitParallelOrTeamsOutlinedFunction( 1233 CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS, 1234 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, 1235 const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) { 1236 assert(ThreadIDVar->getType()->isPointerType() && 1237 "thread id variable must be of type kmp_int32 *"); 1238 CodeGenFunction CGF(CGM, true); 1239 bool HasCancel = false; 1240 if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D)) 1241 HasCancel = OPD->hasCancel(); 1242 else if (const auto *OPD = dyn_cast<OMPTargetParallelDirective>(&D)) 1243 HasCancel = OPD->hasCancel(); 1244 else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D)) 1245 HasCancel = OPSD->hasCancel(); 1246 else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D)) 1247 HasCancel = OPFD->hasCancel(); 1248 else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D)) 1249 HasCancel = OPFD->hasCancel(); 1250 else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D)) 1251 HasCancel = OPFD->hasCancel(); 1252 else if (const auto *OPFD = 1253 dyn_cast<OMPTeamsDistributeParallelForDirective>(&D)) 1254 HasCancel = OPFD->hasCancel(); 1255 else if (const auto *OPFD = 1256 dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D)) 1257 HasCancel = OPFD->hasCancel(); 1258 1259 // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new 1260 // parallel region to make cancellation barriers work properly. 1261 llvm::OpenMPIRBuilder &OMPBuilder = CGM.getOpenMPRuntime().getOMPBuilder(); 1262 PushAndPopStackRAII PSR(&OMPBuilder, CGF, HasCancel, InnermostKind); 1263 CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind, 1264 HasCancel, OutlinedHelperName); 1265 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1266 return CGF.GenerateOpenMPCapturedStmtFunction(*CS, D.getBeginLoc()); 1267 } 1268 1269 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction( 1270 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1271 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1272 const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel); 1273 return emitParallelOrTeamsOutlinedFunction( 1274 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1275 } 1276 1277 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction( 1278 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1279 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1280 const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams); 1281 return emitParallelOrTeamsOutlinedFunction( 1282 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1283 } 1284 1285 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction( 1286 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1287 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 1288 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 1289 bool Tied, unsigned &NumberOfParts) { 1290 auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF, 1291 PrePostActionTy &) { 1292 llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc()); 1293 llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 1294 llvm::Value *TaskArgs[] = { 1295 UpLoc, ThreadID, 1296 CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar), 1297 TaskTVar->getType()->castAs<PointerType>()) 1298 .getPointer(CGF)}; 1299 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 1300 CGM.getModule(), OMPRTL___kmpc_omp_task), 1301 TaskArgs); 1302 }; 1303 CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar, 1304 UntiedCodeGen); 1305 CodeGen.setAction(Action); 1306 assert(!ThreadIDVar->getType()->isPointerType() && 1307 "thread id variable must be of type kmp_int32 for tasks"); 1308 const OpenMPDirectiveKind Region = 1309 isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop 1310 : OMPD_task; 1311 const CapturedStmt *CS = D.getCapturedStmt(Region); 1312 bool HasCancel = false; 1313 if (const auto *TD = dyn_cast<OMPTaskDirective>(&D)) 1314 HasCancel = TD->hasCancel(); 1315 else if (const auto *TD = dyn_cast<OMPTaskLoopDirective>(&D)) 1316 HasCancel = TD->hasCancel(); 1317 else if (const auto *TD = dyn_cast<OMPMasterTaskLoopDirective>(&D)) 1318 HasCancel = TD->hasCancel(); 1319 else if (const auto *TD = dyn_cast<OMPParallelMasterTaskLoopDirective>(&D)) 1320 HasCancel = TD->hasCancel(); 1321 1322 CodeGenFunction CGF(CGM, true); 1323 CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, 1324 InnermostKind, HasCancel, Action); 1325 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1326 llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS); 1327 if (!Tied) 1328 NumberOfParts = Action.getNumberOfParts(); 1329 return Res; 1330 } 1331 1332 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM, 1333 const RecordDecl *RD, const CGRecordLayout &RL, 1334 ArrayRef<llvm::Constant *> Data) { 1335 llvm::StructType *StructTy = RL.getLLVMType(); 1336 unsigned PrevIdx = 0; 1337 ConstantInitBuilder CIBuilder(CGM); 1338 const auto *DI = Data.begin(); 1339 for (const FieldDecl *FD : RD->fields()) { 1340 unsigned Idx = RL.getLLVMFieldNo(FD); 1341 // Fill the alignment. 1342 for (unsigned I = PrevIdx; I < Idx; ++I) 1343 Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I))); 1344 PrevIdx = Idx + 1; 1345 Fields.add(*DI); 1346 ++DI; 1347 } 1348 } 1349 1350 template <class... As> 1351 static llvm::GlobalVariable * 1352 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant, 1353 ArrayRef<llvm::Constant *> Data, const Twine &Name, 1354 As &&... Args) { 1355 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1356 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1357 ConstantInitBuilder CIBuilder(CGM); 1358 ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType()); 1359 buildStructValue(Fields, CGM, RD, RL, Data); 1360 return Fields.finishAndCreateGlobal( 1361 Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant, 1362 std::forward<As>(Args)...); 1363 } 1364 1365 template <typename T> 1366 static void 1367 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty, 1368 ArrayRef<llvm::Constant *> Data, 1369 T &Parent) { 1370 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1371 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1372 ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType()); 1373 buildStructValue(Fields, CGM, RD, RL, Data); 1374 Fields.finishAndAddTo(Parent); 1375 } 1376 1377 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF, 1378 bool AtCurrentPoint) { 1379 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1380 assert(!Elem.second.ServiceInsertPt && "Insert point is set already."); 1381 1382 llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty); 1383 if (AtCurrentPoint) { 1384 Elem.second.ServiceInsertPt = new llvm::BitCastInst( 1385 Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock()); 1386 } else { 1387 Elem.second.ServiceInsertPt = 1388 new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt"); 1389 Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt); 1390 } 1391 } 1392 1393 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) { 1394 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1395 if (Elem.second.ServiceInsertPt) { 1396 llvm::Instruction *Ptr = Elem.second.ServiceInsertPt; 1397 Elem.second.ServiceInsertPt = nullptr; 1398 Ptr->eraseFromParent(); 1399 } 1400 } 1401 1402 static StringRef getIdentStringFromSourceLocation(CodeGenFunction &CGF, 1403 SourceLocation Loc, 1404 SmallString<128> &Buffer) { 1405 llvm::raw_svector_ostream OS(Buffer); 1406 // Build debug location 1407 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc); 1408 OS << ";" << PLoc.getFilename() << ";"; 1409 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) 1410 OS << FD->getQualifiedNameAsString(); 1411 OS << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;"; 1412 return OS.str(); 1413 } 1414 1415 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF, 1416 SourceLocation Loc, 1417 unsigned Flags) { 1418 uint32_t SrcLocStrSize; 1419 llvm::Constant *SrcLocStr; 1420 if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo || 1421 Loc.isInvalid()) { 1422 SrcLocStr = OMPBuilder.getOrCreateDefaultSrcLocStr(SrcLocStrSize); 1423 } else { 1424 std::string FunctionName; 1425 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) 1426 FunctionName = FD->getQualifiedNameAsString(); 1427 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc); 1428 const char *FileName = PLoc.getFilename(); 1429 unsigned Line = PLoc.getLine(); 1430 unsigned Column = PLoc.getColumn(); 1431 SrcLocStr = OMPBuilder.getOrCreateSrcLocStr(FunctionName, FileName, Line, 1432 Column, SrcLocStrSize); 1433 } 1434 unsigned Reserved2Flags = getDefaultLocationReserved2Flags(); 1435 return OMPBuilder.getOrCreateIdent( 1436 SrcLocStr, SrcLocStrSize, llvm::omp::IdentFlag(Flags), Reserved2Flags); 1437 } 1438 1439 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF, 1440 SourceLocation Loc) { 1441 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1442 // If the OpenMPIRBuilder is used we need to use it for all thread id calls as 1443 // the clang invariants used below might be broken. 1444 if (CGM.getLangOpts().OpenMPIRBuilder) { 1445 SmallString<128> Buffer; 1446 OMPBuilder.updateToLocation(CGF.Builder.saveIP()); 1447 uint32_t SrcLocStrSize; 1448 auto *SrcLocStr = OMPBuilder.getOrCreateSrcLocStr( 1449 getIdentStringFromSourceLocation(CGF, Loc, Buffer), SrcLocStrSize); 1450 return OMPBuilder.getOrCreateThreadID( 1451 OMPBuilder.getOrCreateIdent(SrcLocStr, SrcLocStrSize)); 1452 } 1453 1454 llvm::Value *ThreadID = nullptr; 1455 // Check whether we've already cached a load of the thread id in this 1456 // function. 1457 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1458 if (I != OpenMPLocThreadIDMap.end()) { 1459 ThreadID = I->second.ThreadID; 1460 if (ThreadID != nullptr) 1461 return ThreadID; 1462 } 1463 // If exceptions are enabled, do not use parameter to avoid possible crash. 1464 if (auto *OMPRegionInfo = 1465 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 1466 if (OMPRegionInfo->getThreadIDVariable()) { 1467 // Check if this an outlined function with thread id passed as argument. 1468 LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF); 1469 llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent(); 1470 if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions || 1471 !CGF.getLangOpts().CXXExceptions || 1472 CGF.Builder.GetInsertBlock() == TopBlock || 1473 !isa<llvm::Instruction>(LVal.getPointer(CGF)) || 1474 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() == 1475 TopBlock || 1476 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() == 1477 CGF.Builder.GetInsertBlock()) { 1478 ThreadID = CGF.EmitLoadOfScalar(LVal, Loc); 1479 // If value loaded in entry block, cache it and use it everywhere in 1480 // function. 1481 if (CGF.Builder.GetInsertBlock() == TopBlock) { 1482 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1483 Elem.second.ThreadID = ThreadID; 1484 } 1485 return ThreadID; 1486 } 1487 } 1488 } 1489 1490 // This is not an outlined function region - need to call __kmpc_int32 1491 // kmpc_global_thread_num(ident_t *loc). 1492 // Generate thread id value and cache this value for use across the 1493 // function. 1494 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1495 if (!Elem.second.ServiceInsertPt) 1496 setLocThreadIdInsertPt(CGF); 1497 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1498 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1499 llvm::CallInst *Call = CGF.Builder.CreateCall( 1500 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 1501 OMPRTL___kmpc_global_thread_num), 1502 emitUpdateLocation(CGF, Loc)); 1503 Call->setCallingConv(CGF.getRuntimeCC()); 1504 Elem.second.ThreadID = Call; 1505 return Call; 1506 } 1507 1508 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) { 1509 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1510 if (OpenMPLocThreadIDMap.count(CGF.CurFn)) { 1511 clearLocThreadIdInsertPt(CGF); 1512 OpenMPLocThreadIDMap.erase(CGF.CurFn); 1513 } 1514 if (FunctionUDRMap.count(CGF.CurFn) > 0) { 1515 for(const auto *D : FunctionUDRMap[CGF.CurFn]) 1516 UDRMap.erase(D); 1517 FunctionUDRMap.erase(CGF.CurFn); 1518 } 1519 auto I = FunctionUDMMap.find(CGF.CurFn); 1520 if (I != FunctionUDMMap.end()) { 1521 for(const auto *D : I->second) 1522 UDMMap.erase(D); 1523 FunctionUDMMap.erase(I); 1524 } 1525 LastprivateConditionalToTypes.erase(CGF.CurFn); 1526 FunctionToUntiedTaskStackMap.erase(CGF.CurFn); 1527 } 1528 1529 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() { 1530 return OMPBuilder.IdentPtr; 1531 } 1532 1533 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() { 1534 if (!Kmpc_MicroTy) { 1535 // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...) 1536 llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty), 1537 llvm::PointerType::getUnqual(CGM.Int32Ty)}; 1538 Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true); 1539 } 1540 return llvm::PointerType::getUnqual(Kmpc_MicroTy); 1541 } 1542 1543 llvm::FunctionCallee 1544 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned, 1545 bool IsGPUDistribute) { 1546 assert((IVSize == 32 || IVSize == 64) && 1547 "IV size is not compatible with the omp runtime"); 1548 StringRef Name; 1549 if (IsGPUDistribute) 1550 Name = IVSize == 32 ? (IVSigned ? "__kmpc_distribute_static_init_4" 1551 : "__kmpc_distribute_static_init_4u") 1552 : (IVSigned ? "__kmpc_distribute_static_init_8" 1553 : "__kmpc_distribute_static_init_8u"); 1554 else 1555 Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4" 1556 : "__kmpc_for_static_init_4u") 1557 : (IVSigned ? "__kmpc_for_static_init_8" 1558 : "__kmpc_for_static_init_8u"); 1559 1560 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 1561 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 1562 llvm::Type *TypeParams[] = { 1563 getIdentTyPointerTy(), // loc 1564 CGM.Int32Ty, // tid 1565 CGM.Int32Ty, // schedtype 1566 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 1567 PtrTy, // p_lower 1568 PtrTy, // p_upper 1569 PtrTy, // p_stride 1570 ITy, // incr 1571 ITy // chunk 1572 }; 1573 auto *FnTy = 1574 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1575 return CGM.CreateRuntimeFunction(FnTy, Name); 1576 } 1577 1578 llvm::FunctionCallee 1579 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) { 1580 assert((IVSize == 32 || IVSize == 64) && 1581 "IV size is not compatible with the omp runtime"); 1582 StringRef Name = 1583 IVSize == 32 1584 ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u") 1585 : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u"); 1586 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 1587 llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc 1588 CGM.Int32Ty, // tid 1589 CGM.Int32Ty, // schedtype 1590 ITy, // lower 1591 ITy, // upper 1592 ITy, // stride 1593 ITy // chunk 1594 }; 1595 auto *FnTy = 1596 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1597 return CGM.CreateRuntimeFunction(FnTy, Name); 1598 } 1599 1600 llvm::FunctionCallee 1601 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) { 1602 assert((IVSize == 32 || IVSize == 64) && 1603 "IV size is not compatible with the omp runtime"); 1604 StringRef Name = 1605 IVSize == 32 1606 ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u") 1607 : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u"); 1608 llvm::Type *TypeParams[] = { 1609 getIdentTyPointerTy(), // loc 1610 CGM.Int32Ty, // tid 1611 }; 1612 auto *FnTy = 1613 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1614 return CGM.CreateRuntimeFunction(FnTy, Name); 1615 } 1616 1617 llvm::FunctionCallee 1618 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) { 1619 assert((IVSize == 32 || IVSize == 64) && 1620 "IV size is not compatible with the omp runtime"); 1621 StringRef Name = 1622 IVSize == 32 1623 ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u") 1624 : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u"); 1625 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 1626 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 1627 llvm::Type *TypeParams[] = { 1628 getIdentTyPointerTy(), // loc 1629 CGM.Int32Ty, // tid 1630 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 1631 PtrTy, // p_lower 1632 PtrTy, // p_upper 1633 PtrTy // p_stride 1634 }; 1635 auto *FnTy = 1636 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1637 return CGM.CreateRuntimeFunction(FnTy, Name); 1638 } 1639 1640 /// Obtain information that uniquely identifies a target entry. This 1641 /// consists of the file and device IDs as well as line number associated with 1642 /// the relevant entry source location. 1643 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc, 1644 unsigned &DeviceID, unsigned &FileID, 1645 unsigned &LineNum) { 1646 SourceManager &SM = C.getSourceManager(); 1647 1648 // The loc should be always valid and have a file ID (the user cannot use 1649 // #pragma directives in macros) 1650 1651 assert(Loc.isValid() && "Source location is expected to be always valid."); 1652 1653 PresumedLoc PLoc = SM.getPresumedLoc(Loc); 1654 assert(PLoc.isValid() && "Source location is expected to be always valid."); 1655 1656 llvm::sys::fs::UniqueID ID; 1657 if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID)) { 1658 PLoc = SM.getPresumedLoc(Loc, /*UseLineDirectives=*/false); 1659 assert(PLoc.isValid() && "Source location is expected to be always valid."); 1660 if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID)) 1661 SM.getDiagnostics().Report(diag::err_cannot_open_file) 1662 << PLoc.getFilename() << EC.message(); 1663 } 1664 1665 DeviceID = ID.getDevice(); 1666 FileID = ID.getFile(); 1667 LineNum = PLoc.getLine(); 1668 } 1669 1670 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) { 1671 if (CGM.getLangOpts().OpenMPSimd) 1672 return Address::invalid(); 1673 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 1674 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 1675 if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link || 1676 (*Res == OMPDeclareTargetDeclAttr::MT_To && 1677 HasRequiresUnifiedSharedMemory))) { 1678 SmallString<64> PtrName; 1679 { 1680 llvm::raw_svector_ostream OS(PtrName); 1681 OS << CGM.getMangledName(GlobalDecl(VD)); 1682 if (!VD->isExternallyVisible()) { 1683 unsigned DeviceID, FileID, Line; 1684 getTargetEntryUniqueInfo(CGM.getContext(), 1685 VD->getCanonicalDecl()->getBeginLoc(), 1686 DeviceID, FileID, Line); 1687 OS << llvm::format("_%x", FileID); 1688 } 1689 OS << "_decl_tgt_ref_ptr"; 1690 } 1691 llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName); 1692 QualType PtrTy = CGM.getContext().getPointerType(VD->getType()); 1693 llvm::Type *LlvmPtrTy = CGM.getTypes().ConvertTypeForMem(PtrTy); 1694 if (!Ptr) { 1695 Ptr = getOrCreateInternalVariable(LlvmPtrTy, PtrName); 1696 1697 auto *GV = cast<llvm::GlobalVariable>(Ptr); 1698 GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 1699 1700 if (!CGM.getLangOpts().OpenMPIsDevice) 1701 GV->setInitializer(CGM.GetAddrOfGlobal(VD)); 1702 registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr)); 1703 } 1704 return Address(Ptr, LlvmPtrTy, CGM.getContext().getDeclAlign(VD)); 1705 } 1706 return Address::invalid(); 1707 } 1708 1709 llvm::Constant * 1710 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) { 1711 assert(!CGM.getLangOpts().OpenMPUseTLS || 1712 !CGM.getContext().getTargetInfo().isTLSSupported()); 1713 // Lookup the entry, lazily creating it if necessary. 1714 std::string Suffix = getName({"cache", ""}); 1715 return getOrCreateInternalVariable( 1716 CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix)); 1717 } 1718 1719 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 1720 const VarDecl *VD, 1721 Address VDAddr, 1722 SourceLocation Loc) { 1723 if (CGM.getLangOpts().OpenMPUseTLS && 1724 CGM.getContext().getTargetInfo().isTLSSupported()) 1725 return VDAddr; 1726 1727 llvm::Type *VarTy = VDAddr.getElementType(); 1728 llvm::Value *Args[] = { 1729 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 1730 CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.Int8PtrTy), 1731 CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)), 1732 getOrCreateThreadPrivateCache(VD)}; 1733 return Address( 1734 CGF.EmitRuntimeCall( 1735 OMPBuilder.getOrCreateRuntimeFunction( 1736 CGM.getModule(), OMPRTL___kmpc_threadprivate_cached), 1737 Args), 1738 CGF.Int8Ty, VDAddr.getAlignment()); 1739 } 1740 1741 void CGOpenMPRuntime::emitThreadPrivateVarInit( 1742 CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor, 1743 llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) { 1744 // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime 1745 // library. 1746 llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc); 1747 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 1748 CGM.getModule(), OMPRTL___kmpc_global_thread_num), 1749 OMPLoc); 1750 // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor) 1751 // to register constructor/destructor for variable. 1752 llvm::Value *Args[] = { 1753 OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy), 1754 Ctor, CopyCtor, Dtor}; 1755 CGF.EmitRuntimeCall( 1756 OMPBuilder.getOrCreateRuntimeFunction( 1757 CGM.getModule(), OMPRTL___kmpc_threadprivate_register), 1758 Args); 1759 } 1760 1761 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition( 1762 const VarDecl *VD, Address VDAddr, SourceLocation Loc, 1763 bool PerformInit, CodeGenFunction *CGF) { 1764 if (CGM.getLangOpts().OpenMPUseTLS && 1765 CGM.getContext().getTargetInfo().isTLSSupported()) 1766 return nullptr; 1767 1768 VD = VD->getDefinition(CGM.getContext()); 1769 if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) { 1770 QualType ASTTy = VD->getType(); 1771 1772 llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr; 1773 const Expr *Init = VD->getAnyInitializer(); 1774 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 1775 // Generate function that re-emits the declaration's initializer into the 1776 // threadprivate copy of the variable VD 1777 CodeGenFunction CtorCGF(CGM); 1778 FunctionArgList Args; 1779 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 1780 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 1781 ImplicitParamDecl::Other); 1782 Args.push_back(&Dst); 1783 1784 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 1785 CGM.getContext().VoidPtrTy, Args); 1786 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 1787 std::string Name = getName({"__kmpc_global_ctor_", ""}); 1788 llvm::Function *Fn = 1789 CGM.CreateGlobalInitOrCleanUpFunction(FTy, Name, FI, Loc); 1790 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI, 1791 Args, Loc, Loc); 1792 llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar( 1793 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 1794 CGM.getContext().VoidPtrTy, Dst.getLocation()); 1795 Address Arg(ArgVal, CtorCGF.Int8Ty, VDAddr.getAlignment()); 1796 Arg = CtorCGF.Builder.CreateElementBitCast( 1797 Arg, CtorCGF.ConvertTypeForMem(ASTTy)); 1798 CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(), 1799 /*IsInitializer=*/true); 1800 ArgVal = CtorCGF.EmitLoadOfScalar( 1801 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 1802 CGM.getContext().VoidPtrTy, Dst.getLocation()); 1803 CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue); 1804 CtorCGF.FinishFunction(); 1805 Ctor = Fn; 1806 } 1807 if (VD->getType().isDestructedType() != QualType::DK_none) { 1808 // Generate function that emits destructor call for the threadprivate copy 1809 // of the variable VD 1810 CodeGenFunction DtorCGF(CGM); 1811 FunctionArgList Args; 1812 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 1813 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 1814 ImplicitParamDecl::Other); 1815 Args.push_back(&Dst); 1816 1817 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 1818 CGM.getContext().VoidTy, Args); 1819 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 1820 std::string Name = getName({"__kmpc_global_dtor_", ""}); 1821 llvm::Function *Fn = 1822 CGM.CreateGlobalInitOrCleanUpFunction(FTy, Name, FI, Loc); 1823 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 1824 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args, 1825 Loc, Loc); 1826 // Create a scope with an artificial location for the body of this function. 1827 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 1828 llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar( 1829 DtorCGF.GetAddrOfLocalVar(&Dst), 1830 /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation()); 1831 DtorCGF.emitDestroy( 1832 Address(ArgVal, DtorCGF.Int8Ty, VDAddr.getAlignment()), ASTTy, 1833 DtorCGF.getDestroyer(ASTTy.isDestructedType()), 1834 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 1835 DtorCGF.FinishFunction(); 1836 Dtor = Fn; 1837 } 1838 // Do not emit init function if it is not required. 1839 if (!Ctor && !Dtor) 1840 return nullptr; 1841 1842 llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 1843 auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs, 1844 /*isVarArg=*/false) 1845 ->getPointerTo(); 1846 // Copying constructor for the threadprivate variable. 1847 // Must be NULL - reserved by runtime, but currently it requires that this 1848 // parameter is always NULL. Otherwise it fires assertion. 1849 CopyCtor = llvm::Constant::getNullValue(CopyCtorTy); 1850 if (Ctor == nullptr) { 1851 auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 1852 /*isVarArg=*/false) 1853 ->getPointerTo(); 1854 Ctor = llvm::Constant::getNullValue(CtorTy); 1855 } 1856 if (Dtor == nullptr) { 1857 auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, 1858 /*isVarArg=*/false) 1859 ->getPointerTo(); 1860 Dtor = llvm::Constant::getNullValue(DtorTy); 1861 } 1862 if (!CGF) { 1863 auto *InitFunctionTy = 1864 llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false); 1865 std::string Name = getName({"__omp_threadprivate_init_", ""}); 1866 llvm::Function *InitFunction = CGM.CreateGlobalInitOrCleanUpFunction( 1867 InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction()); 1868 CodeGenFunction InitCGF(CGM); 1869 FunctionArgList ArgList; 1870 InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction, 1871 CGM.getTypes().arrangeNullaryFunction(), ArgList, 1872 Loc, Loc); 1873 emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 1874 InitCGF.FinishFunction(); 1875 return InitFunction; 1876 } 1877 emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 1878 } 1879 return nullptr; 1880 } 1881 1882 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD, 1883 llvm::GlobalVariable *Addr, 1884 bool PerformInit) { 1885 if (CGM.getLangOpts().OMPTargetTriples.empty() && 1886 !CGM.getLangOpts().OpenMPIsDevice) 1887 return false; 1888 Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 1889 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 1890 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 1891 (*Res == OMPDeclareTargetDeclAttr::MT_To && 1892 HasRequiresUnifiedSharedMemory)) 1893 return CGM.getLangOpts().OpenMPIsDevice; 1894 VD = VD->getDefinition(CGM.getContext()); 1895 assert(VD && "Unknown VarDecl"); 1896 1897 if (!DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second) 1898 return CGM.getLangOpts().OpenMPIsDevice; 1899 1900 QualType ASTTy = VD->getType(); 1901 SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc(); 1902 1903 // Produce the unique prefix to identify the new target regions. We use 1904 // the source location of the variable declaration which we know to not 1905 // conflict with any target region. 1906 unsigned DeviceID; 1907 unsigned FileID; 1908 unsigned Line; 1909 getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line); 1910 SmallString<128> Buffer, Out; 1911 { 1912 llvm::raw_svector_ostream OS(Buffer); 1913 OS << "__omp_offloading_" << llvm::format("_%x", DeviceID) 1914 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 1915 } 1916 1917 const Expr *Init = VD->getAnyInitializer(); 1918 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 1919 llvm::Constant *Ctor; 1920 llvm::Constant *ID; 1921 if (CGM.getLangOpts().OpenMPIsDevice) { 1922 // Generate function that re-emits the declaration's initializer into 1923 // the threadprivate copy of the variable VD 1924 CodeGenFunction CtorCGF(CGM); 1925 1926 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 1927 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 1928 llvm::Function *Fn = CGM.CreateGlobalInitOrCleanUpFunction( 1929 FTy, Twine(Buffer, "_ctor"), FI, Loc); 1930 auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF); 1931 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 1932 FunctionArgList(), Loc, Loc); 1933 auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF); 1934 llvm::Constant *AddrInAS0 = Addr; 1935 if (Addr->getAddressSpace() != 0) 1936 AddrInAS0 = llvm::ConstantExpr::getAddrSpaceCast( 1937 Addr, llvm::PointerType::getWithSamePointeeType( 1938 cast<llvm::PointerType>(Addr->getType()), 0)); 1939 CtorCGF.EmitAnyExprToMem(Init, 1940 Address(AddrInAS0, Addr->getValueType(), 1941 CGM.getContext().getDeclAlign(VD)), 1942 Init->getType().getQualifiers(), 1943 /*IsInitializer=*/true); 1944 CtorCGF.FinishFunction(); 1945 Ctor = Fn; 1946 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 1947 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor)); 1948 } else { 1949 Ctor = new llvm::GlobalVariable( 1950 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 1951 llvm::GlobalValue::PrivateLinkage, 1952 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor")); 1953 ID = Ctor; 1954 } 1955 1956 // Register the information for the entry associated with the constructor. 1957 Out.clear(); 1958 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 1959 DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor, 1960 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor); 1961 } 1962 if (VD->getType().isDestructedType() != QualType::DK_none) { 1963 llvm::Constant *Dtor; 1964 llvm::Constant *ID; 1965 if (CGM.getLangOpts().OpenMPIsDevice) { 1966 // Generate function that emits destructor call for the threadprivate 1967 // copy of the variable VD 1968 CodeGenFunction DtorCGF(CGM); 1969 1970 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 1971 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 1972 llvm::Function *Fn = CGM.CreateGlobalInitOrCleanUpFunction( 1973 FTy, Twine(Buffer, "_dtor"), FI, Loc); 1974 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 1975 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 1976 FunctionArgList(), Loc, Loc); 1977 // Create a scope with an artificial location for the body of this 1978 // function. 1979 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 1980 llvm::Constant *AddrInAS0 = Addr; 1981 if (Addr->getAddressSpace() != 0) 1982 AddrInAS0 = llvm::ConstantExpr::getAddrSpaceCast( 1983 Addr, llvm::PointerType::getWithSamePointeeType( 1984 cast<llvm::PointerType>(Addr->getType()), 0)); 1985 DtorCGF.emitDestroy(Address(AddrInAS0, Addr->getValueType(), 1986 CGM.getContext().getDeclAlign(VD)), 1987 ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()), 1988 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 1989 DtorCGF.FinishFunction(); 1990 Dtor = Fn; 1991 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 1992 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor)); 1993 } else { 1994 Dtor = new llvm::GlobalVariable( 1995 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 1996 llvm::GlobalValue::PrivateLinkage, 1997 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor")); 1998 ID = Dtor; 1999 } 2000 // Register the information for the entry associated with the destructor. 2001 Out.clear(); 2002 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 2003 DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor, 2004 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor); 2005 } 2006 return CGM.getLangOpts().OpenMPIsDevice; 2007 } 2008 2009 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF, 2010 QualType VarType, 2011 StringRef Name) { 2012 std::string Suffix = getName({"artificial", ""}); 2013 llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType); 2014 llvm::GlobalVariable *GAddr = 2015 getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix)); 2016 if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS && 2017 CGM.getTarget().isTLSSupported()) { 2018 GAddr->setThreadLocal(/*Val=*/true); 2019 return Address(GAddr, GAddr->getValueType(), 2020 CGM.getContext().getTypeAlignInChars(VarType)); 2021 } 2022 std::string CacheSuffix = getName({"cache", ""}); 2023 llvm::Value *Args[] = { 2024 emitUpdateLocation(CGF, SourceLocation()), 2025 getThreadID(CGF, SourceLocation()), 2026 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy), 2027 CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy, 2028 /*isSigned=*/false), 2029 getOrCreateInternalVariable( 2030 CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))}; 2031 return Address( 2032 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 2033 CGF.EmitRuntimeCall( 2034 OMPBuilder.getOrCreateRuntimeFunction( 2035 CGM.getModule(), OMPRTL___kmpc_threadprivate_cached), 2036 Args), 2037 VarLVType->getPointerTo(/*AddrSpace=*/0)), 2038 VarLVType, CGM.getContext().getTypeAlignInChars(VarType)); 2039 } 2040 2041 void CGOpenMPRuntime::emitIfClause(CodeGenFunction &CGF, const Expr *Cond, 2042 const RegionCodeGenTy &ThenGen, 2043 const RegionCodeGenTy &ElseGen) { 2044 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange()); 2045 2046 // If the condition constant folds and can be elided, try to avoid emitting 2047 // the condition and the dead arm of the if/else. 2048 bool CondConstant; 2049 if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) { 2050 if (CondConstant) 2051 ThenGen(CGF); 2052 else 2053 ElseGen(CGF); 2054 return; 2055 } 2056 2057 // Otherwise, the condition did not fold, or we couldn't elide it. Just 2058 // emit the conditional branch. 2059 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then"); 2060 llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else"); 2061 llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end"); 2062 CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0); 2063 2064 // Emit the 'then' code. 2065 CGF.EmitBlock(ThenBlock); 2066 ThenGen(CGF); 2067 CGF.EmitBranch(ContBlock); 2068 // Emit the 'else' code if present. 2069 // There is no need to emit line number for unconditional branch. 2070 (void)ApplyDebugLocation::CreateEmpty(CGF); 2071 CGF.EmitBlock(ElseBlock); 2072 ElseGen(CGF); 2073 // There is no need to emit line number for unconditional branch. 2074 (void)ApplyDebugLocation::CreateEmpty(CGF); 2075 CGF.EmitBranch(ContBlock); 2076 // Emit the continuation block for code after the if. 2077 CGF.EmitBlock(ContBlock, /*IsFinished=*/true); 2078 } 2079 2080 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc, 2081 llvm::Function *OutlinedFn, 2082 ArrayRef<llvm::Value *> CapturedVars, 2083 const Expr *IfCond, 2084 llvm::Value *NumThreads) { 2085 if (!CGF.HaveInsertPoint()) 2086 return; 2087 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 2088 auto &M = CGM.getModule(); 2089 auto &&ThenGen = [&M, OutlinedFn, CapturedVars, RTLoc, 2090 this](CodeGenFunction &CGF, PrePostActionTy &) { 2091 // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn); 2092 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 2093 llvm::Value *Args[] = { 2094 RTLoc, 2095 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 2096 CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())}; 2097 llvm::SmallVector<llvm::Value *, 16> RealArgs; 2098 RealArgs.append(std::begin(Args), std::end(Args)); 2099 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 2100 2101 llvm::FunctionCallee RTLFn = 2102 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_fork_call); 2103 CGF.EmitRuntimeCall(RTLFn, RealArgs); 2104 }; 2105 auto &&ElseGen = [&M, OutlinedFn, CapturedVars, RTLoc, Loc, 2106 this](CodeGenFunction &CGF, PrePostActionTy &) { 2107 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 2108 llvm::Value *ThreadID = RT.getThreadID(CGF, Loc); 2109 // Build calls: 2110 // __kmpc_serialized_parallel(&Loc, GTid); 2111 llvm::Value *Args[] = {RTLoc, ThreadID}; 2112 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2113 M, OMPRTL___kmpc_serialized_parallel), 2114 Args); 2115 2116 // OutlinedFn(>id, &zero_bound, CapturedStruct); 2117 Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc); 2118 Address ZeroAddrBound = 2119 CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty, 2120 /*Name=*/".bound.zero.addr"); 2121 CGF.Builder.CreateStore(CGF.Builder.getInt32(/*C*/ 0), ZeroAddrBound); 2122 llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs; 2123 // ThreadId for serialized parallels is 0. 2124 OutlinedFnArgs.push_back(ThreadIDAddr.getPointer()); 2125 OutlinedFnArgs.push_back(ZeroAddrBound.getPointer()); 2126 OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end()); 2127 2128 // Ensure we do not inline the function. This is trivially true for the ones 2129 // passed to __kmpc_fork_call but the ones called in serialized regions 2130 // could be inlined. This is not a perfect but it is closer to the invariant 2131 // we want, namely, every data environment starts with a new function. 2132 // TODO: We should pass the if condition to the runtime function and do the 2133 // handling there. Much cleaner code. 2134 OutlinedFn->removeFnAttr(llvm::Attribute::AlwaysInline); 2135 OutlinedFn->addFnAttr(llvm::Attribute::NoInline); 2136 RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs); 2137 2138 // __kmpc_end_serialized_parallel(&Loc, GTid); 2139 llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID}; 2140 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2141 M, OMPRTL___kmpc_end_serialized_parallel), 2142 EndArgs); 2143 }; 2144 if (IfCond) { 2145 emitIfClause(CGF, IfCond, ThenGen, ElseGen); 2146 } else { 2147 RegionCodeGenTy ThenRCG(ThenGen); 2148 ThenRCG(CGF); 2149 } 2150 } 2151 2152 // If we're inside an (outlined) parallel region, use the region info's 2153 // thread-ID variable (it is passed in a first argument of the outlined function 2154 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in 2155 // regular serial code region, get thread ID by calling kmp_int32 2156 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and 2157 // return the address of that temp. 2158 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF, 2159 SourceLocation Loc) { 2160 if (auto *OMPRegionInfo = 2161 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 2162 if (OMPRegionInfo->getThreadIDVariable()) 2163 return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(CGF); 2164 2165 llvm::Value *ThreadID = getThreadID(CGF, Loc); 2166 QualType Int32Ty = 2167 CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true); 2168 Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp."); 2169 CGF.EmitStoreOfScalar(ThreadID, 2170 CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty)); 2171 2172 return ThreadIDTemp; 2173 } 2174 2175 llvm::GlobalVariable *CGOpenMPRuntime::getOrCreateInternalVariable( 2176 llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) { 2177 SmallString<256> Buffer; 2178 llvm::raw_svector_ostream Out(Buffer); 2179 Out << Name; 2180 StringRef RuntimeName = Out.str(); 2181 auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first; 2182 if (Elem.second) { 2183 assert(Elem.second->getType()->isOpaqueOrPointeeTypeMatches(Ty) && 2184 "OMP internal variable has different type than requested"); 2185 return &*Elem.second; 2186 } 2187 2188 return Elem.second = new llvm::GlobalVariable( 2189 CGM.getModule(), Ty, /*IsConstant*/ false, 2190 llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty), 2191 Elem.first(), /*InsertBefore=*/nullptr, 2192 llvm::GlobalValue::NotThreadLocal, AddressSpace); 2193 } 2194 2195 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) { 2196 std::string Prefix = Twine("gomp_critical_user_", CriticalName).str(); 2197 std::string Name = getName({Prefix, "var"}); 2198 return getOrCreateInternalVariable(KmpCriticalNameTy, Name); 2199 } 2200 2201 namespace { 2202 /// Common pre(post)-action for different OpenMP constructs. 2203 class CommonActionTy final : public PrePostActionTy { 2204 llvm::FunctionCallee EnterCallee; 2205 ArrayRef<llvm::Value *> EnterArgs; 2206 llvm::FunctionCallee ExitCallee; 2207 ArrayRef<llvm::Value *> ExitArgs; 2208 bool Conditional; 2209 llvm::BasicBlock *ContBlock = nullptr; 2210 2211 public: 2212 CommonActionTy(llvm::FunctionCallee EnterCallee, 2213 ArrayRef<llvm::Value *> EnterArgs, 2214 llvm::FunctionCallee ExitCallee, 2215 ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false) 2216 : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee), 2217 ExitArgs(ExitArgs), Conditional(Conditional) {} 2218 void Enter(CodeGenFunction &CGF) override { 2219 llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs); 2220 if (Conditional) { 2221 llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes); 2222 auto *ThenBlock = CGF.createBasicBlock("omp_if.then"); 2223 ContBlock = CGF.createBasicBlock("omp_if.end"); 2224 // Generate the branch (If-stmt) 2225 CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock); 2226 CGF.EmitBlock(ThenBlock); 2227 } 2228 } 2229 void Done(CodeGenFunction &CGF) { 2230 // Emit the rest of blocks/branches 2231 CGF.EmitBranch(ContBlock); 2232 CGF.EmitBlock(ContBlock, true); 2233 } 2234 void Exit(CodeGenFunction &CGF) override { 2235 CGF.EmitRuntimeCall(ExitCallee, ExitArgs); 2236 } 2237 }; 2238 } // anonymous namespace 2239 2240 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF, 2241 StringRef CriticalName, 2242 const RegionCodeGenTy &CriticalOpGen, 2243 SourceLocation Loc, const Expr *Hint) { 2244 // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]); 2245 // CriticalOpGen(); 2246 // __kmpc_end_critical(ident_t *, gtid, Lock); 2247 // Prepare arguments and build a call to __kmpc_critical 2248 if (!CGF.HaveInsertPoint()) 2249 return; 2250 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2251 getCriticalRegionLock(CriticalName)}; 2252 llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args), 2253 std::end(Args)); 2254 if (Hint) { 2255 EnterArgs.push_back(CGF.Builder.CreateIntCast( 2256 CGF.EmitScalarExpr(Hint), CGM.Int32Ty, /*isSigned=*/false)); 2257 } 2258 CommonActionTy Action( 2259 OMPBuilder.getOrCreateRuntimeFunction( 2260 CGM.getModule(), 2261 Hint ? OMPRTL___kmpc_critical_with_hint : OMPRTL___kmpc_critical), 2262 EnterArgs, 2263 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 2264 OMPRTL___kmpc_end_critical), 2265 Args); 2266 CriticalOpGen.setAction(Action); 2267 emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen); 2268 } 2269 2270 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF, 2271 const RegionCodeGenTy &MasterOpGen, 2272 SourceLocation Loc) { 2273 if (!CGF.HaveInsertPoint()) 2274 return; 2275 // if(__kmpc_master(ident_t *, gtid)) { 2276 // MasterOpGen(); 2277 // __kmpc_end_master(ident_t *, gtid); 2278 // } 2279 // Prepare arguments and build a call to __kmpc_master 2280 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 2281 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 2282 CGM.getModule(), OMPRTL___kmpc_master), 2283 Args, 2284 OMPBuilder.getOrCreateRuntimeFunction( 2285 CGM.getModule(), OMPRTL___kmpc_end_master), 2286 Args, 2287 /*Conditional=*/true); 2288 MasterOpGen.setAction(Action); 2289 emitInlinedDirective(CGF, OMPD_master, MasterOpGen); 2290 Action.Done(CGF); 2291 } 2292 2293 void CGOpenMPRuntime::emitMaskedRegion(CodeGenFunction &CGF, 2294 const RegionCodeGenTy &MaskedOpGen, 2295 SourceLocation Loc, const Expr *Filter) { 2296 if (!CGF.HaveInsertPoint()) 2297 return; 2298 // if(__kmpc_masked(ident_t *, gtid, filter)) { 2299 // MaskedOpGen(); 2300 // __kmpc_end_masked(iden_t *, gtid); 2301 // } 2302 // Prepare arguments and build a call to __kmpc_masked 2303 llvm::Value *FilterVal = Filter 2304 ? CGF.EmitScalarExpr(Filter, CGF.Int32Ty) 2305 : llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/0); 2306 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2307 FilterVal}; 2308 llvm::Value *ArgsEnd[] = {emitUpdateLocation(CGF, Loc), 2309 getThreadID(CGF, Loc)}; 2310 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 2311 CGM.getModule(), OMPRTL___kmpc_masked), 2312 Args, 2313 OMPBuilder.getOrCreateRuntimeFunction( 2314 CGM.getModule(), OMPRTL___kmpc_end_masked), 2315 ArgsEnd, 2316 /*Conditional=*/true); 2317 MaskedOpGen.setAction(Action); 2318 emitInlinedDirective(CGF, OMPD_masked, MaskedOpGen); 2319 Action.Done(CGF); 2320 } 2321 2322 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 2323 SourceLocation Loc) { 2324 if (!CGF.HaveInsertPoint()) 2325 return; 2326 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) { 2327 OMPBuilder.createTaskyield(CGF.Builder); 2328 } else { 2329 // Build call __kmpc_omp_taskyield(loc, thread_id, 0); 2330 llvm::Value *Args[] = { 2331 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2332 llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)}; 2333 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2334 CGM.getModule(), OMPRTL___kmpc_omp_taskyield), 2335 Args); 2336 } 2337 2338 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 2339 Region->emitUntiedSwitch(CGF); 2340 } 2341 2342 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF, 2343 const RegionCodeGenTy &TaskgroupOpGen, 2344 SourceLocation Loc) { 2345 if (!CGF.HaveInsertPoint()) 2346 return; 2347 // __kmpc_taskgroup(ident_t *, gtid); 2348 // TaskgroupOpGen(); 2349 // __kmpc_end_taskgroup(ident_t *, gtid); 2350 // Prepare arguments and build a call to __kmpc_taskgroup 2351 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 2352 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 2353 CGM.getModule(), OMPRTL___kmpc_taskgroup), 2354 Args, 2355 OMPBuilder.getOrCreateRuntimeFunction( 2356 CGM.getModule(), OMPRTL___kmpc_end_taskgroup), 2357 Args); 2358 TaskgroupOpGen.setAction(Action); 2359 emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen); 2360 } 2361 2362 /// Given an array of pointers to variables, project the address of a 2363 /// given variable. 2364 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array, 2365 unsigned Index, const VarDecl *Var) { 2366 // Pull out the pointer to the variable. 2367 Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index); 2368 llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr); 2369 2370 llvm::Type *ElemTy = CGF.ConvertTypeForMem(Var->getType()); 2371 return Address( 2372 CGF.Builder.CreateBitCast( 2373 Ptr, ElemTy->getPointerTo(Ptr->getType()->getPointerAddressSpace())), 2374 ElemTy, CGF.getContext().getDeclAlign(Var)); 2375 } 2376 2377 static llvm::Value *emitCopyprivateCopyFunction( 2378 CodeGenModule &CGM, llvm::Type *ArgsElemType, 2379 ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs, 2380 ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps, 2381 SourceLocation Loc) { 2382 ASTContext &C = CGM.getContext(); 2383 // void copy_func(void *LHSArg, void *RHSArg); 2384 FunctionArgList Args; 2385 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 2386 ImplicitParamDecl::Other); 2387 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 2388 ImplicitParamDecl::Other); 2389 Args.push_back(&LHSArg); 2390 Args.push_back(&RHSArg); 2391 const auto &CGFI = 2392 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 2393 std::string Name = 2394 CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"}); 2395 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 2396 llvm::GlobalValue::InternalLinkage, Name, 2397 &CGM.getModule()); 2398 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 2399 Fn->setDoesNotRecurse(); 2400 CodeGenFunction CGF(CGM); 2401 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 2402 // Dest = (void*[n])(LHSArg); 2403 // Src = (void*[n])(RHSArg); 2404 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 2405 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 2406 ArgsElemType->getPointerTo()), 2407 ArgsElemType, CGF.getPointerAlign()); 2408 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 2409 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 2410 ArgsElemType->getPointerTo()), 2411 ArgsElemType, CGF.getPointerAlign()); 2412 // *(Type0*)Dst[0] = *(Type0*)Src[0]; 2413 // *(Type1*)Dst[1] = *(Type1*)Src[1]; 2414 // ... 2415 // *(Typen*)Dst[n] = *(Typen*)Src[n]; 2416 for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) { 2417 const auto *DestVar = 2418 cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl()); 2419 Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar); 2420 2421 const auto *SrcVar = 2422 cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl()); 2423 Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar); 2424 2425 const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl(); 2426 QualType Type = VD->getType(); 2427 CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]); 2428 } 2429 CGF.FinishFunction(); 2430 return Fn; 2431 } 2432 2433 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF, 2434 const RegionCodeGenTy &SingleOpGen, 2435 SourceLocation Loc, 2436 ArrayRef<const Expr *> CopyprivateVars, 2437 ArrayRef<const Expr *> SrcExprs, 2438 ArrayRef<const Expr *> DstExprs, 2439 ArrayRef<const Expr *> AssignmentOps) { 2440 if (!CGF.HaveInsertPoint()) 2441 return; 2442 assert(CopyprivateVars.size() == SrcExprs.size() && 2443 CopyprivateVars.size() == DstExprs.size() && 2444 CopyprivateVars.size() == AssignmentOps.size()); 2445 ASTContext &C = CGM.getContext(); 2446 // int32 did_it = 0; 2447 // if(__kmpc_single(ident_t *, gtid)) { 2448 // SingleOpGen(); 2449 // __kmpc_end_single(ident_t *, gtid); 2450 // did_it = 1; 2451 // } 2452 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 2453 // <copy_func>, did_it); 2454 2455 Address DidIt = Address::invalid(); 2456 if (!CopyprivateVars.empty()) { 2457 // int32 did_it = 0; 2458 QualType KmpInt32Ty = 2459 C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 2460 DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it"); 2461 CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt); 2462 } 2463 // Prepare arguments and build a call to __kmpc_single 2464 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 2465 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 2466 CGM.getModule(), OMPRTL___kmpc_single), 2467 Args, 2468 OMPBuilder.getOrCreateRuntimeFunction( 2469 CGM.getModule(), OMPRTL___kmpc_end_single), 2470 Args, 2471 /*Conditional=*/true); 2472 SingleOpGen.setAction(Action); 2473 emitInlinedDirective(CGF, OMPD_single, SingleOpGen); 2474 if (DidIt.isValid()) { 2475 // did_it = 1; 2476 CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt); 2477 } 2478 Action.Done(CGF); 2479 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 2480 // <copy_func>, did_it); 2481 if (DidIt.isValid()) { 2482 llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size()); 2483 QualType CopyprivateArrayTy = C.getConstantArrayType( 2484 C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 2485 /*IndexTypeQuals=*/0); 2486 // Create a list of all private variables for copyprivate. 2487 Address CopyprivateList = 2488 CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list"); 2489 for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) { 2490 Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I); 2491 CGF.Builder.CreateStore( 2492 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 2493 CGF.EmitLValue(CopyprivateVars[I]).getPointer(CGF), 2494 CGF.VoidPtrTy), 2495 Elem); 2496 } 2497 // Build function that copies private values from single region to all other 2498 // threads in the corresponding parallel region. 2499 llvm::Value *CpyFn = emitCopyprivateCopyFunction( 2500 CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy), CopyprivateVars, 2501 SrcExprs, DstExprs, AssignmentOps, Loc); 2502 llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy); 2503 Address CL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 2504 CopyprivateList, CGF.VoidPtrTy, CGF.Int8Ty); 2505 llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt); 2506 llvm::Value *Args[] = { 2507 emitUpdateLocation(CGF, Loc), // ident_t *<loc> 2508 getThreadID(CGF, Loc), // i32 <gtid> 2509 BufSize, // size_t <buf_size> 2510 CL.getPointer(), // void *<copyprivate list> 2511 CpyFn, // void (*) (void *, void *) <copy_func> 2512 DidItVal // i32 did_it 2513 }; 2514 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2515 CGM.getModule(), OMPRTL___kmpc_copyprivate), 2516 Args); 2517 } 2518 } 2519 2520 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF, 2521 const RegionCodeGenTy &OrderedOpGen, 2522 SourceLocation Loc, bool IsThreads) { 2523 if (!CGF.HaveInsertPoint()) 2524 return; 2525 // __kmpc_ordered(ident_t *, gtid); 2526 // OrderedOpGen(); 2527 // __kmpc_end_ordered(ident_t *, gtid); 2528 // Prepare arguments and build a call to __kmpc_ordered 2529 if (IsThreads) { 2530 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 2531 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 2532 CGM.getModule(), OMPRTL___kmpc_ordered), 2533 Args, 2534 OMPBuilder.getOrCreateRuntimeFunction( 2535 CGM.getModule(), OMPRTL___kmpc_end_ordered), 2536 Args); 2537 OrderedOpGen.setAction(Action); 2538 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 2539 return; 2540 } 2541 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 2542 } 2543 2544 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) { 2545 unsigned Flags; 2546 if (Kind == OMPD_for) 2547 Flags = OMP_IDENT_BARRIER_IMPL_FOR; 2548 else if (Kind == OMPD_sections) 2549 Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS; 2550 else if (Kind == OMPD_single) 2551 Flags = OMP_IDENT_BARRIER_IMPL_SINGLE; 2552 else if (Kind == OMPD_barrier) 2553 Flags = OMP_IDENT_BARRIER_EXPL; 2554 else 2555 Flags = OMP_IDENT_BARRIER_IMPL; 2556 return Flags; 2557 } 2558 2559 void CGOpenMPRuntime::getDefaultScheduleAndChunk( 2560 CodeGenFunction &CGF, const OMPLoopDirective &S, 2561 OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const { 2562 // Check if the loop directive is actually a doacross loop directive. In this 2563 // case choose static, 1 schedule. 2564 if (llvm::any_of( 2565 S.getClausesOfKind<OMPOrderedClause>(), 2566 [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) { 2567 ScheduleKind = OMPC_SCHEDULE_static; 2568 // Chunk size is 1 in this case. 2569 llvm::APInt ChunkSize(32, 1); 2570 ChunkExpr = IntegerLiteral::Create( 2571 CGF.getContext(), ChunkSize, 2572 CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0), 2573 SourceLocation()); 2574 } 2575 } 2576 2577 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc, 2578 OpenMPDirectiveKind Kind, bool EmitChecks, 2579 bool ForceSimpleCall) { 2580 // Check if we should use the OMPBuilder 2581 auto *OMPRegionInfo = 2582 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo); 2583 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) { 2584 CGF.Builder.restoreIP(OMPBuilder.createBarrier( 2585 CGF.Builder, Kind, ForceSimpleCall, EmitChecks)); 2586 return; 2587 } 2588 2589 if (!CGF.HaveInsertPoint()) 2590 return; 2591 // Build call __kmpc_cancel_barrier(loc, thread_id); 2592 // Build call __kmpc_barrier(loc, thread_id); 2593 unsigned Flags = getDefaultFlagsForBarriers(Kind); 2594 // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc, 2595 // thread_id); 2596 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags), 2597 getThreadID(CGF, Loc)}; 2598 if (OMPRegionInfo) { 2599 if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) { 2600 llvm::Value *Result = CGF.EmitRuntimeCall( 2601 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 2602 OMPRTL___kmpc_cancel_barrier), 2603 Args); 2604 if (EmitChecks) { 2605 // if (__kmpc_cancel_barrier()) { 2606 // exit from construct; 2607 // } 2608 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 2609 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 2610 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 2611 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 2612 CGF.EmitBlock(ExitBB); 2613 // exit from construct; 2614 CodeGenFunction::JumpDest CancelDestination = 2615 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 2616 CGF.EmitBranchThroughCleanup(CancelDestination); 2617 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 2618 } 2619 return; 2620 } 2621 } 2622 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2623 CGM.getModule(), OMPRTL___kmpc_barrier), 2624 Args); 2625 } 2626 2627 /// Map the OpenMP loop schedule to the runtime enumeration. 2628 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind, 2629 bool Chunked, bool Ordered) { 2630 switch (ScheduleKind) { 2631 case OMPC_SCHEDULE_static: 2632 return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked) 2633 : (Ordered ? OMP_ord_static : OMP_sch_static); 2634 case OMPC_SCHEDULE_dynamic: 2635 return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked; 2636 case OMPC_SCHEDULE_guided: 2637 return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked; 2638 case OMPC_SCHEDULE_runtime: 2639 return Ordered ? OMP_ord_runtime : OMP_sch_runtime; 2640 case OMPC_SCHEDULE_auto: 2641 return Ordered ? OMP_ord_auto : OMP_sch_auto; 2642 case OMPC_SCHEDULE_unknown: 2643 assert(!Chunked && "chunk was specified but schedule kind not known"); 2644 return Ordered ? OMP_ord_static : OMP_sch_static; 2645 } 2646 llvm_unreachable("Unexpected runtime schedule"); 2647 } 2648 2649 /// Map the OpenMP distribute schedule to the runtime enumeration. 2650 static OpenMPSchedType 2651 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) { 2652 // only static is allowed for dist_schedule 2653 return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static; 2654 } 2655 2656 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind, 2657 bool Chunked) const { 2658 OpenMPSchedType Schedule = 2659 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 2660 return Schedule == OMP_sch_static; 2661 } 2662 2663 bool CGOpenMPRuntime::isStaticNonchunked( 2664 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 2665 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 2666 return Schedule == OMP_dist_sch_static; 2667 } 2668 2669 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind, 2670 bool Chunked) const { 2671 OpenMPSchedType Schedule = 2672 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 2673 return Schedule == OMP_sch_static_chunked; 2674 } 2675 2676 bool CGOpenMPRuntime::isStaticChunked( 2677 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 2678 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 2679 return Schedule == OMP_dist_sch_static_chunked; 2680 } 2681 2682 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const { 2683 OpenMPSchedType Schedule = 2684 getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false); 2685 assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here"); 2686 return Schedule != OMP_sch_static; 2687 } 2688 2689 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule, 2690 OpenMPScheduleClauseModifier M1, 2691 OpenMPScheduleClauseModifier M2) { 2692 int Modifier = 0; 2693 switch (M1) { 2694 case OMPC_SCHEDULE_MODIFIER_monotonic: 2695 Modifier = OMP_sch_modifier_monotonic; 2696 break; 2697 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 2698 Modifier = OMP_sch_modifier_nonmonotonic; 2699 break; 2700 case OMPC_SCHEDULE_MODIFIER_simd: 2701 if (Schedule == OMP_sch_static_chunked) 2702 Schedule = OMP_sch_static_balanced_chunked; 2703 break; 2704 case OMPC_SCHEDULE_MODIFIER_last: 2705 case OMPC_SCHEDULE_MODIFIER_unknown: 2706 break; 2707 } 2708 switch (M2) { 2709 case OMPC_SCHEDULE_MODIFIER_monotonic: 2710 Modifier = OMP_sch_modifier_monotonic; 2711 break; 2712 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 2713 Modifier = OMP_sch_modifier_nonmonotonic; 2714 break; 2715 case OMPC_SCHEDULE_MODIFIER_simd: 2716 if (Schedule == OMP_sch_static_chunked) 2717 Schedule = OMP_sch_static_balanced_chunked; 2718 break; 2719 case OMPC_SCHEDULE_MODIFIER_last: 2720 case OMPC_SCHEDULE_MODIFIER_unknown: 2721 break; 2722 } 2723 // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription. 2724 // If the static schedule kind is specified or if the ordered clause is 2725 // specified, and if the nonmonotonic modifier is not specified, the effect is 2726 // as if the monotonic modifier is specified. Otherwise, unless the monotonic 2727 // modifier is specified, the effect is as if the nonmonotonic modifier is 2728 // specified. 2729 if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) { 2730 if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static || 2731 Schedule == OMP_sch_static_balanced_chunked || 2732 Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static || 2733 Schedule == OMP_dist_sch_static_chunked || 2734 Schedule == OMP_dist_sch_static)) 2735 Modifier = OMP_sch_modifier_nonmonotonic; 2736 } 2737 return Schedule | Modifier; 2738 } 2739 2740 void CGOpenMPRuntime::emitForDispatchInit( 2741 CodeGenFunction &CGF, SourceLocation Loc, 2742 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 2743 bool Ordered, const DispatchRTInput &DispatchValues) { 2744 if (!CGF.HaveInsertPoint()) 2745 return; 2746 OpenMPSchedType Schedule = getRuntimeSchedule( 2747 ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered); 2748 assert(Ordered || 2749 (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked && 2750 Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked && 2751 Schedule != OMP_sch_static_balanced_chunked)); 2752 // Call __kmpc_dispatch_init( 2753 // ident_t *loc, kmp_int32 tid, kmp_int32 schedule, 2754 // kmp_int[32|64] lower, kmp_int[32|64] upper, 2755 // kmp_int[32|64] stride, kmp_int[32|64] chunk); 2756 2757 // If the Chunk was not specified in the clause - use default value 1. 2758 llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk 2759 : CGF.Builder.getIntN(IVSize, 1); 2760 llvm::Value *Args[] = { 2761 emitUpdateLocation(CGF, Loc), 2762 getThreadID(CGF, Loc), 2763 CGF.Builder.getInt32(addMonoNonMonoModifier( 2764 CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type 2765 DispatchValues.LB, // Lower 2766 DispatchValues.UB, // Upper 2767 CGF.Builder.getIntN(IVSize, 1), // Stride 2768 Chunk // Chunk 2769 }; 2770 CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args); 2771 } 2772 2773 static void emitForStaticInitCall( 2774 CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId, 2775 llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule, 2776 OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2, 2777 const CGOpenMPRuntime::StaticRTInput &Values) { 2778 if (!CGF.HaveInsertPoint()) 2779 return; 2780 2781 assert(!Values.Ordered); 2782 assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked || 2783 Schedule == OMP_sch_static_balanced_chunked || 2784 Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked || 2785 Schedule == OMP_dist_sch_static || 2786 Schedule == OMP_dist_sch_static_chunked); 2787 2788 // Call __kmpc_for_static_init( 2789 // ident_t *loc, kmp_int32 tid, kmp_int32 schedtype, 2790 // kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower, 2791 // kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride, 2792 // kmp_int[32|64] incr, kmp_int[32|64] chunk); 2793 llvm::Value *Chunk = Values.Chunk; 2794 if (Chunk == nullptr) { 2795 assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static || 2796 Schedule == OMP_dist_sch_static) && 2797 "expected static non-chunked schedule"); 2798 // If the Chunk was not specified in the clause - use default value 1. 2799 Chunk = CGF.Builder.getIntN(Values.IVSize, 1); 2800 } else { 2801 assert((Schedule == OMP_sch_static_chunked || 2802 Schedule == OMP_sch_static_balanced_chunked || 2803 Schedule == OMP_ord_static_chunked || 2804 Schedule == OMP_dist_sch_static_chunked) && 2805 "expected static chunked schedule"); 2806 } 2807 llvm::Value *Args[] = { 2808 UpdateLocation, 2809 ThreadId, 2810 CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1, 2811 M2)), // Schedule type 2812 Values.IL.getPointer(), // &isLastIter 2813 Values.LB.getPointer(), // &LB 2814 Values.UB.getPointer(), // &UB 2815 Values.ST.getPointer(), // &Stride 2816 CGF.Builder.getIntN(Values.IVSize, 1), // Incr 2817 Chunk // Chunk 2818 }; 2819 CGF.EmitRuntimeCall(ForStaticInitFunction, Args); 2820 } 2821 2822 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF, 2823 SourceLocation Loc, 2824 OpenMPDirectiveKind DKind, 2825 const OpenMPScheduleTy &ScheduleKind, 2826 const StaticRTInput &Values) { 2827 OpenMPSchedType ScheduleNum = getRuntimeSchedule( 2828 ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered); 2829 assert(isOpenMPWorksharingDirective(DKind) && 2830 "Expected loop-based or sections-based directive."); 2831 llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc, 2832 isOpenMPLoopDirective(DKind) 2833 ? OMP_IDENT_WORK_LOOP 2834 : OMP_IDENT_WORK_SECTIONS); 2835 llvm::Value *ThreadId = getThreadID(CGF, Loc); 2836 llvm::FunctionCallee StaticInitFunction = 2837 createForStaticInitFunction(Values.IVSize, Values.IVSigned, false); 2838 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 2839 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 2840 ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values); 2841 } 2842 2843 void CGOpenMPRuntime::emitDistributeStaticInit( 2844 CodeGenFunction &CGF, SourceLocation Loc, 2845 OpenMPDistScheduleClauseKind SchedKind, 2846 const CGOpenMPRuntime::StaticRTInput &Values) { 2847 OpenMPSchedType ScheduleNum = 2848 getRuntimeSchedule(SchedKind, Values.Chunk != nullptr); 2849 llvm::Value *UpdatedLocation = 2850 emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE); 2851 llvm::Value *ThreadId = getThreadID(CGF, Loc); 2852 llvm::FunctionCallee StaticInitFunction; 2853 bool isGPUDistribute = 2854 CGM.getLangOpts().OpenMPIsDevice && 2855 (CGM.getTriple().isAMDGCN() || CGM.getTriple().isNVPTX()); 2856 StaticInitFunction = createForStaticInitFunction( 2857 Values.IVSize, Values.IVSigned, isGPUDistribute); 2858 2859 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 2860 ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown, 2861 OMPC_SCHEDULE_MODIFIER_unknown, Values); 2862 } 2863 2864 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF, 2865 SourceLocation Loc, 2866 OpenMPDirectiveKind DKind) { 2867 if (!CGF.HaveInsertPoint()) 2868 return; 2869 // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid); 2870 llvm::Value *Args[] = { 2871 emitUpdateLocation(CGF, Loc, 2872 isOpenMPDistributeDirective(DKind) 2873 ? OMP_IDENT_WORK_DISTRIBUTE 2874 : isOpenMPLoopDirective(DKind) 2875 ? OMP_IDENT_WORK_LOOP 2876 : OMP_IDENT_WORK_SECTIONS), 2877 getThreadID(CGF, Loc)}; 2878 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 2879 if (isOpenMPDistributeDirective(DKind) && CGM.getLangOpts().OpenMPIsDevice && 2880 (CGM.getTriple().isAMDGCN() || CGM.getTriple().isNVPTX())) 2881 CGF.EmitRuntimeCall( 2882 OMPBuilder.getOrCreateRuntimeFunction( 2883 CGM.getModule(), OMPRTL___kmpc_distribute_static_fini), 2884 Args); 2885 else 2886 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2887 CGM.getModule(), OMPRTL___kmpc_for_static_fini), 2888 Args); 2889 } 2890 2891 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 2892 SourceLocation Loc, 2893 unsigned IVSize, 2894 bool IVSigned) { 2895 if (!CGF.HaveInsertPoint()) 2896 return; 2897 // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid); 2898 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 2899 CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args); 2900 } 2901 2902 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF, 2903 SourceLocation Loc, unsigned IVSize, 2904 bool IVSigned, Address IL, 2905 Address LB, Address UB, 2906 Address ST) { 2907 // Call __kmpc_dispatch_next( 2908 // ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter, 2909 // kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper, 2910 // kmp_int[32|64] *p_stride); 2911 llvm::Value *Args[] = { 2912 emitUpdateLocation(CGF, Loc), 2913 getThreadID(CGF, Loc), 2914 IL.getPointer(), // &isLastIter 2915 LB.getPointer(), // &Lower 2916 UB.getPointer(), // &Upper 2917 ST.getPointer() // &Stride 2918 }; 2919 llvm::Value *Call = 2920 CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args); 2921 return CGF.EmitScalarConversion( 2922 Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1), 2923 CGF.getContext().BoolTy, Loc); 2924 } 2925 2926 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 2927 llvm::Value *NumThreads, 2928 SourceLocation Loc) { 2929 if (!CGF.HaveInsertPoint()) 2930 return; 2931 // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads) 2932 llvm::Value *Args[] = { 2933 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2934 CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)}; 2935 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2936 CGM.getModule(), OMPRTL___kmpc_push_num_threads), 2937 Args); 2938 } 2939 2940 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF, 2941 ProcBindKind ProcBind, 2942 SourceLocation Loc) { 2943 if (!CGF.HaveInsertPoint()) 2944 return; 2945 assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value."); 2946 // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind) 2947 llvm::Value *Args[] = { 2948 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2949 llvm::ConstantInt::get(CGM.IntTy, unsigned(ProcBind), /*isSigned=*/true)}; 2950 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2951 CGM.getModule(), OMPRTL___kmpc_push_proc_bind), 2952 Args); 2953 } 2954 2955 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>, 2956 SourceLocation Loc, llvm::AtomicOrdering AO) { 2957 if (CGF.CGM.getLangOpts().OpenMPIRBuilder) { 2958 OMPBuilder.createFlush(CGF.Builder); 2959 } else { 2960 if (!CGF.HaveInsertPoint()) 2961 return; 2962 // Build call void __kmpc_flush(ident_t *loc) 2963 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 2964 CGM.getModule(), OMPRTL___kmpc_flush), 2965 emitUpdateLocation(CGF, Loc)); 2966 } 2967 } 2968 2969 namespace { 2970 /// Indexes of fields for type kmp_task_t. 2971 enum KmpTaskTFields { 2972 /// List of shared variables. 2973 KmpTaskTShareds, 2974 /// Task routine. 2975 KmpTaskTRoutine, 2976 /// Partition id for the untied tasks. 2977 KmpTaskTPartId, 2978 /// Function with call of destructors for private variables. 2979 Data1, 2980 /// Task priority. 2981 Data2, 2982 /// (Taskloops only) Lower bound. 2983 KmpTaskTLowerBound, 2984 /// (Taskloops only) Upper bound. 2985 KmpTaskTUpperBound, 2986 /// (Taskloops only) Stride. 2987 KmpTaskTStride, 2988 /// (Taskloops only) Is last iteration flag. 2989 KmpTaskTLastIter, 2990 /// (Taskloops only) Reduction data. 2991 KmpTaskTReductions, 2992 }; 2993 } // anonymous namespace 2994 2995 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const { 2996 return OffloadEntriesTargetRegion.empty() && 2997 OffloadEntriesDeviceGlobalVar.empty(); 2998 } 2999 3000 /// Initialize target region entry. 3001 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3002 initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3003 StringRef ParentName, unsigned LineNum, 3004 unsigned Order) { 3005 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3006 "only required for the device " 3007 "code generation."); 3008 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = 3009 OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr, 3010 OMPTargetRegionEntryTargetRegion); 3011 ++OffloadingEntriesNum; 3012 } 3013 3014 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3015 registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3016 StringRef ParentName, unsigned LineNum, 3017 llvm::Constant *Addr, llvm::Constant *ID, 3018 OMPTargetRegionEntryKind Flags) { 3019 // If we are emitting code for a target, the entry is already initialized, 3020 // only has to be registered. 3021 if (CGM.getLangOpts().OpenMPIsDevice) { 3022 // This could happen if the device compilation is invoked standalone. 3023 if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) 3024 return; 3025 auto &Entry = 3026 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum]; 3027 Entry.setAddress(Addr); 3028 Entry.setID(ID); 3029 Entry.setFlags(Flags); 3030 } else { 3031 if (Flags == 3032 OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion && 3033 hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum, 3034 /*IgnoreAddressId*/ true)) 3035 return; 3036 assert(!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum) && 3037 "Target region entry already registered!"); 3038 OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags); 3039 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry; 3040 ++OffloadingEntriesNum; 3041 } 3042 } 3043 3044 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo( 3045 unsigned DeviceID, unsigned FileID, StringRef ParentName, unsigned LineNum, 3046 bool IgnoreAddressId) const { 3047 auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID); 3048 if (PerDevice == OffloadEntriesTargetRegion.end()) 3049 return false; 3050 auto PerFile = PerDevice->second.find(FileID); 3051 if (PerFile == PerDevice->second.end()) 3052 return false; 3053 auto PerParentName = PerFile->second.find(ParentName); 3054 if (PerParentName == PerFile->second.end()) 3055 return false; 3056 auto PerLine = PerParentName->second.find(LineNum); 3057 if (PerLine == PerParentName->second.end()) 3058 return false; 3059 // Fail if this entry is already registered. 3060 if (!IgnoreAddressId && 3061 (PerLine->second.getAddress() || PerLine->second.getID())) 3062 return false; 3063 return true; 3064 } 3065 3066 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo( 3067 const OffloadTargetRegionEntryInfoActTy &Action) { 3068 // Scan all target region entries and perform the provided action. 3069 for (const auto &D : OffloadEntriesTargetRegion) 3070 for (const auto &F : D.second) 3071 for (const auto &P : F.second) 3072 for (const auto &L : P.second) 3073 Action(D.first, F.first, P.first(), L.first, L.second); 3074 } 3075 3076 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3077 initializeDeviceGlobalVarEntryInfo(StringRef Name, 3078 OMPTargetGlobalVarEntryKind Flags, 3079 unsigned Order) { 3080 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3081 "only required for the device " 3082 "code generation."); 3083 OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags); 3084 ++OffloadingEntriesNum; 3085 } 3086 3087 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3088 registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr, 3089 CharUnits VarSize, 3090 OMPTargetGlobalVarEntryKind Flags, 3091 llvm::GlobalValue::LinkageTypes Linkage) { 3092 if (CGM.getLangOpts().OpenMPIsDevice) { 3093 // This could happen if the device compilation is invoked standalone. 3094 if (!hasDeviceGlobalVarEntryInfo(VarName)) 3095 return; 3096 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 3097 if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) { 3098 if (Entry.getVarSize().isZero()) { 3099 Entry.setVarSize(VarSize); 3100 Entry.setLinkage(Linkage); 3101 } 3102 return; 3103 } 3104 Entry.setVarSize(VarSize); 3105 Entry.setLinkage(Linkage); 3106 Entry.setAddress(Addr); 3107 } else { 3108 if (hasDeviceGlobalVarEntryInfo(VarName)) { 3109 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 3110 assert(Entry.isValid() && Entry.getFlags() == Flags && 3111 "Entry not initialized!"); 3112 if (Entry.getVarSize().isZero()) { 3113 Entry.setVarSize(VarSize); 3114 Entry.setLinkage(Linkage); 3115 } 3116 return; 3117 } 3118 OffloadEntriesDeviceGlobalVar.try_emplace( 3119 VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage); 3120 ++OffloadingEntriesNum; 3121 } 3122 } 3123 3124 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3125 actOnDeviceGlobalVarEntriesInfo( 3126 const OffloadDeviceGlobalVarEntryInfoActTy &Action) { 3127 // Scan all target region entries and perform the provided action. 3128 for (const auto &E : OffloadEntriesDeviceGlobalVar) 3129 Action(E.getKey(), E.getValue()); 3130 } 3131 3132 void CGOpenMPRuntime::createOffloadEntry( 3133 llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags, 3134 llvm::GlobalValue::LinkageTypes Linkage) { 3135 StringRef Name = Addr->getName(); 3136 llvm::Module &M = CGM.getModule(); 3137 llvm::LLVMContext &C = M.getContext(); 3138 3139 // Create constant string with the name. 3140 llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name); 3141 3142 std::string StringName = getName({"omp_offloading", "entry_name"}); 3143 auto *Str = new llvm::GlobalVariable( 3144 M, StrPtrInit->getType(), /*isConstant=*/true, 3145 llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName); 3146 Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 3147 3148 llvm::Constant *Data[] = { 3149 llvm::ConstantExpr::getPointerBitCastOrAddrSpaceCast(ID, CGM.VoidPtrTy), 3150 llvm::ConstantExpr::getPointerBitCastOrAddrSpaceCast(Str, CGM.Int8PtrTy), 3151 llvm::ConstantInt::get(CGM.SizeTy, Size), 3152 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 3153 llvm::ConstantInt::get(CGM.Int32Ty, 0)}; 3154 std::string EntryName = getName({"omp_offloading", "entry", ""}); 3155 llvm::GlobalVariable *Entry = createGlobalStruct( 3156 CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data, 3157 Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage); 3158 3159 // The entry has to be created in the section the linker expects it to be. 3160 Entry->setSection("omp_offloading_entries"); 3161 } 3162 3163 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() { 3164 // Emit the offloading entries and metadata so that the device codegen side 3165 // can easily figure out what to emit. The produced metadata looks like 3166 // this: 3167 // 3168 // !omp_offload.info = !{!1, ...} 3169 // 3170 // Right now we only generate metadata for function that contain target 3171 // regions. 3172 3173 // If we are in simd mode or there are no entries, we don't need to do 3174 // anything. 3175 if (CGM.getLangOpts().OpenMPSimd || OffloadEntriesInfoManager.empty()) 3176 return; 3177 3178 llvm::Module &M = CGM.getModule(); 3179 llvm::LLVMContext &C = M.getContext(); 3180 SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *, 3181 SourceLocation, StringRef>, 3182 16> 3183 OrderedEntries(OffloadEntriesInfoManager.size()); 3184 llvm::SmallVector<StringRef, 16> ParentFunctions( 3185 OffloadEntriesInfoManager.size()); 3186 3187 // Auxiliary methods to create metadata values and strings. 3188 auto &&GetMDInt = [this](unsigned V) { 3189 return llvm::ConstantAsMetadata::get( 3190 llvm::ConstantInt::get(CGM.Int32Ty, V)); 3191 }; 3192 3193 auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); }; 3194 3195 // Create the offloading info metadata node. 3196 llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info"); 3197 3198 // Create function that emits metadata for each target region entry; 3199 auto &&TargetRegionMetadataEmitter = 3200 [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt, 3201 &GetMDString]( 3202 unsigned DeviceID, unsigned FileID, StringRef ParentName, 3203 unsigned Line, 3204 const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) { 3205 // Generate metadata for target regions. Each entry of this metadata 3206 // contains: 3207 // - Entry 0 -> Kind of this type of metadata (0). 3208 // - Entry 1 -> Device ID of the file where the entry was identified. 3209 // - Entry 2 -> File ID of the file where the entry was identified. 3210 // - Entry 3 -> Mangled name of the function where the entry was 3211 // identified. 3212 // - Entry 4 -> Line in the file where the entry was identified. 3213 // - Entry 5 -> Order the entry was created. 3214 // The first element of the metadata node is the kind. 3215 llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID), 3216 GetMDInt(FileID), GetMDString(ParentName), 3217 GetMDInt(Line), GetMDInt(E.getOrder())}; 3218 3219 SourceLocation Loc; 3220 for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(), 3221 E = CGM.getContext().getSourceManager().fileinfo_end(); 3222 I != E; ++I) { 3223 if (I->getFirst()->getUniqueID().getDevice() == DeviceID && 3224 I->getFirst()->getUniqueID().getFile() == FileID) { 3225 Loc = CGM.getContext().getSourceManager().translateFileLineCol( 3226 I->getFirst(), Line, 1); 3227 break; 3228 } 3229 } 3230 // Save this entry in the right position of the ordered entries array. 3231 OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName); 3232 ParentFunctions[E.getOrder()] = ParentName; 3233 3234 // Add metadata to the named metadata node. 3235 MD->addOperand(llvm::MDNode::get(C, Ops)); 3236 }; 3237 3238 OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo( 3239 TargetRegionMetadataEmitter); 3240 3241 // Create function that emits metadata for each device global variable entry; 3242 auto &&DeviceGlobalVarMetadataEmitter = 3243 [&C, &OrderedEntries, &GetMDInt, &GetMDString, 3244 MD](StringRef MangledName, 3245 const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar 3246 &E) { 3247 // Generate metadata for global variables. Each entry of this metadata 3248 // contains: 3249 // - Entry 0 -> Kind of this type of metadata (1). 3250 // - Entry 1 -> Mangled name of the variable. 3251 // - Entry 2 -> Declare target kind. 3252 // - Entry 3 -> Order the entry was created. 3253 // The first element of the metadata node is the kind. 3254 llvm::Metadata *Ops[] = { 3255 GetMDInt(E.getKind()), GetMDString(MangledName), 3256 GetMDInt(E.getFlags()), GetMDInt(E.getOrder())}; 3257 3258 // Save this entry in the right position of the ordered entries array. 3259 OrderedEntries[E.getOrder()] = 3260 std::make_tuple(&E, SourceLocation(), MangledName); 3261 3262 // Add metadata to the named metadata node. 3263 MD->addOperand(llvm::MDNode::get(C, Ops)); 3264 }; 3265 3266 OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo( 3267 DeviceGlobalVarMetadataEmitter); 3268 3269 for (const auto &E : OrderedEntries) { 3270 assert(std::get<0>(E) && "All ordered entries must exist!"); 3271 if (const auto *CE = 3272 dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>( 3273 std::get<0>(E))) { 3274 if (!CE->getID() || !CE->getAddress()) { 3275 // Do not blame the entry if the parent funtion is not emitted. 3276 StringRef FnName = ParentFunctions[CE->getOrder()]; 3277 if (!CGM.GetGlobalValue(FnName)) 3278 continue; 3279 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3280 DiagnosticsEngine::Error, 3281 "Offloading entry for target region in %0 is incorrect: either the " 3282 "address or the ID is invalid."); 3283 CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName; 3284 continue; 3285 } 3286 createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0, 3287 CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage); 3288 } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy:: 3289 OffloadEntryInfoDeviceGlobalVar>( 3290 std::get<0>(E))) { 3291 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags = 3292 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 3293 CE->getFlags()); 3294 switch (Flags) { 3295 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: { 3296 if (CGM.getLangOpts().OpenMPIsDevice && 3297 CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory()) 3298 continue; 3299 if (!CE->getAddress()) { 3300 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3301 DiagnosticsEngine::Error, "Offloading entry for declare target " 3302 "variable %0 is incorrect: the " 3303 "address is invalid."); 3304 CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E); 3305 continue; 3306 } 3307 // The vaiable has no definition - no need to add the entry. 3308 if (CE->getVarSize().isZero()) 3309 continue; 3310 break; 3311 } 3312 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink: 3313 assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) || 3314 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) && 3315 "Declaret target link address is set."); 3316 if (CGM.getLangOpts().OpenMPIsDevice) 3317 continue; 3318 if (!CE->getAddress()) { 3319 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3320 DiagnosticsEngine::Error, 3321 "Offloading entry for declare target variable is incorrect: the " 3322 "address is invalid."); 3323 CGM.getDiags().Report(DiagID); 3324 continue; 3325 } 3326 break; 3327 } 3328 createOffloadEntry(CE->getAddress(), CE->getAddress(), 3329 CE->getVarSize().getQuantity(), Flags, 3330 CE->getLinkage()); 3331 } else { 3332 llvm_unreachable("Unsupported entry kind."); 3333 } 3334 } 3335 } 3336 3337 /// Loads all the offload entries information from the host IR 3338 /// metadata. 3339 void CGOpenMPRuntime::loadOffloadInfoMetadata() { 3340 // If we are in target mode, load the metadata from the host IR. This code has 3341 // to match the metadaata creation in createOffloadEntriesAndInfoMetadata(). 3342 3343 if (!CGM.getLangOpts().OpenMPIsDevice) 3344 return; 3345 3346 if (CGM.getLangOpts().OMPHostIRFile.empty()) 3347 return; 3348 3349 auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile); 3350 if (auto EC = Buf.getError()) { 3351 CGM.getDiags().Report(diag::err_cannot_open_file) 3352 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 3353 return; 3354 } 3355 3356 llvm::LLVMContext C; 3357 auto ME = expectedToErrorOrAndEmitErrors( 3358 C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C)); 3359 3360 if (auto EC = ME.getError()) { 3361 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3362 DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'"); 3363 CGM.getDiags().Report(DiagID) 3364 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 3365 return; 3366 } 3367 3368 llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info"); 3369 if (!MD) 3370 return; 3371 3372 for (llvm::MDNode *MN : MD->operands()) { 3373 auto &&GetMDInt = [MN](unsigned Idx) { 3374 auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx)); 3375 return cast<llvm::ConstantInt>(V->getValue())->getZExtValue(); 3376 }; 3377 3378 auto &&GetMDString = [MN](unsigned Idx) { 3379 auto *V = cast<llvm::MDString>(MN->getOperand(Idx)); 3380 return V->getString(); 3381 }; 3382 3383 switch (GetMDInt(0)) { 3384 default: 3385 llvm_unreachable("Unexpected metadata!"); 3386 break; 3387 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 3388 OffloadingEntryInfoTargetRegion: 3389 OffloadEntriesInfoManager.initializeTargetRegionEntryInfo( 3390 /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2), 3391 /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4), 3392 /*Order=*/GetMDInt(5)); 3393 break; 3394 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 3395 OffloadingEntryInfoDeviceGlobalVar: 3396 OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo( 3397 /*MangledName=*/GetMDString(1), 3398 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 3399 /*Flags=*/GetMDInt(2)), 3400 /*Order=*/GetMDInt(3)); 3401 break; 3402 } 3403 } 3404 } 3405 3406 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) { 3407 if (!KmpRoutineEntryPtrTy) { 3408 // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type. 3409 ASTContext &C = CGM.getContext(); 3410 QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy}; 3411 FunctionProtoType::ExtProtoInfo EPI; 3412 KmpRoutineEntryPtrQTy = C.getPointerType( 3413 C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI)); 3414 KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy); 3415 } 3416 } 3417 3418 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() { 3419 // Make sure the type of the entry is already created. This is the type we 3420 // have to create: 3421 // struct __tgt_offload_entry{ 3422 // void *addr; // Pointer to the offload entry info. 3423 // // (function or global) 3424 // char *name; // Name of the function or global. 3425 // size_t size; // Size of the entry info (0 if it a function). 3426 // int32_t flags; // Flags associated with the entry, e.g. 'link'. 3427 // int32_t reserved; // Reserved, to use by the runtime library. 3428 // }; 3429 if (TgtOffloadEntryQTy.isNull()) { 3430 ASTContext &C = CGM.getContext(); 3431 RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry"); 3432 RD->startDefinition(); 3433 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 3434 addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy)); 3435 addFieldToRecordDecl(C, RD, C.getSizeType()); 3436 addFieldToRecordDecl( 3437 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 3438 addFieldToRecordDecl( 3439 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 3440 RD->completeDefinition(); 3441 RD->addAttr(PackedAttr::CreateImplicit(C)); 3442 TgtOffloadEntryQTy = C.getRecordType(RD); 3443 } 3444 return TgtOffloadEntryQTy; 3445 } 3446 3447 namespace { 3448 struct PrivateHelpersTy { 3449 PrivateHelpersTy(const Expr *OriginalRef, const VarDecl *Original, 3450 const VarDecl *PrivateCopy, const VarDecl *PrivateElemInit) 3451 : OriginalRef(OriginalRef), Original(Original), PrivateCopy(PrivateCopy), 3452 PrivateElemInit(PrivateElemInit) {} 3453 PrivateHelpersTy(const VarDecl *Original) : Original(Original) {} 3454 const Expr *OriginalRef = nullptr; 3455 const VarDecl *Original = nullptr; 3456 const VarDecl *PrivateCopy = nullptr; 3457 const VarDecl *PrivateElemInit = nullptr; 3458 bool isLocalPrivate() const { 3459 return !OriginalRef && !PrivateCopy && !PrivateElemInit; 3460 } 3461 }; 3462 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy; 3463 } // anonymous namespace 3464 3465 static bool isAllocatableDecl(const VarDecl *VD) { 3466 const VarDecl *CVD = VD->getCanonicalDecl(); 3467 if (!CVD->hasAttr<OMPAllocateDeclAttr>()) 3468 return false; 3469 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>(); 3470 // Use the default allocation. 3471 return !(AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc && 3472 !AA->getAllocator()); 3473 } 3474 3475 static RecordDecl * 3476 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) { 3477 if (!Privates.empty()) { 3478 ASTContext &C = CGM.getContext(); 3479 // Build struct .kmp_privates_t. { 3480 // /* private vars */ 3481 // }; 3482 RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t"); 3483 RD->startDefinition(); 3484 for (const auto &Pair : Privates) { 3485 const VarDecl *VD = Pair.second.Original; 3486 QualType Type = VD->getType().getNonReferenceType(); 3487 // If the private variable is a local variable with lvalue ref type, 3488 // allocate the pointer instead of the pointee type. 3489 if (Pair.second.isLocalPrivate()) { 3490 if (VD->getType()->isLValueReferenceType()) 3491 Type = C.getPointerType(Type); 3492 if (isAllocatableDecl(VD)) 3493 Type = C.getPointerType(Type); 3494 } 3495 FieldDecl *FD = addFieldToRecordDecl(C, RD, Type); 3496 if (VD->hasAttrs()) { 3497 for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()), 3498 E(VD->getAttrs().end()); 3499 I != E; ++I) 3500 FD->addAttr(*I); 3501 } 3502 } 3503 RD->completeDefinition(); 3504 return RD; 3505 } 3506 return nullptr; 3507 } 3508 3509 static RecordDecl * 3510 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind, 3511 QualType KmpInt32Ty, 3512 QualType KmpRoutineEntryPointerQTy) { 3513 ASTContext &C = CGM.getContext(); 3514 // Build struct kmp_task_t { 3515 // void * shareds; 3516 // kmp_routine_entry_t routine; 3517 // kmp_int32 part_id; 3518 // kmp_cmplrdata_t data1; 3519 // kmp_cmplrdata_t data2; 3520 // For taskloops additional fields: 3521 // kmp_uint64 lb; 3522 // kmp_uint64 ub; 3523 // kmp_int64 st; 3524 // kmp_int32 liter; 3525 // void * reductions; 3526 // }; 3527 RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union); 3528 UD->startDefinition(); 3529 addFieldToRecordDecl(C, UD, KmpInt32Ty); 3530 addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy); 3531 UD->completeDefinition(); 3532 QualType KmpCmplrdataTy = C.getRecordType(UD); 3533 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t"); 3534 RD->startDefinition(); 3535 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 3536 addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy); 3537 addFieldToRecordDecl(C, RD, KmpInt32Ty); 3538 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 3539 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 3540 if (isOpenMPTaskLoopDirective(Kind)) { 3541 QualType KmpUInt64Ty = 3542 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0); 3543 QualType KmpInt64Ty = 3544 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 3545 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 3546 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 3547 addFieldToRecordDecl(C, RD, KmpInt64Ty); 3548 addFieldToRecordDecl(C, RD, KmpInt32Ty); 3549 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 3550 } 3551 RD->completeDefinition(); 3552 return RD; 3553 } 3554 3555 static RecordDecl * 3556 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy, 3557 ArrayRef<PrivateDataTy> Privates) { 3558 ASTContext &C = CGM.getContext(); 3559 // Build struct kmp_task_t_with_privates { 3560 // kmp_task_t task_data; 3561 // .kmp_privates_t. privates; 3562 // }; 3563 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates"); 3564 RD->startDefinition(); 3565 addFieldToRecordDecl(C, RD, KmpTaskTQTy); 3566 if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates)) 3567 addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD)); 3568 RD->completeDefinition(); 3569 return RD; 3570 } 3571 3572 /// Emit a proxy function which accepts kmp_task_t as the second 3573 /// argument. 3574 /// \code 3575 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) { 3576 /// TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt, 3577 /// For taskloops: 3578 /// tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 3579 /// tt->reductions, tt->shareds); 3580 /// return 0; 3581 /// } 3582 /// \endcode 3583 static llvm::Function * 3584 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc, 3585 OpenMPDirectiveKind Kind, QualType KmpInt32Ty, 3586 QualType KmpTaskTWithPrivatesPtrQTy, 3587 QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy, 3588 QualType SharedsPtrTy, llvm::Function *TaskFunction, 3589 llvm::Value *TaskPrivatesMap) { 3590 ASTContext &C = CGM.getContext(); 3591 FunctionArgList Args; 3592 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 3593 ImplicitParamDecl::Other); 3594 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3595 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 3596 ImplicitParamDecl::Other); 3597 Args.push_back(&GtidArg); 3598 Args.push_back(&TaskTypeArg); 3599 const auto &TaskEntryFnInfo = 3600 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 3601 llvm::FunctionType *TaskEntryTy = 3602 CGM.getTypes().GetFunctionType(TaskEntryFnInfo); 3603 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""}); 3604 auto *TaskEntry = llvm::Function::Create( 3605 TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 3606 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo); 3607 TaskEntry->setDoesNotRecurse(); 3608 CodeGenFunction CGF(CGM); 3609 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args, 3610 Loc, Loc); 3611 3612 // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map, 3613 // tt, 3614 // For taskloops: 3615 // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 3616 // tt->task_data.shareds); 3617 llvm::Value *GtidParam = CGF.EmitLoadOfScalar( 3618 CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc); 3619 LValue TDBase = CGF.EmitLoadOfPointerLValue( 3620 CGF.GetAddrOfLocalVar(&TaskTypeArg), 3621 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 3622 const auto *KmpTaskTWithPrivatesQTyRD = 3623 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 3624 LValue Base = 3625 CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 3626 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 3627 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 3628 LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI); 3629 llvm::Value *PartidParam = PartIdLVal.getPointer(CGF); 3630 3631 auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds); 3632 LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI); 3633 llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3634 CGF.EmitLoadOfScalar(SharedsLVal, Loc), 3635 CGF.ConvertTypeForMem(SharedsPtrTy)); 3636 3637 auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 3638 llvm::Value *PrivatesParam; 3639 if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) { 3640 LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI); 3641 PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3642 PrivatesLVal.getPointer(CGF), CGF.VoidPtrTy); 3643 } else { 3644 PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 3645 } 3646 3647 llvm::Value *CommonArgs[] = { 3648 GtidParam, PartidParam, PrivatesParam, TaskPrivatesMap, 3649 CGF.Builder 3650 .CreatePointerBitCastOrAddrSpaceCast(TDBase.getAddress(CGF), 3651 CGF.VoidPtrTy, CGF.Int8Ty) 3652 .getPointer()}; 3653 SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs), 3654 std::end(CommonArgs)); 3655 if (isOpenMPTaskLoopDirective(Kind)) { 3656 auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound); 3657 LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI); 3658 llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc); 3659 auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound); 3660 LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI); 3661 llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc); 3662 auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride); 3663 LValue StLVal = CGF.EmitLValueForField(Base, *StFI); 3664 llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc); 3665 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 3666 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 3667 llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc); 3668 auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions); 3669 LValue RLVal = CGF.EmitLValueForField(Base, *RFI); 3670 llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc); 3671 CallArgs.push_back(LBParam); 3672 CallArgs.push_back(UBParam); 3673 CallArgs.push_back(StParam); 3674 CallArgs.push_back(LIParam); 3675 CallArgs.push_back(RParam); 3676 } 3677 CallArgs.push_back(SharedsParam); 3678 3679 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction, 3680 CallArgs); 3681 CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)), 3682 CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty)); 3683 CGF.FinishFunction(); 3684 return TaskEntry; 3685 } 3686 3687 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM, 3688 SourceLocation Loc, 3689 QualType KmpInt32Ty, 3690 QualType KmpTaskTWithPrivatesPtrQTy, 3691 QualType KmpTaskTWithPrivatesQTy) { 3692 ASTContext &C = CGM.getContext(); 3693 FunctionArgList Args; 3694 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 3695 ImplicitParamDecl::Other); 3696 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3697 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 3698 ImplicitParamDecl::Other); 3699 Args.push_back(&GtidArg); 3700 Args.push_back(&TaskTypeArg); 3701 const auto &DestructorFnInfo = 3702 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 3703 llvm::FunctionType *DestructorFnTy = 3704 CGM.getTypes().GetFunctionType(DestructorFnInfo); 3705 std::string Name = 3706 CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""}); 3707 auto *DestructorFn = 3708 llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage, 3709 Name, &CGM.getModule()); 3710 CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn, 3711 DestructorFnInfo); 3712 DestructorFn->setDoesNotRecurse(); 3713 CodeGenFunction CGF(CGM); 3714 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo, 3715 Args, Loc, Loc); 3716 3717 LValue Base = CGF.EmitLoadOfPointerLValue( 3718 CGF.GetAddrOfLocalVar(&TaskTypeArg), 3719 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 3720 const auto *KmpTaskTWithPrivatesQTyRD = 3721 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 3722 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 3723 Base = CGF.EmitLValueForField(Base, *FI); 3724 for (const auto *Field : 3725 cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) { 3726 if (QualType::DestructionKind DtorKind = 3727 Field->getType().isDestructedType()) { 3728 LValue FieldLValue = CGF.EmitLValueForField(Base, Field); 3729 CGF.pushDestroy(DtorKind, FieldLValue.getAddress(CGF), Field->getType()); 3730 } 3731 } 3732 CGF.FinishFunction(); 3733 return DestructorFn; 3734 } 3735 3736 /// Emit a privates mapping function for correct handling of private and 3737 /// firstprivate variables. 3738 /// \code 3739 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1> 3740 /// **noalias priv1,..., <tyn> **noalias privn) { 3741 /// *priv1 = &.privates.priv1; 3742 /// ...; 3743 /// *privn = &.privates.privn; 3744 /// } 3745 /// \endcode 3746 static llvm::Value * 3747 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc, 3748 const OMPTaskDataTy &Data, QualType PrivatesQTy, 3749 ArrayRef<PrivateDataTy> Privates) { 3750 ASTContext &C = CGM.getContext(); 3751 FunctionArgList Args; 3752 ImplicitParamDecl TaskPrivatesArg( 3753 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3754 C.getPointerType(PrivatesQTy).withConst().withRestrict(), 3755 ImplicitParamDecl::Other); 3756 Args.push_back(&TaskPrivatesArg); 3757 llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, unsigned> PrivateVarsPos; 3758 unsigned Counter = 1; 3759 for (const Expr *E : Data.PrivateVars) { 3760 Args.push_back(ImplicitParamDecl::Create( 3761 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3762 C.getPointerType(C.getPointerType(E->getType())) 3763 .withConst() 3764 .withRestrict(), 3765 ImplicitParamDecl::Other)); 3766 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 3767 PrivateVarsPos[VD] = Counter; 3768 ++Counter; 3769 } 3770 for (const Expr *E : Data.FirstprivateVars) { 3771 Args.push_back(ImplicitParamDecl::Create( 3772 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3773 C.getPointerType(C.getPointerType(E->getType())) 3774 .withConst() 3775 .withRestrict(), 3776 ImplicitParamDecl::Other)); 3777 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 3778 PrivateVarsPos[VD] = Counter; 3779 ++Counter; 3780 } 3781 for (const Expr *E : Data.LastprivateVars) { 3782 Args.push_back(ImplicitParamDecl::Create( 3783 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3784 C.getPointerType(C.getPointerType(E->getType())) 3785 .withConst() 3786 .withRestrict(), 3787 ImplicitParamDecl::Other)); 3788 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 3789 PrivateVarsPos[VD] = Counter; 3790 ++Counter; 3791 } 3792 for (const VarDecl *VD : Data.PrivateLocals) { 3793 QualType Ty = VD->getType().getNonReferenceType(); 3794 if (VD->getType()->isLValueReferenceType()) 3795 Ty = C.getPointerType(Ty); 3796 if (isAllocatableDecl(VD)) 3797 Ty = C.getPointerType(Ty); 3798 Args.push_back(ImplicitParamDecl::Create( 3799 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 3800 C.getPointerType(C.getPointerType(Ty)).withConst().withRestrict(), 3801 ImplicitParamDecl::Other)); 3802 PrivateVarsPos[VD] = Counter; 3803 ++Counter; 3804 } 3805 const auto &TaskPrivatesMapFnInfo = 3806 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 3807 llvm::FunctionType *TaskPrivatesMapTy = 3808 CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo); 3809 std::string Name = 3810 CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""}); 3811 auto *TaskPrivatesMap = llvm::Function::Create( 3812 TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name, 3813 &CGM.getModule()); 3814 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap, 3815 TaskPrivatesMapFnInfo); 3816 if (CGM.getLangOpts().Optimize) { 3817 TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline); 3818 TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone); 3819 TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline); 3820 } 3821 CodeGenFunction CGF(CGM); 3822 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap, 3823 TaskPrivatesMapFnInfo, Args, Loc, Loc); 3824 3825 // *privi = &.privates.privi; 3826 LValue Base = CGF.EmitLoadOfPointerLValue( 3827 CGF.GetAddrOfLocalVar(&TaskPrivatesArg), 3828 TaskPrivatesArg.getType()->castAs<PointerType>()); 3829 const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl()); 3830 Counter = 0; 3831 for (const FieldDecl *Field : PrivatesQTyRD->fields()) { 3832 LValue FieldLVal = CGF.EmitLValueForField(Base, Field); 3833 const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]]; 3834 LValue RefLVal = 3835 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType()); 3836 LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue( 3837 RefLVal.getAddress(CGF), RefLVal.getType()->castAs<PointerType>()); 3838 CGF.EmitStoreOfScalar(FieldLVal.getPointer(CGF), RefLoadLVal); 3839 ++Counter; 3840 } 3841 CGF.FinishFunction(); 3842 return TaskPrivatesMap; 3843 } 3844 3845 /// Emit initialization for private variables in task-based directives. 3846 static void emitPrivatesInit(CodeGenFunction &CGF, 3847 const OMPExecutableDirective &D, 3848 Address KmpTaskSharedsPtr, LValue TDBase, 3849 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 3850 QualType SharedsTy, QualType SharedsPtrTy, 3851 const OMPTaskDataTy &Data, 3852 ArrayRef<PrivateDataTy> Privates, bool ForDup) { 3853 ASTContext &C = CGF.getContext(); 3854 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 3855 LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI); 3856 OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind()) 3857 ? OMPD_taskloop 3858 : OMPD_task; 3859 const CapturedStmt &CS = *D.getCapturedStmt(Kind); 3860 CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS); 3861 LValue SrcBase; 3862 bool IsTargetTask = 3863 isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) || 3864 isOpenMPTargetExecutionDirective(D.getDirectiveKind()); 3865 // For target-based directives skip 4 firstprivate arrays BasePointersArray, 3866 // PointersArray, SizesArray, and MappersArray. The original variables for 3867 // these arrays are not captured and we get their addresses explicitly. 3868 if ((!IsTargetTask && !Data.FirstprivateVars.empty() && ForDup) || 3869 (IsTargetTask && KmpTaskSharedsPtr.isValid())) { 3870 SrcBase = CGF.MakeAddrLValue( 3871 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3872 KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy), 3873 CGF.ConvertTypeForMem(SharedsTy)), 3874 SharedsTy); 3875 } 3876 FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin(); 3877 for (const PrivateDataTy &Pair : Privates) { 3878 // Do not initialize private locals. 3879 if (Pair.second.isLocalPrivate()) { 3880 ++FI; 3881 continue; 3882 } 3883 const VarDecl *VD = Pair.second.PrivateCopy; 3884 const Expr *Init = VD->getAnyInitializer(); 3885 if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) && 3886 !CGF.isTrivialInitializer(Init)))) { 3887 LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI); 3888 if (const VarDecl *Elem = Pair.second.PrivateElemInit) { 3889 const VarDecl *OriginalVD = Pair.second.Original; 3890 // Check if the variable is the target-based BasePointersArray, 3891 // PointersArray, SizesArray, or MappersArray. 3892 LValue SharedRefLValue; 3893 QualType Type = PrivateLValue.getType(); 3894 const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD); 3895 if (IsTargetTask && !SharedField) { 3896 assert(isa<ImplicitParamDecl>(OriginalVD) && 3897 isa<CapturedDecl>(OriginalVD->getDeclContext()) && 3898 cast<CapturedDecl>(OriginalVD->getDeclContext()) 3899 ->getNumParams() == 0 && 3900 isa<TranslationUnitDecl>( 3901 cast<CapturedDecl>(OriginalVD->getDeclContext()) 3902 ->getDeclContext()) && 3903 "Expected artificial target data variable."); 3904 SharedRefLValue = 3905 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type); 3906 } else if (ForDup) { 3907 SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField); 3908 SharedRefLValue = CGF.MakeAddrLValue( 3909 SharedRefLValue.getAddress(CGF).withAlignment( 3910 C.getDeclAlign(OriginalVD)), 3911 SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl), 3912 SharedRefLValue.getTBAAInfo()); 3913 } else if (CGF.LambdaCaptureFields.count( 3914 Pair.second.Original->getCanonicalDecl()) > 0 || 3915 isa_and_nonnull<BlockDecl>(CGF.CurCodeDecl)) { 3916 SharedRefLValue = CGF.EmitLValue(Pair.second.OriginalRef); 3917 } else { 3918 // Processing for implicitly captured variables. 3919 InlinedOpenMPRegionRAII Region( 3920 CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown, 3921 /*HasCancel=*/false, /*NoInheritance=*/true); 3922 SharedRefLValue = CGF.EmitLValue(Pair.second.OriginalRef); 3923 } 3924 if (Type->isArrayType()) { 3925 // Initialize firstprivate array. 3926 if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) { 3927 // Perform simple memcpy. 3928 CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type); 3929 } else { 3930 // Initialize firstprivate array using element-by-element 3931 // initialization. 3932 CGF.EmitOMPAggregateAssign( 3933 PrivateLValue.getAddress(CGF), SharedRefLValue.getAddress(CGF), 3934 Type, 3935 [&CGF, Elem, Init, &CapturesInfo](Address DestElement, 3936 Address SrcElement) { 3937 // Clean up any temporaries needed by the initialization. 3938 CodeGenFunction::OMPPrivateScope InitScope(CGF); 3939 InitScope.addPrivate(Elem, SrcElement); 3940 (void)InitScope.Privatize(); 3941 // Emit initialization for single element. 3942 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII( 3943 CGF, &CapturesInfo); 3944 CGF.EmitAnyExprToMem(Init, DestElement, 3945 Init->getType().getQualifiers(), 3946 /*IsInitializer=*/false); 3947 }); 3948 } 3949 } else { 3950 CodeGenFunction::OMPPrivateScope InitScope(CGF); 3951 InitScope.addPrivate(Elem, SharedRefLValue.getAddress(CGF)); 3952 (void)InitScope.Privatize(); 3953 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo); 3954 CGF.EmitExprAsInit(Init, VD, PrivateLValue, 3955 /*capturedByInit=*/false); 3956 } 3957 } else { 3958 CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false); 3959 } 3960 } 3961 ++FI; 3962 } 3963 } 3964 3965 /// Check if duplication function is required for taskloops. 3966 static bool checkInitIsRequired(CodeGenFunction &CGF, 3967 ArrayRef<PrivateDataTy> Privates) { 3968 bool InitRequired = false; 3969 for (const PrivateDataTy &Pair : Privates) { 3970 if (Pair.second.isLocalPrivate()) 3971 continue; 3972 const VarDecl *VD = Pair.second.PrivateCopy; 3973 const Expr *Init = VD->getAnyInitializer(); 3974 InitRequired = InitRequired || (isa_and_nonnull<CXXConstructExpr>(Init) && 3975 !CGF.isTrivialInitializer(Init)); 3976 if (InitRequired) 3977 break; 3978 } 3979 return InitRequired; 3980 } 3981 3982 3983 /// Emit task_dup function (for initialization of 3984 /// private/firstprivate/lastprivate vars and last_iter flag) 3985 /// \code 3986 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int 3987 /// lastpriv) { 3988 /// // setup lastprivate flag 3989 /// task_dst->last = lastpriv; 3990 /// // could be constructor calls here... 3991 /// } 3992 /// \endcode 3993 static llvm::Value * 3994 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc, 3995 const OMPExecutableDirective &D, 3996 QualType KmpTaskTWithPrivatesPtrQTy, 3997 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 3998 const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy, 3999 QualType SharedsPtrTy, const OMPTaskDataTy &Data, 4000 ArrayRef<PrivateDataTy> Privates, bool WithLastIter) { 4001 ASTContext &C = CGM.getContext(); 4002 FunctionArgList Args; 4003 ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4004 KmpTaskTWithPrivatesPtrQTy, 4005 ImplicitParamDecl::Other); 4006 ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4007 KmpTaskTWithPrivatesPtrQTy, 4008 ImplicitParamDecl::Other); 4009 ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy, 4010 ImplicitParamDecl::Other); 4011 Args.push_back(&DstArg); 4012 Args.push_back(&SrcArg); 4013 Args.push_back(&LastprivArg); 4014 const auto &TaskDupFnInfo = 4015 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4016 llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo); 4017 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""}); 4018 auto *TaskDup = llvm::Function::Create( 4019 TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 4020 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo); 4021 TaskDup->setDoesNotRecurse(); 4022 CodeGenFunction CGF(CGM); 4023 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc, 4024 Loc); 4025 4026 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4027 CGF.GetAddrOfLocalVar(&DstArg), 4028 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4029 // task_dst->liter = lastpriv; 4030 if (WithLastIter) { 4031 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 4032 LValue Base = CGF.EmitLValueForField( 4033 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4034 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 4035 llvm::Value *Lastpriv = CGF.EmitLoadOfScalar( 4036 CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc); 4037 CGF.EmitStoreOfScalar(Lastpriv, LILVal); 4038 } 4039 4040 // Emit initial values for private copies (if any). 4041 assert(!Privates.empty()); 4042 Address KmpTaskSharedsPtr = Address::invalid(); 4043 if (!Data.FirstprivateVars.empty()) { 4044 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4045 CGF.GetAddrOfLocalVar(&SrcArg), 4046 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4047 LValue Base = CGF.EmitLValueForField( 4048 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4049 KmpTaskSharedsPtr = Address::deprecated( 4050 CGF.EmitLoadOfScalar(CGF.EmitLValueForField( 4051 Base, *std::next(KmpTaskTQTyRD->field_begin(), 4052 KmpTaskTShareds)), 4053 Loc), 4054 CGM.getNaturalTypeAlignment(SharedsTy)); 4055 } 4056 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD, 4057 SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true); 4058 CGF.FinishFunction(); 4059 return TaskDup; 4060 } 4061 4062 /// Checks if destructor function is required to be generated. 4063 /// \return true if cleanups are required, false otherwise. 4064 static bool 4065 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD, 4066 ArrayRef<PrivateDataTy> Privates) { 4067 for (const PrivateDataTy &P : Privates) { 4068 if (P.second.isLocalPrivate()) 4069 continue; 4070 QualType Ty = P.second.Original->getType().getNonReferenceType(); 4071 if (Ty.isDestructedType()) 4072 return true; 4073 } 4074 return false; 4075 } 4076 4077 namespace { 4078 /// Loop generator for OpenMP iterator expression. 4079 class OMPIteratorGeneratorScope final 4080 : public CodeGenFunction::OMPPrivateScope { 4081 CodeGenFunction &CGF; 4082 const OMPIteratorExpr *E = nullptr; 4083 SmallVector<CodeGenFunction::JumpDest, 4> ContDests; 4084 SmallVector<CodeGenFunction::JumpDest, 4> ExitDests; 4085 OMPIteratorGeneratorScope() = delete; 4086 OMPIteratorGeneratorScope(OMPIteratorGeneratorScope &) = delete; 4087 4088 public: 4089 OMPIteratorGeneratorScope(CodeGenFunction &CGF, const OMPIteratorExpr *E) 4090 : CodeGenFunction::OMPPrivateScope(CGF), CGF(CGF), E(E) { 4091 if (!E) 4092 return; 4093 SmallVector<llvm::Value *, 4> Uppers; 4094 for (unsigned I = 0, End = E->numOfIterators(); I < End; ++I) { 4095 Uppers.push_back(CGF.EmitScalarExpr(E->getHelper(I).Upper)); 4096 const auto *VD = cast<VarDecl>(E->getIteratorDecl(I)); 4097 addPrivate(VD, CGF.CreateMemTemp(VD->getType(), VD->getName())); 4098 const OMPIteratorHelperData &HelperData = E->getHelper(I); 4099 addPrivate( 4100 HelperData.CounterVD, 4101 CGF.CreateMemTemp(HelperData.CounterVD->getType(), "counter.addr")); 4102 } 4103 Privatize(); 4104 4105 for (unsigned I = 0, End = E->numOfIterators(); I < End; ++I) { 4106 const OMPIteratorHelperData &HelperData = E->getHelper(I); 4107 LValue CLVal = 4108 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(HelperData.CounterVD), 4109 HelperData.CounterVD->getType()); 4110 // Counter = 0; 4111 CGF.EmitStoreOfScalar( 4112 llvm::ConstantInt::get(CLVal.getAddress(CGF).getElementType(), 0), 4113 CLVal); 4114 CodeGenFunction::JumpDest &ContDest = 4115 ContDests.emplace_back(CGF.getJumpDestInCurrentScope("iter.cont")); 4116 CodeGenFunction::JumpDest &ExitDest = 4117 ExitDests.emplace_back(CGF.getJumpDestInCurrentScope("iter.exit")); 4118 // N = <number-of_iterations>; 4119 llvm::Value *N = Uppers[I]; 4120 // cont: 4121 // if (Counter < N) goto body; else goto exit; 4122 CGF.EmitBlock(ContDest.getBlock()); 4123 auto *CVal = 4124 CGF.EmitLoadOfScalar(CLVal, HelperData.CounterVD->getLocation()); 4125 llvm::Value *Cmp = 4126 HelperData.CounterVD->getType()->isSignedIntegerOrEnumerationType() 4127 ? CGF.Builder.CreateICmpSLT(CVal, N) 4128 : CGF.Builder.CreateICmpULT(CVal, N); 4129 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("iter.body"); 4130 CGF.Builder.CreateCondBr(Cmp, BodyBB, ExitDest.getBlock()); 4131 // body: 4132 CGF.EmitBlock(BodyBB); 4133 // Iteri = Begini + Counter * Stepi; 4134 CGF.EmitIgnoredExpr(HelperData.Update); 4135 } 4136 } 4137 ~OMPIteratorGeneratorScope() { 4138 if (!E) 4139 return; 4140 for (unsigned I = E->numOfIterators(); I > 0; --I) { 4141 // Counter = Counter + 1; 4142 const OMPIteratorHelperData &HelperData = E->getHelper(I - 1); 4143 CGF.EmitIgnoredExpr(HelperData.CounterUpdate); 4144 // goto cont; 4145 CGF.EmitBranchThroughCleanup(ContDests[I - 1]); 4146 // exit: 4147 CGF.EmitBlock(ExitDests[I - 1].getBlock(), /*IsFinished=*/I == 1); 4148 } 4149 } 4150 }; 4151 } // namespace 4152 4153 static std::pair<llvm::Value *, llvm::Value *> 4154 getPointerAndSize(CodeGenFunction &CGF, const Expr *E) { 4155 const auto *OASE = dyn_cast<OMPArrayShapingExpr>(E); 4156 llvm::Value *Addr; 4157 if (OASE) { 4158 const Expr *Base = OASE->getBase(); 4159 Addr = CGF.EmitScalarExpr(Base); 4160 } else { 4161 Addr = CGF.EmitLValue(E).getPointer(CGF); 4162 } 4163 llvm::Value *SizeVal; 4164 QualType Ty = E->getType(); 4165 if (OASE) { 4166 SizeVal = CGF.getTypeSize(OASE->getBase()->getType()->getPointeeType()); 4167 for (const Expr *SE : OASE->getDimensions()) { 4168 llvm::Value *Sz = CGF.EmitScalarExpr(SE); 4169 Sz = CGF.EmitScalarConversion( 4170 Sz, SE->getType(), CGF.getContext().getSizeType(), SE->getExprLoc()); 4171 SizeVal = CGF.Builder.CreateNUWMul(SizeVal, Sz); 4172 } 4173 } else if (const auto *ASE = 4174 dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) { 4175 LValue UpAddrLVal = 4176 CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false); 4177 Address UpAddrAddress = UpAddrLVal.getAddress(CGF); 4178 llvm::Value *UpAddr = CGF.Builder.CreateConstGEP1_32( 4179 UpAddrAddress.getElementType(), UpAddrAddress.getPointer(), /*Idx0=*/1); 4180 llvm::Value *LowIntPtr = CGF.Builder.CreatePtrToInt(Addr, CGF.SizeTy); 4181 llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGF.SizeTy); 4182 SizeVal = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr); 4183 } else { 4184 SizeVal = CGF.getTypeSize(Ty); 4185 } 4186 return std::make_pair(Addr, SizeVal); 4187 } 4188 4189 /// Builds kmp_depend_info, if it is not built yet, and builds flags type. 4190 static void getKmpAffinityType(ASTContext &C, QualType &KmpTaskAffinityInfoTy) { 4191 QualType FlagsTy = C.getIntTypeForBitwidth(32, /*Signed=*/false); 4192 if (KmpTaskAffinityInfoTy.isNull()) { 4193 RecordDecl *KmpAffinityInfoRD = 4194 C.buildImplicitRecord("kmp_task_affinity_info_t"); 4195 KmpAffinityInfoRD->startDefinition(); 4196 addFieldToRecordDecl(C, KmpAffinityInfoRD, C.getIntPtrType()); 4197 addFieldToRecordDecl(C, KmpAffinityInfoRD, C.getSizeType()); 4198 addFieldToRecordDecl(C, KmpAffinityInfoRD, FlagsTy); 4199 KmpAffinityInfoRD->completeDefinition(); 4200 KmpTaskAffinityInfoTy = C.getRecordType(KmpAffinityInfoRD); 4201 } 4202 } 4203 4204 CGOpenMPRuntime::TaskResultTy 4205 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc, 4206 const OMPExecutableDirective &D, 4207 llvm::Function *TaskFunction, QualType SharedsTy, 4208 Address Shareds, const OMPTaskDataTy &Data) { 4209 ASTContext &C = CGM.getContext(); 4210 llvm::SmallVector<PrivateDataTy, 4> Privates; 4211 // Aggregate privates and sort them by the alignment. 4212 const auto *I = Data.PrivateCopies.begin(); 4213 for (const Expr *E : Data.PrivateVars) { 4214 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4215 Privates.emplace_back( 4216 C.getDeclAlign(VD), 4217 PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4218 /*PrivateElemInit=*/nullptr)); 4219 ++I; 4220 } 4221 I = Data.FirstprivateCopies.begin(); 4222 const auto *IElemInitRef = Data.FirstprivateInits.begin(); 4223 for (const Expr *E : Data.FirstprivateVars) { 4224 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4225 Privates.emplace_back( 4226 C.getDeclAlign(VD), 4227 PrivateHelpersTy( 4228 E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4229 cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl()))); 4230 ++I; 4231 ++IElemInitRef; 4232 } 4233 I = Data.LastprivateCopies.begin(); 4234 for (const Expr *E : Data.LastprivateVars) { 4235 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4236 Privates.emplace_back( 4237 C.getDeclAlign(VD), 4238 PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4239 /*PrivateElemInit=*/nullptr)); 4240 ++I; 4241 } 4242 for (const VarDecl *VD : Data.PrivateLocals) { 4243 if (isAllocatableDecl(VD)) 4244 Privates.emplace_back(CGM.getPointerAlign(), PrivateHelpersTy(VD)); 4245 else 4246 Privates.emplace_back(C.getDeclAlign(VD), PrivateHelpersTy(VD)); 4247 } 4248 llvm::stable_sort(Privates, 4249 [](const PrivateDataTy &L, const PrivateDataTy &R) { 4250 return L.first > R.first; 4251 }); 4252 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 4253 // Build type kmp_routine_entry_t (if not built yet). 4254 emitKmpRoutineEntryT(KmpInt32Ty); 4255 // Build type kmp_task_t (if not built yet). 4256 if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) { 4257 if (SavedKmpTaskloopTQTy.isNull()) { 4258 SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl( 4259 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 4260 } 4261 KmpTaskTQTy = SavedKmpTaskloopTQTy; 4262 } else { 4263 assert((D.getDirectiveKind() == OMPD_task || 4264 isOpenMPTargetExecutionDirective(D.getDirectiveKind()) || 4265 isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) && 4266 "Expected taskloop, task or target directive"); 4267 if (SavedKmpTaskTQTy.isNull()) { 4268 SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl( 4269 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 4270 } 4271 KmpTaskTQTy = SavedKmpTaskTQTy; 4272 } 4273 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 4274 // Build particular struct kmp_task_t for the given task. 4275 const RecordDecl *KmpTaskTWithPrivatesQTyRD = 4276 createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates); 4277 QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD); 4278 QualType KmpTaskTWithPrivatesPtrQTy = 4279 C.getPointerType(KmpTaskTWithPrivatesQTy); 4280 llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy); 4281 llvm::Type *KmpTaskTWithPrivatesPtrTy = 4282 KmpTaskTWithPrivatesTy->getPointerTo(); 4283 llvm::Value *KmpTaskTWithPrivatesTySize = 4284 CGF.getTypeSize(KmpTaskTWithPrivatesQTy); 4285 QualType SharedsPtrTy = C.getPointerType(SharedsTy); 4286 4287 // Emit initial values for private copies (if any). 4288 llvm::Value *TaskPrivatesMap = nullptr; 4289 llvm::Type *TaskPrivatesMapTy = 4290 std::next(TaskFunction->arg_begin(), 3)->getType(); 4291 if (!Privates.empty()) { 4292 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4293 TaskPrivatesMap = 4294 emitTaskPrivateMappingFunction(CGM, Loc, Data, FI->getType(), Privates); 4295 TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4296 TaskPrivatesMap, TaskPrivatesMapTy); 4297 } else { 4298 TaskPrivatesMap = llvm::ConstantPointerNull::get( 4299 cast<llvm::PointerType>(TaskPrivatesMapTy)); 4300 } 4301 // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid, 4302 // kmp_task_t *tt); 4303 llvm::Function *TaskEntry = emitProxyTaskFunction( 4304 CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 4305 KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction, 4306 TaskPrivatesMap); 4307 4308 // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 4309 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 4310 // kmp_routine_entry_t *task_entry); 4311 // Task flags. Format is taken from 4312 // https://github.com/llvm/llvm-project/blob/main/openmp/runtime/src/kmp.h, 4313 // description of kmp_tasking_flags struct. 4314 enum { 4315 TiedFlag = 0x1, 4316 FinalFlag = 0x2, 4317 DestructorsFlag = 0x8, 4318 PriorityFlag = 0x20, 4319 DetachableFlag = 0x40, 4320 }; 4321 unsigned Flags = Data.Tied ? TiedFlag : 0; 4322 bool NeedsCleanup = false; 4323 if (!Privates.empty()) { 4324 NeedsCleanup = 4325 checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD, Privates); 4326 if (NeedsCleanup) 4327 Flags = Flags | DestructorsFlag; 4328 } 4329 if (Data.Priority.getInt()) 4330 Flags = Flags | PriorityFlag; 4331 if (D.hasClausesOfKind<OMPDetachClause>()) 4332 Flags = Flags | DetachableFlag; 4333 llvm::Value *TaskFlags = 4334 Data.Final.getPointer() 4335 ? CGF.Builder.CreateSelect(Data.Final.getPointer(), 4336 CGF.Builder.getInt32(FinalFlag), 4337 CGF.Builder.getInt32(/*C=*/0)) 4338 : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0); 4339 TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags)); 4340 llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy)); 4341 SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc), 4342 getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize, 4343 SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4344 TaskEntry, KmpRoutineEntryPtrTy)}; 4345 llvm::Value *NewTask; 4346 if (D.hasClausesOfKind<OMPNowaitClause>()) { 4347 // Check if we have any device clause associated with the directive. 4348 const Expr *Device = nullptr; 4349 if (auto *C = D.getSingleClause<OMPDeviceClause>()) 4350 Device = C->getDevice(); 4351 // Emit device ID if any otherwise use default value. 4352 llvm::Value *DeviceID; 4353 if (Device) 4354 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 4355 CGF.Int64Ty, /*isSigned=*/true); 4356 else 4357 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 4358 AllocArgs.push_back(DeviceID); 4359 NewTask = CGF.EmitRuntimeCall( 4360 OMPBuilder.getOrCreateRuntimeFunction( 4361 CGM.getModule(), OMPRTL___kmpc_omp_target_task_alloc), 4362 AllocArgs); 4363 } else { 4364 NewTask = 4365 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 4366 CGM.getModule(), OMPRTL___kmpc_omp_task_alloc), 4367 AllocArgs); 4368 } 4369 // Emit detach clause initialization. 4370 // evt = (typeof(evt))__kmpc_task_allow_completion_event(loc, tid, 4371 // task_descriptor); 4372 if (const auto *DC = D.getSingleClause<OMPDetachClause>()) { 4373 const Expr *Evt = DC->getEventHandler()->IgnoreParenImpCasts(); 4374 LValue EvtLVal = CGF.EmitLValue(Evt); 4375 4376 // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref, 4377 // int gtid, kmp_task_t *task); 4378 llvm::Value *Loc = emitUpdateLocation(CGF, DC->getBeginLoc()); 4379 llvm::Value *Tid = getThreadID(CGF, DC->getBeginLoc()); 4380 Tid = CGF.Builder.CreateIntCast(Tid, CGF.IntTy, /*isSigned=*/false); 4381 llvm::Value *EvtVal = CGF.EmitRuntimeCall( 4382 OMPBuilder.getOrCreateRuntimeFunction( 4383 CGM.getModule(), OMPRTL___kmpc_task_allow_completion_event), 4384 {Loc, Tid, NewTask}); 4385 EvtVal = CGF.EmitScalarConversion(EvtVal, C.VoidPtrTy, Evt->getType(), 4386 Evt->getExprLoc()); 4387 CGF.EmitStoreOfScalar(EvtVal, EvtLVal); 4388 } 4389 // Process affinity clauses. 4390 if (D.hasClausesOfKind<OMPAffinityClause>()) { 4391 // Process list of affinity data. 4392 ASTContext &C = CGM.getContext(); 4393 Address AffinitiesArray = Address::invalid(); 4394 // Calculate number of elements to form the array of affinity data. 4395 llvm::Value *NumOfElements = nullptr; 4396 unsigned NumAffinities = 0; 4397 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) { 4398 if (const Expr *Modifier = C->getModifier()) { 4399 const auto *IE = cast<OMPIteratorExpr>(Modifier->IgnoreParenImpCasts()); 4400 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) { 4401 llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper); 4402 Sz = CGF.Builder.CreateIntCast(Sz, CGF.SizeTy, /*isSigned=*/false); 4403 NumOfElements = 4404 NumOfElements ? CGF.Builder.CreateNUWMul(NumOfElements, Sz) : Sz; 4405 } 4406 } else { 4407 NumAffinities += C->varlist_size(); 4408 } 4409 } 4410 getKmpAffinityType(CGM.getContext(), KmpTaskAffinityInfoTy); 4411 // Fields ids in kmp_task_affinity_info record. 4412 enum RTLAffinityInfoFieldsTy { BaseAddr, Len, Flags }; 4413 4414 QualType KmpTaskAffinityInfoArrayTy; 4415 if (NumOfElements) { 4416 NumOfElements = CGF.Builder.CreateNUWAdd( 4417 llvm::ConstantInt::get(CGF.SizeTy, NumAffinities), NumOfElements); 4418 auto *OVE = new (C) OpaqueValueExpr( 4419 Loc, 4420 C.getIntTypeForBitwidth(C.getTypeSize(C.getSizeType()), /*Signed=*/0), 4421 VK_PRValue); 4422 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, OVE, 4423 RValue::get(NumOfElements)); 4424 KmpTaskAffinityInfoArrayTy = 4425 C.getVariableArrayType(KmpTaskAffinityInfoTy, OVE, ArrayType::Normal, 4426 /*IndexTypeQuals=*/0, SourceRange(Loc, Loc)); 4427 // Properly emit variable-sized array. 4428 auto *PD = ImplicitParamDecl::Create(C, KmpTaskAffinityInfoArrayTy, 4429 ImplicitParamDecl::Other); 4430 CGF.EmitVarDecl(*PD); 4431 AffinitiesArray = CGF.GetAddrOfLocalVar(PD); 4432 NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty, 4433 /*isSigned=*/false); 4434 } else { 4435 KmpTaskAffinityInfoArrayTy = C.getConstantArrayType( 4436 KmpTaskAffinityInfoTy, 4437 llvm::APInt(C.getTypeSize(C.getSizeType()), NumAffinities), nullptr, 4438 ArrayType::Normal, /*IndexTypeQuals=*/0); 4439 AffinitiesArray = 4440 CGF.CreateMemTemp(KmpTaskAffinityInfoArrayTy, ".affs.arr.addr"); 4441 AffinitiesArray = CGF.Builder.CreateConstArrayGEP(AffinitiesArray, 0); 4442 NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumAffinities, 4443 /*isSigned=*/false); 4444 } 4445 4446 const auto *KmpAffinityInfoRD = KmpTaskAffinityInfoTy->getAsRecordDecl(); 4447 // Fill array by elements without iterators. 4448 unsigned Pos = 0; 4449 bool HasIterator = false; 4450 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) { 4451 if (C->getModifier()) { 4452 HasIterator = true; 4453 continue; 4454 } 4455 for (const Expr *E : C->varlists()) { 4456 llvm::Value *Addr; 4457 llvm::Value *Size; 4458 std::tie(Addr, Size) = getPointerAndSize(CGF, E); 4459 LValue Base = 4460 CGF.MakeAddrLValue(CGF.Builder.CreateConstGEP(AffinitiesArray, Pos), 4461 KmpTaskAffinityInfoTy); 4462 // affs[i].base_addr = &<Affinities[i].second>; 4463 LValue BaseAddrLVal = CGF.EmitLValueForField( 4464 Base, *std::next(KmpAffinityInfoRD->field_begin(), BaseAddr)); 4465 CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy), 4466 BaseAddrLVal); 4467 // affs[i].len = sizeof(<Affinities[i].second>); 4468 LValue LenLVal = CGF.EmitLValueForField( 4469 Base, *std::next(KmpAffinityInfoRD->field_begin(), Len)); 4470 CGF.EmitStoreOfScalar(Size, LenLVal); 4471 ++Pos; 4472 } 4473 } 4474 LValue PosLVal; 4475 if (HasIterator) { 4476 PosLVal = CGF.MakeAddrLValue( 4477 CGF.CreateMemTemp(C.getSizeType(), "affs.counter.addr"), 4478 C.getSizeType()); 4479 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Pos), PosLVal); 4480 } 4481 // Process elements with iterators. 4482 for (const auto *C : D.getClausesOfKind<OMPAffinityClause>()) { 4483 const Expr *Modifier = C->getModifier(); 4484 if (!Modifier) 4485 continue; 4486 OMPIteratorGeneratorScope IteratorScope( 4487 CGF, cast_or_null<OMPIteratorExpr>(Modifier->IgnoreParenImpCasts())); 4488 for (const Expr *E : C->varlists()) { 4489 llvm::Value *Addr; 4490 llvm::Value *Size; 4491 std::tie(Addr, Size) = getPointerAndSize(CGF, E); 4492 llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc()); 4493 LValue Base = CGF.MakeAddrLValue( 4494 CGF.Builder.CreateGEP(AffinitiesArray, Idx), KmpTaskAffinityInfoTy); 4495 // affs[i].base_addr = &<Affinities[i].second>; 4496 LValue BaseAddrLVal = CGF.EmitLValueForField( 4497 Base, *std::next(KmpAffinityInfoRD->field_begin(), BaseAddr)); 4498 CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy), 4499 BaseAddrLVal); 4500 // affs[i].len = sizeof(<Affinities[i].second>); 4501 LValue LenLVal = CGF.EmitLValueForField( 4502 Base, *std::next(KmpAffinityInfoRD->field_begin(), Len)); 4503 CGF.EmitStoreOfScalar(Size, LenLVal); 4504 Idx = CGF.Builder.CreateNUWAdd( 4505 Idx, llvm::ConstantInt::get(Idx->getType(), 1)); 4506 CGF.EmitStoreOfScalar(Idx, PosLVal); 4507 } 4508 } 4509 // Call to kmp_int32 __kmpc_omp_reg_task_with_affinity(ident_t *loc_ref, 4510 // kmp_int32 gtid, kmp_task_t *new_task, kmp_int32 4511 // naffins, kmp_task_affinity_info_t *affin_list); 4512 llvm::Value *LocRef = emitUpdateLocation(CGF, Loc); 4513 llvm::Value *GTid = getThreadID(CGF, Loc); 4514 llvm::Value *AffinListPtr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4515 AffinitiesArray.getPointer(), CGM.VoidPtrTy); 4516 // FIXME: Emit the function and ignore its result for now unless the 4517 // runtime function is properly implemented. 4518 (void)CGF.EmitRuntimeCall( 4519 OMPBuilder.getOrCreateRuntimeFunction( 4520 CGM.getModule(), OMPRTL___kmpc_omp_reg_task_with_affinity), 4521 {LocRef, GTid, NewTask, NumOfElements, AffinListPtr}); 4522 } 4523 llvm::Value *NewTaskNewTaskTTy = 4524 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4525 NewTask, KmpTaskTWithPrivatesPtrTy); 4526 LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy, 4527 KmpTaskTWithPrivatesQTy); 4528 LValue TDBase = 4529 CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4530 // Fill the data in the resulting kmp_task_t record. 4531 // Copy shareds if there are any. 4532 Address KmpTaskSharedsPtr = Address::invalid(); 4533 if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) { 4534 KmpTaskSharedsPtr = Address::deprecated( 4535 CGF.EmitLoadOfScalar( 4536 CGF.EmitLValueForField( 4537 TDBase, 4538 *std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds)), 4539 Loc), 4540 CGM.getNaturalTypeAlignment(SharedsTy)); 4541 LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy); 4542 LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy); 4543 CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap); 4544 } 4545 // Emit initial values for private copies (if any). 4546 TaskResultTy Result; 4547 if (!Privates.empty()) { 4548 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD, 4549 SharedsTy, SharedsPtrTy, Data, Privates, 4550 /*ForDup=*/false); 4551 if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) && 4552 (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) { 4553 Result.TaskDupFn = emitTaskDupFunction( 4554 CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD, 4555 KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates, 4556 /*WithLastIter=*/!Data.LastprivateVars.empty()); 4557 } 4558 } 4559 // Fields of union "kmp_cmplrdata_t" for destructors and priority. 4560 enum { Priority = 0, Destructors = 1 }; 4561 // Provide pointer to function with destructors for privates. 4562 auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1); 4563 const RecordDecl *KmpCmplrdataUD = 4564 (*FI)->getType()->getAsUnionType()->getDecl(); 4565 if (NeedsCleanup) { 4566 llvm::Value *DestructorFn = emitDestructorsFunction( 4567 CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 4568 KmpTaskTWithPrivatesQTy); 4569 LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI); 4570 LValue DestructorsLV = CGF.EmitLValueForField( 4571 Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors)); 4572 CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4573 DestructorFn, KmpRoutineEntryPtrTy), 4574 DestructorsLV); 4575 } 4576 // Set priority. 4577 if (Data.Priority.getInt()) { 4578 LValue Data2LV = CGF.EmitLValueForField( 4579 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2)); 4580 LValue PriorityLV = CGF.EmitLValueForField( 4581 Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority)); 4582 CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV); 4583 } 4584 Result.NewTask = NewTask; 4585 Result.TaskEntry = TaskEntry; 4586 Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy; 4587 Result.TDBase = TDBase; 4588 Result.KmpTaskTQTyRD = KmpTaskTQTyRD; 4589 return Result; 4590 } 4591 4592 namespace { 4593 /// Dependence kind for RTL. 4594 enum RTLDependenceKindTy { 4595 DepIn = 0x01, 4596 DepInOut = 0x3, 4597 DepMutexInOutSet = 0x4, 4598 DepInOutSet = 0x8 4599 }; 4600 /// Fields ids in kmp_depend_info record. 4601 enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags }; 4602 } // namespace 4603 4604 /// Translates internal dependency kind into the runtime kind. 4605 static RTLDependenceKindTy translateDependencyKind(OpenMPDependClauseKind K) { 4606 RTLDependenceKindTy DepKind; 4607 switch (K) { 4608 case OMPC_DEPEND_in: 4609 DepKind = DepIn; 4610 break; 4611 // Out and InOut dependencies must use the same code. 4612 case OMPC_DEPEND_out: 4613 case OMPC_DEPEND_inout: 4614 DepKind = DepInOut; 4615 break; 4616 case OMPC_DEPEND_mutexinoutset: 4617 DepKind = DepMutexInOutSet; 4618 break; 4619 case OMPC_DEPEND_inoutset: 4620 DepKind = DepInOutSet; 4621 break; 4622 case OMPC_DEPEND_source: 4623 case OMPC_DEPEND_sink: 4624 case OMPC_DEPEND_depobj: 4625 case OMPC_DEPEND_unknown: 4626 llvm_unreachable("Unknown task dependence type"); 4627 } 4628 return DepKind; 4629 } 4630 4631 /// Builds kmp_depend_info, if it is not built yet, and builds flags type. 4632 static void getDependTypes(ASTContext &C, QualType &KmpDependInfoTy, 4633 QualType &FlagsTy) { 4634 FlagsTy = C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false); 4635 if (KmpDependInfoTy.isNull()) { 4636 RecordDecl *KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info"); 4637 KmpDependInfoRD->startDefinition(); 4638 addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType()); 4639 addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType()); 4640 addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy); 4641 KmpDependInfoRD->completeDefinition(); 4642 KmpDependInfoTy = C.getRecordType(KmpDependInfoRD); 4643 } 4644 } 4645 4646 std::pair<llvm::Value *, LValue> 4647 CGOpenMPRuntime::getDepobjElements(CodeGenFunction &CGF, LValue DepobjLVal, 4648 SourceLocation Loc) { 4649 ASTContext &C = CGM.getContext(); 4650 QualType FlagsTy; 4651 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4652 RecordDecl *KmpDependInfoRD = 4653 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 4654 LValue Base = CGF.EmitLoadOfPointerLValue( 4655 DepobjLVal.getAddress(CGF), C.VoidPtrTy.castAs<PointerType>()); 4656 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 4657 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4658 Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy), 4659 CGF.ConvertTypeForMem(KmpDependInfoTy)); 4660 Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(), 4661 Base.getTBAAInfo()); 4662 Address DepObjAddr = CGF.Builder.CreateGEP( 4663 Addr, llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 4664 LValue NumDepsBase = CGF.MakeAddrLValue( 4665 DepObjAddr, KmpDependInfoTy, Base.getBaseInfo(), Base.getTBAAInfo()); 4666 // NumDeps = deps[i].base_addr; 4667 LValue BaseAddrLVal = CGF.EmitLValueForField( 4668 NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 4669 llvm::Value *NumDeps = CGF.EmitLoadOfScalar(BaseAddrLVal, Loc); 4670 return std::make_pair(NumDeps, Base); 4671 } 4672 4673 static void emitDependData(CodeGenFunction &CGF, QualType &KmpDependInfoTy, 4674 llvm::PointerUnion<unsigned *, LValue *> Pos, 4675 const OMPTaskDataTy::DependData &Data, 4676 Address DependenciesArray) { 4677 CodeGenModule &CGM = CGF.CGM; 4678 ASTContext &C = CGM.getContext(); 4679 QualType FlagsTy; 4680 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4681 RecordDecl *KmpDependInfoRD = 4682 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 4683 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 4684 4685 OMPIteratorGeneratorScope IteratorScope( 4686 CGF, cast_or_null<OMPIteratorExpr>( 4687 Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts() 4688 : nullptr)); 4689 for (const Expr *E : Data.DepExprs) { 4690 llvm::Value *Addr; 4691 llvm::Value *Size; 4692 std::tie(Addr, Size) = getPointerAndSize(CGF, E); 4693 LValue Base; 4694 if (unsigned *P = Pos.dyn_cast<unsigned *>()) { 4695 Base = CGF.MakeAddrLValue( 4696 CGF.Builder.CreateConstGEP(DependenciesArray, *P), KmpDependInfoTy); 4697 } else { 4698 LValue &PosLVal = *Pos.get<LValue *>(); 4699 llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc()); 4700 Base = CGF.MakeAddrLValue( 4701 CGF.Builder.CreateGEP(DependenciesArray, Idx), KmpDependInfoTy); 4702 } 4703 // deps[i].base_addr = &<Dependencies[i].second>; 4704 LValue BaseAddrLVal = CGF.EmitLValueForField( 4705 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 4706 CGF.EmitStoreOfScalar(CGF.Builder.CreatePtrToInt(Addr, CGF.IntPtrTy), 4707 BaseAddrLVal); 4708 // deps[i].len = sizeof(<Dependencies[i].second>); 4709 LValue LenLVal = CGF.EmitLValueForField( 4710 Base, *std::next(KmpDependInfoRD->field_begin(), Len)); 4711 CGF.EmitStoreOfScalar(Size, LenLVal); 4712 // deps[i].flags = <Dependencies[i].first>; 4713 RTLDependenceKindTy DepKind = translateDependencyKind(Data.DepKind); 4714 LValue FlagsLVal = CGF.EmitLValueForField( 4715 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 4716 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 4717 FlagsLVal); 4718 if (unsigned *P = Pos.dyn_cast<unsigned *>()) { 4719 ++(*P); 4720 } else { 4721 LValue &PosLVal = *Pos.get<LValue *>(); 4722 llvm::Value *Idx = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc()); 4723 Idx = CGF.Builder.CreateNUWAdd(Idx, 4724 llvm::ConstantInt::get(Idx->getType(), 1)); 4725 CGF.EmitStoreOfScalar(Idx, PosLVal); 4726 } 4727 } 4728 } 4729 4730 static SmallVector<llvm::Value *, 4> 4731 emitDepobjElementsSizes(CodeGenFunction &CGF, QualType &KmpDependInfoTy, 4732 const OMPTaskDataTy::DependData &Data) { 4733 assert(Data.DepKind == OMPC_DEPEND_depobj && 4734 "Expected depobj dependecy kind."); 4735 SmallVector<llvm::Value *, 4> Sizes; 4736 SmallVector<LValue, 4> SizeLVals; 4737 ASTContext &C = CGF.getContext(); 4738 QualType FlagsTy; 4739 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4740 RecordDecl *KmpDependInfoRD = 4741 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 4742 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 4743 llvm::Type *KmpDependInfoPtrT = CGF.ConvertTypeForMem(KmpDependInfoPtrTy); 4744 { 4745 OMPIteratorGeneratorScope IteratorScope( 4746 CGF, cast_or_null<OMPIteratorExpr>( 4747 Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts() 4748 : nullptr)); 4749 for (const Expr *E : Data.DepExprs) { 4750 LValue DepobjLVal = CGF.EmitLValue(E->IgnoreParenImpCasts()); 4751 LValue Base = CGF.EmitLoadOfPointerLValue( 4752 DepobjLVal.getAddress(CGF), C.VoidPtrTy.castAs<PointerType>()); 4753 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4754 Base.getAddress(CGF), KmpDependInfoPtrT, 4755 CGF.ConvertTypeForMem(KmpDependInfoTy)); 4756 Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(), 4757 Base.getTBAAInfo()); 4758 Address DepObjAddr = CGF.Builder.CreateGEP( 4759 Addr, llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 4760 LValue NumDepsBase = CGF.MakeAddrLValue( 4761 DepObjAddr, KmpDependInfoTy, Base.getBaseInfo(), Base.getTBAAInfo()); 4762 // NumDeps = deps[i].base_addr; 4763 LValue BaseAddrLVal = CGF.EmitLValueForField( 4764 NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 4765 llvm::Value *NumDeps = 4766 CGF.EmitLoadOfScalar(BaseAddrLVal, E->getExprLoc()); 4767 LValue NumLVal = CGF.MakeAddrLValue( 4768 CGF.CreateMemTemp(C.getUIntPtrType(), "depobj.size.addr"), 4769 C.getUIntPtrType()); 4770 CGF.Builder.CreateStore(llvm::ConstantInt::get(CGF.IntPtrTy, 0), 4771 NumLVal.getAddress(CGF)); 4772 llvm::Value *PrevVal = CGF.EmitLoadOfScalar(NumLVal, E->getExprLoc()); 4773 llvm::Value *Add = CGF.Builder.CreateNUWAdd(PrevVal, NumDeps); 4774 CGF.EmitStoreOfScalar(Add, NumLVal); 4775 SizeLVals.push_back(NumLVal); 4776 } 4777 } 4778 for (unsigned I = 0, E = SizeLVals.size(); I < E; ++I) { 4779 llvm::Value *Size = 4780 CGF.EmitLoadOfScalar(SizeLVals[I], Data.DepExprs[I]->getExprLoc()); 4781 Sizes.push_back(Size); 4782 } 4783 return Sizes; 4784 } 4785 4786 static void emitDepobjElements(CodeGenFunction &CGF, QualType &KmpDependInfoTy, 4787 LValue PosLVal, 4788 const OMPTaskDataTy::DependData &Data, 4789 Address DependenciesArray) { 4790 assert(Data.DepKind == OMPC_DEPEND_depobj && 4791 "Expected depobj dependecy kind."); 4792 ASTContext &C = CGF.getContext(); 4793 QualType FlagsTy; 4794 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4795 RecordDecl *KmpDependInfoRD = 4796 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 4797 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 4798 llvm::Type *KmpDependInfoPtrT = CGF.ConvertTypeForMem(KmpDependInfoPtrTy); 4799 llvm::Value *ElSize = CGF.getTypeSize(KmpDependInfoTy); 4800 { 4801 OMPIteratorGeneratorScope IteratorScope( 4802 CGF, cast_or_null<OMPIteratorExpr>( 4803 Data.IteratorExpr ? Data.IteratorExpr->IgnoreParenImpCasts() 4804 : nullptr)); 4805 for (unsigned I = 0, End = Data.DepExprs.size(); I < End; ++I) { 4806 const Expr *E = Data.DepExprs[I]; 4807 LValue DepobjLVal = CGF.EmitLValue(E->IgnoreParenImpCasts()); 4808 LValue Base = CGF.EmitLoadOfPointerLValue( 4809 DepobjLVal.getAddress(CGF), C.VoidPtrTy.castAs<PointerType>()); 4810 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4811 Base.getAddress(CGF), KmpDependInfoPtrT, 4812 CGF.ConvertTypeForMem(KmpDependInfoTy)); 4813 Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(), 4814 Base.getTBAAInfo()); 4815 4816 // Get number of elements in a single depobj. 4817 Address DepObjAddr = CGF.Builder.CreateGEP( 4818 Addr, llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 4819 LValue NumDepsBase = CGF.MakeAddrLValue( 4820 DepObjAddr, KmpDependInfoTy, Base.getBaseInfo(), Base.getTBAAInfo()); 4821 // NumDeps = deps[i].base_addr; 4822 LValue BaseAddrLVal = CGF.EmitLValueForField( 4823 NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 4824 llvm::Value *NumDeps = 4825 CGF.EmitLoadOfScalar(BaseAddrLVal, E->getExprLoc()); 4826 4827 // memcopy dependency data. 4828 llvm::Value *Size = CGF.Builder.CreateNUWMul( 4829 ElSize, 4830 CGF.Builder.CreateIntCast(NumDeps, CGF.SizeTy, /*isSigned=*/false)); 4831 llvm::Value *Pos = CGF.EmitLoadOfScalar(PosLVal, E->getExprLoc()); 4832 Address DepAddr = CGF.Builder.CreateGEP(DependenciesArray, Pos); 4833 CGF.Builder.CreateMemCpy(DepAddr, Base.getAddress(CGF), Size); 4834 4835 // Increase pos. 4836 // pos += size; 4837 llvm::Value *Add = CGF.Builder.CreateNUWAdd(Pos, NumDeps); 4838 CGF.EmitStoreOfScalar(Add, PosLVal); 4839 } 4840 } 4841 } 4842 4843 std::pair<llvm::Value *, Address> CGOpenMPRuntime::emitDependClause( 4844 CodeGenFunction &CGF, ArrayRef<OMPTaskDataTy::DependData> Dependencies, 4845 SourceLocation Loc) { 4846 if (llvm::all_of(Dependencies, [](const OMPTaskDataTy::DependData &D) { 4847 return D.DepExprs.empty(); 4848 })) 4849 return std::make_pair(nullptr, Address::invalid()); 4850 // Process list of dependencies. 4851 ASTContext &C = CGM.getContext(); 4852 Address DependenciesArray = Address::invalid(); 4853 llvm::Value *NumOfElements = nullptr; 4854 unsigned NumDependencies = std::accumulate( 4855 Dependencies.begin(), Dependencies.end(), 0, 4856 [](unsigned V, const OMPTaskDataTy::DependData &D) { 4857 return D.DepKind == OMPC_DEPEND_depobj 4858 ? V 4859 : (V + (D.IteratorExpr ? 0 : D.DepExprs.size())); 4860 }); 4861 QualType FlagsTy; 4862 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4863 bool HasDepobjDeps = false; 4864 bool HasRegularWithIterators = false; 4865 llvm::Value *NumOfDepobjElements = llvm::ConstantInt::get(CGF.IntPtrTy, 0); 4866 llvm::Value *NumOfRegularWithIterators = 4867 llvm::ConstantInt::get(CGF.IntPtrTy, 0); 4868 // Calculate number of depobj dependecies and regular deps with the iterators. 4869 for (const OMPTaskDataTy::DependData &D : Dependencies) { 4870 if (D.DepKind == OMPC_DEPEND_depobj) { 4871 SmallVector<llvm::Value *, 4> Sizes = 4872 emitDepobjElementsSizes(CGF, KmpDependInfoTy, D); 4873 for (llvm::Value *Size : Sizes) { 4874 NumOfDepobjElements = 4875 CGF.Builder.CreateNUWAdd(NumOfDepobjElements, Size); 4876 } 4877 HasDepobjDeps = true; 4878 continue; 4879 } 4880 // Include number of iterations, if any. 4881 4882 if (const auto *IE = cast_or_null<OMPIteratorExpr>(D.IteratorExpr)) { 4883 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) { 4884 llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper); 4885 Sz = CGF.Builder.CreateIntCast(Sz, CGF.IntPtrTy, /*isSigned=*/false); 4886 llvm::Value *NumClauseDeps = CGF.Builder.CreateNUWMul( 4887 Sz, llvm::ConstantInt::get(CGF.IntPtrTy, D.DepExprs.size())); 4888 NumOfRegularWithIterators = 4889 CGF.Builder.CreateNUWAdd(NumOfRegularWithIterators, NumClauseDeps); 4890 } 4891 HasRegularWithIterators = true; 4892 continue; 4893 } 4894 } 4895 4896 QualType KmpDependInfoArrayTy; 4897 if (HasDepobjDeps || HasRegularWithIterators) { 4898 NumOfElements = llvm::ConstantInt::get(CGM.IntPtrTy, NumDependencies, 4899 /*isSigned=*/false); 4900 if (HasDepobjDeps) { 4901 NumOfElements = 4902 CGF.Builder.CreateNUWAdd(NumOfDepobjElements, NumOfElements); 4903 } 4904 if (HasRegularWithIterators) { 4905 NumOfElements = 4906 CGF.Builder.CreateNUWAdd(NumOfRegularWithIterators, NumOfElements); 4907 } 4908 auto *OVE = new (C) OpaqueValueExpr( 4909 Loc, C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0), 4910 VK_PRValue); 4911 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, OVE, 4912 RValue::get(NumOfElements)); 4913 KmpDependInfoArrayTy = 4914 C.getVariableArrayType(KmpDependInfoTy, OVE, ArrayType::Normal, 4915 /*IndexTypeQuals=*/0, SourceRange(Loc, Loc)); 4916 // CGF.EmitVariablyModifiedType(KmpDependInfoArrayTy); 4917 // Properly emit variable-sized array. 4918 auto *PD = ImplicitParamDecl::Create(C, KmpDependInfoArrayTy, 4919 ImplicitParamDecl::Other); 4920 CGF.EmitVarDecl(*PD); 4921 DependenciesArray = CGF.GetAddrOfLocalVar(PD); 4922 NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty, 4923 /*isSigned=*/false); 4924 } else { 4925 KmpDependInfoArrayTy = C.getConstantArrayType( 4926 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies), nullptr, 4927 ArrayType::Normal, /*IndexTypeQuals=*/0); 4928 DependenciesArray = 4929 CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr"); 4930 DependenciesArray = CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0); 4931 NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies, 4932 /*isSigned=*/false); 4933 } 4934 unsigned Pos = 0; 4935 for (unsigned I = 0, End = Dependencies.size(); I < End; ++I) { 4936 if (Dependencies[I].DepKind == OMPC_DEPEND_depobj || 4937 Dependencies[I].IteratorExpr) 4938 continue; 4939 emitDependData(CGF, KmpDependInfoTy, &Pos, Dependencies[I], 4940 DependenciesArray); 4941 } 4942 // Copy regular dependecies with iterators. 4943 LValue PosLVal = CGF.MakeAddrLValue( 4944 CGF.CreateMemTemp(C.getSizeType(), "dep.counter.addr"), C.getSizeType()); 4945 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Pos), PosLVal); 4946 for (unsigned I = 0, End = Dependencies.size(); I < End; ++I) { 4947 if (Dependencies[I].DepKind == OMPC_DEPEND_depobj || 4948 !Dependencies[I].IteratorExpr) 4949 continue; 4950 emitDependData(CGF, KmpDependInfoTy, &PosLVal, Dependencies[I], 4951 DependenciesArray); 4952 } 4953 // Copy final depobj arrays without iterators. 4954 if (HasDepobjDeps) { 4955 for (unsigned I = 0, End = Dependencies.size(); I < End; ++I) { 4956 if (Dependencies[I].DepKind != OMPC_DEPEND_depobj) 4957 continue; 4958 emitDepobjElements(CGF, KmpDependInfoTy, PosLVal, Dependencies[I], 4959 DependenciesArray); 4960 } 4961 } 4962 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4963 DependenciesArray, CGF.VoidPtrTy, CGF.Int8Ty); 4964 return std::make_pair(NumOfElements, DependenciesArray); 4965 } 4966 4967 Address CGOpenMPRuntime::emitDepobjDependClause( 4968 CodeGenFunction &CGF, const OMPTaskDataTy::DependData &Dependencies, 4969 SourceLocation Loc) { 4970 if (Dependencies.DepExprs.empty()) 4971 return Address::invalid(); 4972 // Process list of dependencies. 4973 ASTContext &C = CGM.getContext(); 4974 Address DependenciesArray = Address::invalid(); 4975 unsigned NumDependencies = Dependencies.DepExprs.size(); 4976 QualType FlagsTy; 4977 getDependTypes(C, KmpDependInfoTy, FlagsTy); 4978 RecordDecl *KmpDependInfoRD = 4979 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 4980 4981 llvm::Value *Size; 4982 // Define type kmp_depend_info[<Dependencies.size()>]; 4983 // For depobj reserve one extra element to store the number of elements. 4984 // It is required to handle depobj(x) update(in) construct. 4985 // kmp_depend_info[<Dependencies.size()>] deps; 4986 llvm::Value *NumDepsVal; 4987 CharUnits Align = C.getTypeAlignInChars(KmpDependInfoTy); 4988 if (const auto *IE = 4989 cast_or_null<OMPIteratorExpr>(Dependencies.IteratorExpr)) { 4990 NumDepsVal = llvm::ConstantInt::get(CGF.SizeTy, 1); 4991 for (unsigned I = 0, E = IE->numOfIterators(); I < E; ++I) { 4992 llvm::Value *Sz = CGF.EmitScalarExpr(IE->getHelper(I).Upper); 4993 Sz = CGF.Builder.CreateIntCast(Sz, CGF.SizeTy, /*isSigned=*/false); 4994 NumDepsVal = CGF.Builder.CreateNUWMul(NumDepsVal, Sz); 4995 } 4996 Size = CGF.Builder.CreateNUWAdd(llvm::ConstantInt::get(CGF.SizeTy, 1), 4997 NumDepsVal); 4998 CharUnits SizeInBytes = 4999 C.getTypeSizeInChars(KmpDependInfoTy).alignTo(Align); 5000 llvm::Value *RecSize = CGM.getSize(SizeInBytes); 5001 Size = CGF.Builder.CreateNUWMul(Size, RecSize); 5002 NumDepsVal = 5003 CGF.Builder.CreateIntCast(NumDepsVal, CGF.IntPtrTy, /*isSigned=*/false); 5004 } else { 5005 QualType KmpDependInfoArrayTy = C.getConstantArrayType( 5006 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies + 1), 5007 nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 5008 CharUnits Sz = C.getTypeSizeInChars(KmpDependInfoArrayTy); 5009 Size = CGM.getSize(Sz.alignTo(Align)); 5010 NumDepsVal = llvm::ConstantInt::get(CGF.IntPtrTy, NumDependencies); 5011 } 5012 // Need to allocate on the dynamic memory. 5013 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5014 // Use default allocator. 5015 llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5016 llvm::Value *Args[] = {ThreadID, Size, Allocator}; 5017 5018 llvm::Value *Addr = 5019 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 5020 CGM.getModule(), OMPRTL___kmpc_alloc), 5021 Args, ".dep.arr.addr"); 5022 llvm::Type *KmpDependInfoLlvmTy = CGF.ConvertTypeForMem(KmpDependInfoTy); 5023 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5024 Addr, KmpDependInfoLlvmTy->getPointerTo()); 5025 DependenciesArray = Address(Addr, KmpDependInfoLlvmTy, Align); 5026 // Write number of elements in the first element of array for depobj. 5027 LValue Base = CGF.MakeAddrLValue(DependenciesArray, KmpDependInfoTy); 5028 // deps[i].base_addr = NumDependencies; 5029 LValue BaseAddrLVal = CGF.EmitLValueForField( 5030 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 5031 CGF.EmitStoreOfScalar(NumDepsVal, BaseAddrLVal); 5032 llvm::PointerUnion<unsigned *, LValue *> Pos; 5033 unsigned Idx = 1; 5034 LValue PosLVal; 5035 if (Dependencies.IteratorExpr) { 5036 PosLVal = CGF.MakeAddrLValue( 5037 CGF.CreateMemTemp(C.getSizeType(), "iterator.counter.addr"), 5038 C.getSizeType()); 5039 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(CGF.SizeTy, Idx), PosLVal, 5040 /*IsInit=*/true); 5041 Pos = &PosLVal; 5042 } else { 5043 Pos = &Idx; 5044 } 5045 emitDependData(CGF, KmpDependInfoTy, Pos, Dependencies, DependenciesArray); 5046 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5047 CGF.Builder.CreateConstGEP(DependenciesArray, 1), CGF.VoidPtrTy, 5048 CGF.Int8Ty); 5049 return DependenciesArray; 5050 } 5051 5052 void CGOpenMPRuntime::emitDestroyClause(CodeGenFunction &CGF, LValue DepobjLVal, 5053 SourceLocation Loc) { 5054 ASTContext &C = CGM.getContext(); 5055 QualType FlagsTy; 5056 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5057 LValue Base = CGF.EmitLoadOfPointerLValue( 5058 DepobjLVal.getAddress(CGF), C.VoidPtrTy.castAs<PointerType>()); 5059 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 5060 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5061 Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy), 5062 CGF.ConvertTypeForMem(KmpDependInfoTy)); 5063 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP( 5064 Addr.getElementType(), Addr.getPointer(), 5065 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 5066 DepObjAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(DepObjAddr, 5067 CGF.VoidPtrTy); 5068 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5069 // Use default allocator. 5070 llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5071 llvm::Value *Args[] = {ThreadID, DepObjAddr, Allocator}; 5072 5073 // _kmpc_free(gtid, addr, nullptr); 5074 (void)CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 5075 CGM.getModule(), OMPRTL___kmpc_free), 5076 Args); 5077 } 5078 5079 void CGOpenMPRuntime::emitUpdateClause(CodeGenFunction &CGF, LValue DepobjLVal, 5080 OpenMPDependClauseKind NewDepKind, 5081 SourceLocation Loc) { 5082 ASTContext &C = CGM.getContext(); 5083 QualType FlagsTy; 5084 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5085 RecordDecl *KmpDependInfoRD = 5086 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5087 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 5088 llvm::Value *NumDeps; 5089 LValue Base; 5090 std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc); 5091 5092 Address Begin = Base.getAddress(CGF); 5093 // Cast from pointer to array type to pointer to single element. 5094 llvm::Value *End = CGF.Builder.CreateGEP( 5095 Begin.getElementType(), Begin.getPointer(), NumDeps); 5096 // The basic structure here is a while-do loop. 5097 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.body"); 5098 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.done"); 5099 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5100 CGF.EmitBlock(BodyBB); 5101 llvm::PHINode *ElementPHI = 5102 CGF.Builder.CreatePHI(Begin.getType(), 2, "omp.elementPast"); 5103 ElementPHI->addIncoming(Begin.getPointer(), EntryBB); 5104 Begin = Begin.withPointer(ElementPHI); 5105 Base = CGF.MakeAddrLValue(Begin, KmpDependInfoTy, Base.getBaseInfo(), 5106 Base.getTBAAInfo()); 5107 // deps[i].flags = NewDepKind; 5108 RTLDependenceKindTy DepKind = translateDependencyKind(NewDepKind); 5109 LValue FlagsLVal = CGF.EmitLValueForField( 5110 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 5111 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 5112 FlagsLVal); 5113 5114 // Shift the address forward by one element. 5115 Address ElementNext = 5116 CGF.Builder.CreateConstGEP(Begin, /*Index=*/1, "omp.elementNext"); 5117 ElementPHI->addIncoming(ElementNext.getPointer(), 5118 CGF.Builder.GetInsertBlock()); 5119 llvm::Value *IsEmpty = 5120 CGF.Builder.CreateICmpEQ(ElementNext.getPointer(), End, "omp.isempty"); 5121 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5122 // Done. 5123 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5124 } 5125 5126 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 5127 const OMPExecutableDirective &D, 5128 llvm::Function *TaskFunction, 5129 QualType SharedsTy, Address Shareds, 5130 const Expr *IfCond, 5131 const OMPTaskDataTy &Data) { 5132 if (!CGF.HaveInsertPoint()) 5133 return; 5134 5135 TaskResultTy Result = 5136 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5137 llvm::Value *NewTask = Result.NewTask; 5138 llvm::Function *TaskEntry = Result.TaskEntry; 5139 llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy; 5140 LValue TDBase = Result.TDBase; 5141 const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD; 5142 // Process list of dependences. 5143 Address DependenciesArray = Address::invalid(); 5144 llvm::Value *NumOfElements; 5145 std::tie(NumOfElements, DependenciesArray) = 5146 emitDependClause(CGF, Data.Dependences, Loc); 5147 5148 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5149 // libcall. 5150 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 5151 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 5152 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence 5153 // list is not empty 5154 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5155 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5156 llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask }; 5157 llvm::Value *DepTaskArgs[7]; 5158 if (!Data.Dependences.empty()) { 5159 DepTaskArgs[0] = UpLoc; 5160 DepTaskArgs[1] = ThreadID; 5161 DepTaskArgs[2] = NewTask; 5162 DepTaskArgs[3] = NumOfElements; 5163 DepTaskArgs[4] = DependenciesArray.getPointer(); 5164 DepTaskArgs[5] = CGF.Builder.getInt32(0); 5165 DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5166 } 5167 auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, &TaskArgs, 5168 &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) { 5169 if (!Data.Tied) { 5170 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 5171 LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI); 5172 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal); 5173 } 5174 if (!Data.Dependences.empty()) { 5175 CGF.EmitRuntimeCall( 5176 OMPBuilder.getOrCreateRuntimeFunction( 5177 CGM.getModule(), OMPRTL___kmpc_omp_task_with_deps), 5178 DepTaskArgs); 5179 } else { 5180 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 5181 CGM.getModule(), OMPRTL___kmpc_omp_task), 5182 TaskArgs); 5183 } 5184 // Check if parent region is untied and build return for untied task; 5185 if (auto *Region = 5186 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 5187 Region->emitUntiedSwitch(CGF); 5188 }; 5189 5190 llvm::Value *DepWaitTaskArgs[6]; 5191 if (!Data.Dependences.empty()) { 5192 DepWaitTaskArgs[0] = UpLoc; 5193 DepWaitTaskArgs[1] = ThreadID; 5194 DepWaitTaskArgs[2] = NumOfElements; 5195 DepWaitTaskArgs[3] = DependenciesArray.getPointer(); 5196 DepWaitTaskArgs[4] = CGF.Builder.getInt32(0); 5197 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5198 } 5199 auto &M = CGM.getModule(); 5200 auto &&ElseCodeGen = [this, &M, &TaskArgs, ThreadID, NewTaskNewTaskTTy, 5201 TaskEntry, &Data, &DepWaitTaskArgs, 5202 Loc](CodeGenFunction &CGF, PrePostActionTy &) { 5203 CodeGenFunction::RunCleanupsScope LocalScope(CGF); 5204 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 5205 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 5206 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info 5207 // is specified. 5208 if (!Data.Dependences.empty()) 5209 CGF.EmitRuntimeCall( 5210 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_omp_wait_deps), 5211 DepWaitTaskArgs); 5212 // Call proxy_task_entry(gtid, new_task); 5213 auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy, 5214 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 5215 Action.Enter(CGF); 5216 llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy}; 5217 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry, 5218 OutlinedFnArgs); 5219 }; 5220 5221 // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 5222 // kmp_task_t *new_task); 5223 // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 5224 // kmp_task_t *new_task); 5225 RegionCodeGenTy RCG(CodeGen); 5226 CommonActionTy Action(OMPBuilder.getOrCreateRuntimeFunction( 5227 M, OMPRTL___kmpc_omp_task_begin_if0), 5228 TaskArgs, 5229 OMPBuilder.getOrCreateRuntimeFunction( 5230 M, OMPRTL___kmpc_omp_task_complete_if0), 5231 TaskArgs); 5232 RCG.setAction(Action); 5233 RCG(CGF); 5234 }; 5235 5236 if (IfCond) { 5237 emitIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen); 5238 } else { 5239 RegionCodeGenTy ThenRCG(ThenCodeGen); 5240 ThenRCG(CGF); 5241 } 5242 } 5243 5244 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc, 5245 const OMPLoopDirective &D, 5246 llvm::Function *TaskFunction, 5247 QualType SharedsTy, Address Shareds, 5248 const Expr *IfCond, 5249 const OMPTaskDataTy &Data) { 5250 if (!CGF.HaveInsertPoint()) 5251 return; 5252 TaskResultTy Result = 5253 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5254 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5255 // libcall. 5256 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 5257 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 5258 // sched, kmp_uint64 grainsize, void *task_dup); 5259 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5260 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5261 llvm::Value *IfVal; 5262 if (IfCond) { 5263 IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy, 5264 /*isSigned=*/true); 5265 } else { 5266 IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1); 5267 } 5268 5269 LValue LBLVal = CGF.EmitLValueForField( 5270 Result.TDBase, 5271 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound)); 5272 const auto *LBVar = 5273 cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl()); 5274 CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(CGF), 5275 LBLVal.getQuals(), 5276 /*IsInitializer=*/true); 5277 LValue UBLVal = CGF.EmitLValueForField( 5278 Result.TDBase, 5279 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound)); 5280 const auto *UBVar = 5281 cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl()); 5282 CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(CGF), 5283 UBLVal.getQuals(), 5284 /*IsInitializer=*/true); 5285 LValue StLVal = CGF.EmitLValueForField( 5286 Result.TDBase, 5287 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride)); 5288 const auto *StVar = 5289 cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl()); 5290 CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(CGF), 5291 StLVal.getQuals(), 5292 /*IsInitializer=*/true); 5293 // Store reductions address. 5294 LValue RedLVal = CGF.EmitLValueForField( 5295 Result.TDBase, 5296 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions)); 5297 if (Data.Reductions) { 5298 CGF.EmitStoreOfScalar(Data.Reductions, RedLVal); 5299 } else { 5300 CGF.EmitNullInitialization(RedLVal.getAddress(CGF), 5301 CGF.getContext().VoidPtrTy); 5302 } 5303 enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 }; 5304 llvm::Value *TaskArgs[] = { 5305 UpLoc, 5306 ThreadID, 5307 Result.NewTask, 5308 IfVal, 5309 LBLVal.getPointer(CGF), 5310 UBLVal.getPointer(CGF), 5311 CGF.EmitLoadOfScalar(StLVal, Loc), 5312 llvm::ConstantInt::getSigned( 5313 CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler 5314 llvm::ConstantInt::getSigned( 5315 CGF.IntTy, Data.Schedule.getPointer() 5316 ? Data.Schedule.getInt() ? NumTasks : Grainsize 5317 : NoSchedule), 5318 Data.Schedule.getPointer() 5319 ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty, 5320 /*isSigned=*/false) 5321 : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0), 5322 Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5323 Result.TaskDupFn, CGF.VoidPtrTy) 5324 : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)}; 5325 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 5326 CGM.getModule(), OMPRTL___kmpc_taskloop), 5327 TaskArgs); 5328 } 5329 5330 /// Emit reduction operation for each element of array (required for 5331 /// array sections) LHS op = RHS. 5332 /// \param Type Type of array. 5333 /// \param LHSVar Variable on the left side of the reduction operation 5334 /// (references element of array in original variable). 5335 /// \param RHSVar Variable on the right side of the reduction operation 5336 /// (references element of array in original variable). 5337 /// \param RedOpGen Generator of reduction operation with use of LHSVar and 5338 /// RHSVar. 5339 static void EmitOMPAggregateReduction( 5340 CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar, 5341 const VarDecl *RHSVar, 5342 const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *, 5343 const Expr *, const Expr *)> &RedOpGen, 5344 const Expr *XExpr = nullptr, const Expr *EExpr = nullptr, 5345 const Expr *UpExpr = nullptr) { 5346 // Perform element-by-element initialization. 5347 QualType ElementTy; 5348 Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar); 5349 Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar); 5350 5351 // Drill down to the base element type on both arrays. 5352 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 5353 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr); 5354 5355 llvm::Value *RHSBegin = RHSAddr.getPointer(); 5356 llvm::Value *LHSBegin = LHSAddr.getPointer(); 5357 // Cast from pointer to array type to pointer to single element. 5358 llvm::Value *LHSEnd = 5359 CGF.Builder.CreateGEP(LHSAddr.getElementType(), LHSBegin, NumElements); 5360 // The basic structure here is a while-do loop. 5361 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body"); 5362 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done"); 5363 llvm::Value *IsEmpty = 5364 CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty"); 5365 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5366 5367 // Enter the loop body, making that address the current address. 5368 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5369 CGF.EmitBlock(BodyBB); 5370 5371 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 5372 5373 llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI( 5374 RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast"); 5375 RHSElementPHI->addIncoming(RHSBegin, EntryBB); 5376 Address RHSElementCurrent( 5377 RHSElementPHI, RHSAddr.getElementType(), 5378 RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5379 5380 llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI( 5381 LHSBegin->getType(), 2, "omp.arraycpy.destElementPast"); 5382 LHSElementPHI->addIncoming(LHSBegin, EntryBB); 5383 Address LHSElementCurrent( 5384 LHSElementPHI, LHSAddr.getElementType(), 5385 LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5386 5387 // Emit copy. 5388 CodeGenFunction::OMPPrivateScope Scope(CGF); 5389 Scope.addPrivate(LHSVar, LHSElementCurrent); 5390 Scope.addPrivate(RHSVar, RHSElementCurrent); 5391 Scope.Privatize(); 5392 RedOpGen(CGF, XExpr, EExpr, UpExpr); 5393 Scope.ForceCleanup(); 5394 5395 // Shift the address forward by one element. 5396 llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32( 5397 LHSAddr.getElementType(), LHSElementPHI, /*Idx0=*/1, 5398 "omp.arraycpy.dest.element"); 5399 llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32( 5400 RHSAddr.getElementType(), RHSElementPHI, /*Idx0=*/1, 5401 "omp.arraycpy.src.element"); 5402 // Check whether we've reached the end. 5403 llvm::Value *Done = 5404 CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done"); 5405 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 5406 LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock()); 5407 RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock()); 5408 5409 // Done. 5410 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5411 } 5412 5413 /// Emit reduction combiner. If the combiner is a simple expression emit it as 5414 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of 5415 /// UDR combiner function. 5416 static void emitReductionCombiner(CodeGenFunction &CGF, 5417 const Expr *ReductionOp) { 5418 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 5419 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 5420 if (const auto *DRE = 5421 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 5422 if (const auto *DRD = 5423 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) { 5424 std::pair<llvm::Function *, llvm::Function *> Reduction = 5425 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 5426 RValue Func = RValue::get(Reduction.first); 5427 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 5428 CGF.EmitIgnoredExpr(ReductionOp); 5429 return; 5430 } 5431 CGF.EmitIgnoredExpr(ReductionOp); 5432 } 5433 5434 llvm::Function *CGOpenMPRuntime::emitReductionFunction( 5435 SourceLocation Loc, llvm::Type *ArgsElemType, 5436 ArrayRef<const Expr *> Privates, ArrayRef<const Expr *> LHSExprs, 5437 ArrayRef<const Expr *> RHSExprs, ArrayRef<const Expr *> ReductionOps) { 5438 ASTContext &C = CGM.getContext(); 5439 5440 // void reduction_func(void *LHSArg, void *RHSArg); 5441 FunctionArgList Args; 5442 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5443 ImplicitParamDecl::Other); 5444 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5445 ImplicitParamDecl::Other); 5446 Args.push_back(&LHSArg); 5447 Args.push_back(&RHSArg); 5448 const auto &CGFI = 5449 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5450 std::string Name = getName({"omp", "reduction", "reduction_func"}); 5451 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 5452 llvm::GlobalValue::InternalLinkage, Name, 5453 &CGM.getModule()); 5454 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 5455 Fn->setDoesNotRecurse(); 5456 CodeGenFunction CGF(CGM); 5457 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 5458 5459 // Dst = (void*[n])(LHSArg); 5460 // Src = (void*[n])(RHSArg); 5461 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5462 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 5463 ArgsElemType->getPointerTo()), 5464 ArgsElemType, CGF.getPointerAlign()); 5465 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5466 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 5467 ArgsElemType->getPointerTo()), 5468 ArgsElemType, CGF.getPointerAlign()); 5469 5470 // ... 5471 // *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]); 5472 // ... 5473 CodeGenFunction::OMPPrivateScope Scope(CGF); 5474 const auto *IPriv = Privates.begin(); 5475 unsigned Idx = 0; 5476 for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) { 5477 const auto *RHSVar = 5478 cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl()); 5479 Scope.addPrivate(RHSVar, emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar)); 5480 const auto *LHSVar = 5481 cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl()); 5482 Scope.addPrivate(LHSVar, emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar)); 5483 QualType PrivTy = (*IPriv)->getType(); 5484 if (PrivTy->isVariablyModifiedType()) { 5485 // Get array size and emit VLA type. 5486 ++Idx; 5487 Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx); 5488 llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem); 5489 const VariableArrayType *VLA = 5490 CGF.getContext().getAsVariableArrayType(PrivTy); 5491 const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr()); 5492 CodeGenFunction::OpaqueValueMapping OpaqueMap( 5493 CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy))); 5494 CGF.EmitVariablyModifiedType(PrivTy); 5495 } 5496 } 5497 Scope.Privatize(); 5498 IPriv = Privates.begin(); 5499 const auto *ILHS = LHSExprs.begin(); 5500 const auto *IRHS = RHSExprs.begin(); 5501 for (const Expr *E : ReductionOps) { 5502 if ((*IPriv)->getType()->isArrayType()) { 5503 // Emit reduction for array section. 5504 const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5505 const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5506 EmitOMPAggregateReduction( 5507 CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5508 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5509 emitReductionCombiner(CGF, E); 5510 }); 5511 } else { 5512 // Emit reduction for array subscript or single variable. 5513 emitReductionCombiner(CGF, E); 5514 } 5515 ++IPriv; 5516 ++ILHS; 5517 ++IRHS; 5518 } 5519 Scope.ForceCleanup(); 5520 CGF.FinishFunction(); 5521 return Fn; 5522 } 5523 5524 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF, 5525 const Expr *ReductionOp, 5526 const Expr *PrivateRef, 5527 const DeclRefExpr *LHS, 5528 const DeclRefExpr *RHS) { 5529 if (PrivateRef->getType()->isArrayType()) { 5530 // Emit reduction for array section. 5531 const auto *LHSVar = cast<VarDecl>(LHS->getDecl()); 5532 const auto *RHSVar = cast<VarDecl>(RHS->getDecl()); 5533 EmitOMPAggregateReduction( 5534 CGF, PrivateRef->getType(), LHSVar, RHSVar, 5535 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5536 emitReductionCombiner(CGF, ReductionOp); 5537 }); 5538 } else { 5539 // Emit reduction for array subscript or single variable. 5540 emitReductionCombiner(CGF, ReductionOp); 5541 } 5542 } 5543 5544 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc, 5545 ArrayRef<const Expr *> Privates, 5546 ArrayRef<const Expr *> LHSExprs, 5547 ArrayRef<const Expr *> RHSExprs, 5548 ArrayRef<const Expr *> ReductionOps, 5549 ReductionOptionsTy Options) { 5550 if (!CGF.HaveInsertPoint()) 5551 return; 5552 5553 bool WithNowait = Options.WithNowait; 5554 bool SimpleReduction = Options.SimpleReduction; 5555 5556 // Next code should be emitted for reduction: 5557 // 5558 // static kmp_critical_name lock = { 0 }; 5559 // 5560 // void reduce_func(void *lhs[<n>], void *rhs[<n>]) { 5561 // *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]); 5562 // ... 5563 // *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1], 5564 // *(Type<n>-1*)rhs[<n>-1]); 5565 // } 5566 // 5567 // ... 5568 // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]}; 5569 // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5570 // RedList, reduce_func, &<lock>)) { 5571 // case 1: 5572 // ... 5573 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5574 // ... 5575 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5576 // break; 5577 // case 2: 5578 // ... 5579 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5580 // ... 5581 // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);] 5582 // break; 5583 // default:; 5584 // } 5585 // 5586 // if SimpleReduction is true, only the next code is generated: 5587 // ... 5588 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5589 // ... 5590 5591 ASTContext &C = CGM.getContext(); 5592 5593 if (SimpleReduction) { 5594 CodeGenFunction::RunCleanupsScope Scope(CGF); 5595 const auto *IPriv = Privates.begin(); 5596 const auto *ILHS = LHSExprs.begin(); 5597 const auto *IRHS = RHSExprs.begin(); 5598 for (const Expr *E : ReductionOps) { 5599 emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5600 cast<DeclRefExpr>(*IRHS)); 5601 ++IPriv; 5602 ++ILHS; 5603 ++IRHS; 5604 } 5605 return; 5606 } 5607 5608 // 1. Build a list of reduction variables. 5609 // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]}; 5610 auto Size = RHSExprs.size(); 5611 for (const Expr *E : Privates) { 5612 if (E->getType()->isVariablyModifiedType()) 5613 // Reserve place for array size. 5614 ++Size; 5615 } 5616 llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size); 5617 QualType ReductionArrayTy = 5618 C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 5619 /*IndexTypeQuals=*/0); 5620 Address ReductionList = 5621 CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list"); 5622 const auto *IPriv = Privates.begin(); 5623 unsigned Idx = 0; 5624 for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) { 5625 Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 5626 CGF.Builder.CreateStore( 5627 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5628 CGF.EmitLValue(RHSExprs[I]).getPointer(CGF), CGF.VoidPtrTy), 5629 Elem); 5630 if ((*IPriv)->getType()->isVariablyModifiedType()) { 5631 // Store array size. 5632 ++Idx; 5633 Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 5634 llvm::Value *Size = CGF.Builder.CreateIntCast( 5635 CGF.getVLASize( 5636 CGF.getContext().getAsVariableArrayType((*IPriv)->getType())) 5637 .NumElts, 5638 CGF.SizeTy, /*isSigned=*/false); 5639 CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy), 5640 Elem); 5641 } 5642 } 5643 5644 // 2. Emit reduce_func(). 5645 llvm::Function *ReductionFn = 5646 emitReductionFunction(Loc, CGF.ConvertTypeForMem(ReductionArrayTy), 5647 Privates, LHSExprs, RHSExprs, ReductionOps); 5648 5649 // 3. Create static kmp_critical_name lock = { 0 }; 5650 std::string Name = getName({"reduction"}); 5651 llvm::Value *Lock = getCriticalRegionLock(Name); 5652 5653 // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5654 // RedList, reduce_func, &<lock>); 5655 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE); 5656 llvm::Value *ThreadId = getThreadID(CGF, Loc); 5657 llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy); 5658 llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5659 ReductionList.getPointer(), CGF.VoidPtrTy); 5660 llvm::Value *Args[] = { 5661 IdentTLoc, // ident_t *<loc> 5662 ThreadId, // i32 <gtid> 5663 CGF.Builder.getInt32(RHSExprs.size()), // i32 <n> 5664 ReductionArrayTySize, // size_type sizeof(RedList) 5665 RL, // void *RedList 5666 ReductionFn, // void (*) (void *, void *) <reduce_func> 5667 Lock // kmp_critical_name *&<lock> 5668 }; 5669 llvm::Value *Res = CGF.EmitRuntimeCall( 5670 OMPBuilder.getOrCreateRuntimeFunction( 5671 CGM.getModule(), 5672 WithNowait ? OMPRTL___kmpc_reduce_nowait : OMPRTL___kmpc_reduce), 5673 Args); 5674 5675 // 5. Build switch(res) 5676 llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default"); 5677 llvm::SwitchInst *SwInst = 5678 CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2); 5679 5680 // 6. Build case 1: 5681 // ... 5682 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5683 // ... 5684 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5685 // break; 5686 llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1"); 5687 SwInst->addCase(CGF.Builder.getInt32(1), Case1BB); 5688 CGF.EmitBlock(Case1BB); 5689 5690 // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5691 llvm::Value *EndArgs[] = { 5692 IdentTLoc, // ident_t *<loc> 5693 ThreadId, // i32 <gtid> 5694 Lock // kmp_critical_name *&<lock> 5695 }; 5696 auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps]( 5697 CodeGenFunction &CGF, PrePostActionTy &Action) { 5698 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5699 const auto *IPriv = Privates.begin(); 5700 const auto *ILHS = LHSExprs.begin(); 5701 const auto *IRHS = RHSExprs.begin(); 5702 for (const Expr *E : ReductionOps) { 5703 RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5704 cast<DeclRefExpr>(*IRHS)); 5705 ++IPriv; 5706 ++ILHS; 5707 ++IRHS; 5708 } 5709 }; 5710 RegionCodeGenTy RCG(CodeGen); 5711 CommonActionTy Action( 5712 nullptr, llvm::None, 5713 OMPBuilder.getOrCreateRuntimeFunction( 5714 CGM.getModule(), WithNowait ? OMPRTL___kmpc_end_reduce_nowait 5715 : OMPRTL___kmpc_end_reduce), 5716 EndArgs); 5717 RCG.setAction(Action); 5718 RCG(CGF); 5719 5720 CGF.EmitBranch(DefaultBB); 5721 5722 // 7. Build case 2: 5723 // ... 5724 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5725 // ... 5726 // break; 5727 llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2"); 5728 SwInst->addCase(CGF.Builder.getInt32(2), Case2BB); 5729 CGF.EmitBlock(Case2BB); 5730 5731 auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps]( 5732 CodeGenFunction &CGF, PrePostActionTy &Action) { 5733 const auto *ILHS = LHSExprs.begin(); 5734 const auto *IRHS = RHSExprs.begin(); 5735 const auto *IPriv = Privates.begin(); 5736 for (const Expr *E : ReductionOps) { 5737 const Expr *XExpr = nullptr; 5738 const Expr *EExpr = nullptr; 5739 const Expr *UpExpr = nullptr; 5740 BinaryOperatorKind BO = BO_Comma; 5741 if (const auto *BO = dyn_cast<BinaryOperator>(E)) { 5742 if (BO->getOpcode() == BO_Assign) { 5743 XExpr = BO->getLHS(); 5744 UpExpr = BO->getRHS(); 5745 } 5746 } 5747 // Try to emit update expression as a simple atomic. 5748 const Expr *RHSExpr = UpExpr; 5749 if (RHSExpr) { 5750 // Analyze RHS part of the whole expression. 5751 if (const auto *ACO = dyn_cast<AbstractConditionalOperator>( 5752 RHSExpr->IgnoreParenImpCasts())) { 5753 // If this is a conditional operator, analyze its condition for 5754 // min/max reduction operator. 5755 RHSExpr = ACO->getCond(); 5756 } 5757 if (const auto *BORHS = 5758 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) { 5759 EExpr = BORHS->getRHS(); 5760 BO = BORHS->getOpcode(); 5761 } 5762 } 5763 if (XExpr) { 5764 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5765 auto &&AtomicRedGen = [BO, VD, 5766 Loc](CodeGenFunction &CGF, const Expr *XExpr, 5767 const Expr *EExpr, const Expr *UpExpr) { 5768 LValue X = CGF.EmitLValue(XExpr); 5769 RValue E; 5770 if (EExpr) 5771 E = CGF.EmitAnyExpr(EExpr); 5772 CGF.EmitOMPAtomicSimpleUpdateExpr( 5773 X, E, BO, /*IsXLHSInRHSPart=*/true, 5774 llvm::AtomicOrdering::Monotonic, Loc, 5775 [&CGF, UpExpr, VD, Loc](RValue XRValue) { 5776 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 5777 Address LHSTemp = CGF.CreateMemTemp(VD->getType()); 5778 CGF.emitOMPSimpleStore( 5779 CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue, 5780 VD->getType().getNonReferenceType(), Loc); 5781 PrivateScope.addPrivate(VD, LHSTemp); 5782 (void)PrivateScope.Privatize(); 5783 return CGF.EmitAnyExpr(UpExpr); 5784 }); 5785 }; 5786 if ((*IPriv)->getType()->isArrayType()) { 5787 // Emit atomic reduction for array section. 5788 const auto *RHSVar = 5789 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5790 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar, 5791 AtomicRedGen, XExpr, EExpr, UpExpr); 5792 } else { 5793 // Emit atomic reduction for array subscript or single variable. 5794 AtomicRedGen(CGF, XExpr, EExpr, UpExpr); 5795 } 5796 } else { 5797 // Emit as a critical region. 5798 auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *, 5799 const Expr *, const Expr *) { 5800 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5801 std::string Name = RT.getName({"atomic_reduction"}); 5802 RT.emitCriticalRegion( 5803 CGF, Name, 5804 [=](CodeGenFunction &CGF, PrePostActionTy &Action) { 5805 Action.Enter(CGF); 5806 emitReductionCombiner(CGF, E); 5807 }, 5808 Loc); 5809 }; 5810 if ((*IPriv)->getType()->isArrayType()) { 5811 const auto *LHSVar = 5812 cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5813 const auto *RHSVar = 5814 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5815 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5816 CritRedGen); 5817 } else { 5818 CritRedGen(CGF, nullptr, nullptr, nullptr); 5819 } 5820 } 5821 ++ILHS; 5822 ++IRHS; 5823 ++IPriv; 5824 } 5825 }; 5826 RegionCodeGenTy AtomicRCG(AtomicCodeGen); 5827 if (!WithNowait) { 5828 // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>); 5829 llvm::Value *EndArgs[] = { 5830 IdentTLoc, // ident_t *<loc> 5831 ThreadId, // i32 <gtid> 5832 Lock // kmp_critical_name *&<lock> 5833 }; 5834 CommonActionTy Action(nullptr, llvm::None, 5835 OMPBuilder.getOrCreateRuntimeFunction( 5836 CGM.getModule(), OMPRTL___kmpc_end_reduce), 5837 EndArgs); 5838 AtomicRCG.setAction(Action); 5839 AtomicRCG(CGF); 5840 } else { 5841 AtomicRCG(CGF); 5842 } 5843 5844 CGF.EmitBranch(DefaultBB); 5845 CGF.EmitBlock(DefaultBB, /*IsFinished=*/true); 5846 } 5847 5848 /// Generates unique name for artificial threadprivate variables. 5849 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>" 5850 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix, 5851 const Expr *Ref) { 5852 SmallString<256> Buffer; 5853 llvm::raw_svector_ostream Out(Buffer); 5854 const clang::DeclRefExpr *DE; 5855 const VarDecl *D = ::getBaseDecl(Ref, DE); 5856 if (!D) 5857 D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl()); 5858 D = D->getCanonicalDecl(); 5859 std::string Name = CGM.getOpenMPRuntime().getName( 5860 {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)}); 5861 Out << Prefix << Name << "_" 5862 << D->getCanonicalDecl()->getBeginLoc().getRawEncoding(); 5863 return std::string(Out.str()); 5864 } 5865 5866 /// Emits reduction initializer function: 5867 /// \code 5868 /// void @.red_init(void* %arg, void* %orig) { 5869 /// %0 = bitcast void* %arg to <type>* 5870 /// store <type> <init>, <type>* %0 5871 /// ret void 5872 /// } 5873 /// \endcode 5874 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM, 5875 SourceLocation Loc, 5876 ReductionCodeGen &RCG, unsigned N) { 5877 ASTContext &C = CGM.getContext(); 5878 QualType VoidPtrTy = C.VoidPtrTy; 5879 VoidPtrTy.addRestrict(); 5880 FunctionArgList Args; 5881 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, VoidPtrTy, 5882 ImplicitParamDecl::Other); 5883 ImplicitParamDecl ParamOrig(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, VoidPtrTy, 5884 ImplicitParamDecl::Other); 5885 Args.emplace_back(&Param); 5886 Args.emplace_back(&ParamOrig); 5887 const auto &FnInfo = 5888 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5889 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 5890 std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""}); 5891 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 5892 Name, &CGM.getModule()); 5893 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 5894 Fn->setDoesNotRecurse(); 5895 CodeGenFunction CGF(CGM); 5896 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 5897 QualType PrivateType = RCG.getPrivateType(N); 5898 Address PrivateAddr = CGF.EmitLoadOfPointer( 5899 CGF.Builder.CreateElementBitCast( 5900 CGF.GetAddrOfLocalVar(&Param), 5901 CGF.ConvertTypeForMem(PrivateType)->getPointerTo()), 5902 C.getPointerType(PrivateType)->castAs<PointerType>()); 5903 llvm::Value *Size = nullptr; 5904 // If the size of the reduction item is non-constant, load it from global 5905 // threadprivate variable. 5906 if (RCG.getSizes(N).second) { 5907 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 5908 CGF, CGM.getContext().getSizeType(), 5909 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 5910 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 5911 CGM.getContext().getSizeType(), Loc); 5912 } 5913 RCG.emitAggregateType(CGF, N, Size); 5914 Address OrigAddr = Address::invalid(); 5915 // If initializer uses initializer from declare reduction construct, emit a 5916 // pointer to the address of the original reduction item (reuired by reduction 5917 // initializer) 5918 if (RCG.usesReductionInitializer(N)) { 5919 Address SharedAddr = CGF.GetAddrOfLocalVar(&ParamOrig); 5920 OrigAddr = CGF.EmitLoadOfPointer( 5921 SharedAddr, 5922 CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr()); 5923 } 5924 // Emit the initializer: 5925 // %0 = bitcast void* %arg to <type>* 5926 // store <type> <init>, <type>* %0 5927 RCG.emitInitialization(CGF, N, PrivateAddr, OrigAddr, 5928 [](CodeGenFunction &) { return false; }); 5929 CGF.FinishFunction(); 5930 return Fn; 5931 } 5932 5933 /// Emits reduction combiner function: 5934 /// \code 5935 /// void @.red_comb(void* %arg0, void* %arg1) { 5936 /// %lhs = bitcast void* %arg0 to <type>* 5937 /// %rhs = bitcast void* %arg1 to <type>* 5938 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs) 5939 /// store <type> %2, <type>* %lhs 5940 /// ret void 5941 /// } 5942 /// \endcode 5943 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM, 5944 SourceLocation Loc, 5945 ReductionCodeGen &RCG, unsigned N, 5946 const Expr *ReductionOp, 5947 const Expr *LHS, const Expr *RHS, 5948 const Expr *PrivateRef) { 5949 ASTContext &C = CGM.getContext(); 5950 const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl()); 5951 const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl()); 5952 FunctionArgList Args; 5953 ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 5954 C.VoidPtrTy, ImplicitParamDecl::Other); 5955 ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5956 ImplicitParamDecl::Other); 5957 Args.emplace_back(&ParamInOut); 5958 Args.emplace_back(&ParamIn); 5959 const auto &FnInfo = 5960 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5961 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 5962 std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""}); 5963 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 5964 Name, &CGM.getModule()); 5965 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 5966 Fn->setDoesNotRecurse(); 5967 CodeGenFunction CGF(CGM); 5968 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 5969 llvm::Value *Size = nullptr; 5970 // If the size of the reduction item is non-constant, load it from global 5971 // threadprivate variable. 5972 if (RCG.getSizes(N).second) { 5973 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 5974 CGF, CGM.getContext().getSizeType(), 5975 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 5976 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 5977 CGM.getContext().getSizeType(), Loc); 5978 } 5979 RCG.emitAggregateType(CGF, N, Size); 5980 // Remap lhs and rhs variables to the addresses of the function arguments. 5981 // %lhs = bitcast void* %arg0 to <type>* 5982 // %rhs = bitcast void* %arg1 to <type>* 5983 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 5984 PrivateScope.addPrivate( 5985 LHSVD, 5986 // Pull out the pointer to the variable. 5987 CGF.EmitLoadOfPointer( 5988 CGF.Builder.CreateElementBitCast( 5989 CGF.GetAddrOfLocalVar(&ParamInOut), 5990 CGF.ConvertTypeForMem(LHSVD->getType())->getPointerTo()), 5991 C.getPointerType(LHSVD->getType())->castAs<PointerType>())); 5992 PrivateScope.addPrivate( 5993 RHSVD, 5994 // Pull out the pointer to the variable. 5995 CGF.EmitLoadOfPointer( 5996 CGF.Builder.CreateElementBitCast( 5997 CGF.GetAddrOfLocalVar(&ParamIn), 5998 CGF.ConvertTypeForMem(RHSVD->getType())->getPointerTo()), 5999 C.getPointerType(RHSVD->getType())->castAs<PointerType>())); 6000 PrivateScope.Privatize(); 6001 // Emit the combiner body: 6002 // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs) 6003 // store <type> %2, <type>* %lhs 6004 CGM.getOpenMPRuntime().emitSingleReductionCombiner( 6005 CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS), 6006 cast<DeclRefExpr>(RHS)); 6007 CGF.FinishFunction(); 6008 return Fn; 6009 } 6010 6011 /// Emits reduction finalizer function: 6012 /// \code 6013 /// void @.red_fini(void* %arg) { 6014 /// %0 = bitcast void* %arg to <type>* 6015 /// <destroy>(<type>* %0) 6016 /// ret void 6017 /// } 6018 /// \endcode 6019 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM, 6020 SourceLocation Loc, 6021 ReductionCodeGen &RCG, unsigned N) { 6022 if (!RCG.needCleanups(N)) 6023 return nullptr; 6024 ASTContext &C = CGM.getContext(); 6025 FunctionArgList Args; 6026 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6027 ImplicitParamDecl::Other); 6028 Args.emplace_back(&Param); 6029 const auto &FnInfo = 6030 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6031 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6032 std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""}); 6033 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6034 Name, &CGM.getModule()); 6035 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6036 Fn->setDoesNotRecurse(); 6037 CodeGenFunction CGF(CGM); 6038 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6039 Address PrivateAddr = CGF.EmitLoadOfPointer( 6040 CGF.GetAddrOfLocalVar(&Param), C.VoidPtrTy.castAs<PointerType>()); 6041 llvm::Value *Size = nullptr; 6042 // If the size of the reduction item is non-constant, load it from global 6043 // threadprivate variable. 6044 if (RCG.getSizes(N).second) { 6045 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6046 CGF, CGM.getContext().getSizeType(), 6047 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6048 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6049 CGM.getContext().getSizeType(), Loc); 6050 } 6051 RCG.emitAggregateType(CGF, N, Size); 6052 // Emit the finalizer body: 6053 // <destroy>(<type>* %0) 6054 RCG.emitCleanups(CGF, N, PrivateAddr); 6055 CGF.FinishFunction(Loc); 6056 return Fn; 6057 } 6058 6059 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit( 6060 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 6061 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 6062 if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty()) 6063 return nullptr; 6064 6065 // Build typedef struct: 6066 // kmp_taskred_input { 6067 // void *reduce_shar; // shared reduction item 6068 // void *reduce_orig; // original reduction item used for initialization 6069 // size_t reduce_size; // size of data item 6070 // void *reduce_init; // data initialization routine 6071 // void *reduce_fini; // data finalization routine 6072 // void *reduce_comb; // data combiner routine 6073 // kmp_task_red_flags_t flags; // flags for additional info from compiler 6074 // } kmp_taskred_input_t; 6075 ASTContext &C = CGM.getContext(); 6076 RecordDecl *RD = C.buildImplicitRecord("kmp_taskred_input_t"); 6077 RD->startDefinition(); 6078 const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6079 const FieldDecl *OrigFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6080 const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType()); 6081 const FieldDecl *InitFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6082 const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6083 const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6084 const FieldDecl *FlagsFD = addFieldToRecordDecl( 6085 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false)); 6086 RD->completeDefinition(); 6087 QualType RDType = C.getRecordType(RD); 6088 unsigned Size = Data.ReductionVars.size(); 6089 llvm::APInt ArraySize(/*numBits=*/64, Size); 6090 QualType ArrayRDType = C.getConstantArrayType( 6091 RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 6092 // kmp_task_red_input_t .rd_input.[Size]; 6093 Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input."); 6094 ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionOrigs, 6095 Data.ReductionCopies, Data.ReductionOps); 6096 for (unsigned Cnt = 0; Cnt < Size; ++Cnt) { 6097 // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt]; 6098 llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0), 6099 llvm::ConstantInt::get(CGM.SizeTy, Cnt)}; 6100 llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP( 6101 TaskRedInput.getElementType(), TaskRedInput.getPointer(), Idxs, 6102 /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc, 6103 ".rd_input.gep."); 6104 LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType); 6105 // ElemLVal.reduce_shar = &Shareds[Cnt]; 6106 LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD); 6107 RCG.emitSharedOrigLValue(CGF, Cnt); 6108 llvm::Value *CastedShared = 6109 CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer(CGF)); 6110 CGF.EmitStoreOfScalar(CastedShared, SharedLVal); 6111 // ElemLVal.reduce_orig = &Origs[Cnt]; 6112 LValue OrigLVal = CGF.EmitLValueForField(ElemLVal, OrigFD); 6113 llvm::Value *CastedOrig = 6114 CGF.EmitCastToVoidPtr(RCG.getOrigLValue(Cnt).getPointer(CGF)); 6115 CGF.EmitStoreOfScalar(CastedOrig, OrigLVal); 6116 RCG.emitAggregateType(CGF, Cnt); 6117 llvm::Value *SizeValInChars; 6118 llvm::Value *SizeVal; 6119 std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt); 6120 // We use delayed creation/initialization for VLAs and array sections. It is 6121 // required because runtime does not provide the way to pass the sizes of 6122 // VLAs/array sections to initializer/combiner/finalizer functions. Instead 6123 // threadprivate global variables are used to store these values and use 6124 // them in the functions. 6125 bool DelayedCreation = !!SizeVal; 6126 SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy, 6127 /*isSigned=*/false); 6128 LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD); 6129 CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal); 6130 // ElemLVal.reduce_init = init; 6131 LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD); 6132 llvm::Value *InitAddr = 6133 CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt)); 6134 CGF.EmitStoreOfScalar(InitAddr, InitLVal); 6135 // ElemLVal.reduce_fini = fini; 6136 LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD); 6137 llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt); 6138 llvm::Value *FiniAddr = Fini 6139 ? CGF.EmitCastToVoidPtr(Fini) 6140 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy); 6141 CGF.EmitStoreOfScalar(FiniAddr, FiniLVal); 6142 // ElemLVal.reduce_comb = comb; 6143 LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD); 6144 llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction( 6145 CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt], 6146 RHSExprs[Cnt], Data.ReductionCopies[Cnt])); 6147 CGF.EmitStoreOfScalar(CombAddr, CombLVal); 6148 // ElemLVal.flags = 0; 6149 LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD); 6150 if (DelayedCreation) { 6151 CGF.EmitStoreOfScalar( 6152 llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true), 6153 FlagsLVal); 6154 } else 6155 CGF.EmitNullInitialization(FlagsLVal.getAddress(CGF), 6156 FlagsLVal.getType()); 6157 } 6158 if (Data.IsReductionWithTaskMod) { 6159 // Build call void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int 6160 // is_ws, int num, void *data); 6161 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc); 6162 llvm::Value *GTid = CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), 6163 CGM.IntTy, /*isSigned=*/true); 6164 llvm::Value *Args[] = { 6165 IdentTLoc, GTid, 6166 llvm::ConstantInt::get(CGM.IntTy, Data.IsWorksharingReduction ? 1 : 0, 6167 /*isSigned=*/true), 6168 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true), 6169 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6170 TaskRedInput.getPointer(), CGM.VoidPtrTy)}; 6171 return CGF.EmitRuntimeCall( 6172 OMPBuilder.getOrCreateRuntimeFunction( 6173 CGM.getModule(), OMPRTL___kmpc_taskred_modifier_init), 6174 Args); 6175 } 6176 // Build call void *__kmpc_taskred_init(int gtid, int num_data, void *data); 6177 llvm::Value *Args[] = { 6178 CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy, 6179 /*isSigned=*/true), 6180 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true), 6181 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(), 6182 CGM.VoidPtrTy)}; 6183 return CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 6184 CGM.getModule(), OMPRTL___kmpc_taskred_init), 6185 Args); 6186 } 6187 6188 void CGOpenMPRuntime::emitTaskReductionFini(CodeGenFunction &CGF, 6189 SourceLocation Loc, 6190 bool IsWorksharingReduction) { 6191 // Build call void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int 6192 // is_ws, int num, void *data); 6193 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc); 6194 llvm::Value *GTid = CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), 6195 CGM.IntTy, /*isSigned=*/true); 6196 llvm::Value *Args[] = {IdentTLoc, GTid, 6197 llvm::ConstantInt::get(CGM.IntTy, 6198 IsWorksharingReduction ? 1 : 0, 6199 /*isSigned=*/true)}; 6200 (void)CGF.EmitRuntimeCall( 6201 OMPBuilder.getOrCreateRuntimeFunction( 6202 CGM.getModule(), OMPRTL___kmpc_task_reduction_modifier_fini), 6203 Args); 6204 } 6205 6206 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 6207 SourceLocation Loc, 6208 ReductionCodeGen &RCG, 6209 unsigned N) { 6210 auto Sizes = RCG.getSizes(N); 6211 // Emit threadprivate global variable if the type is non-constant 6212 // (Sizes.second = nullptr). 6213 if (Sizes.second) { 6214 llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy, 6215 /*isSigned=*/false); 6216 Address SizeAddr = getAddrOfArtificialThreadPrivate( 6217 CGF, CGM.getContext().getSizeType(), 6218 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6219 CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false); 6220 } 6221 } 6222 6223 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF, 6224 SourceLocation Loc, 6225 llvm::Value *ReductionsPtr, 6226 LValue SharedLVal) { 6227 // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 6228 // *d); 6229 llvm::Value *Args[] = {CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), 6230 CGM.IntTy, 6231 /*isSigned=*/true), 6232 ReductionsPtr, 6233 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6234 SharedLVal.getPointer(CGF), CGM.VoidPtrTy)}; 6235 return Address( 6236 CGF.EmitRuntimeCall( 6237 OMPBuilder.getOrCreateRuntimeFunction( 6238 CGM.getModule(), OMPRTL___kmpc_task_reduction_get_th_data), 6239 Args), 6240 CGF.Int8Ty, SharedLVal.getAlignment()); 6241 } 6242 6243 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF, SourceLocation Loc, 6244 const OMPTaskDataTy &Data) { 6245 if (!CGF.HaveInsertPoint()) 6246 return; 6247 6248 if (CGF.CGM.getLangOpts().OpenMPIRBuilder && Data.Dependences.empty()) { 6249 // TODO: Need to support taskwait with dependences in the OpenMPIRBuilder. 6250 OMPBuilder.createTaskwait(CGF.Builder); 6251 } else { 6252 llvm::Value *ThreadID = getThreadID(CGF, Loc); 6253 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 6254 auto &M = CGM.getModule(); 6255 Address DependenciesArray = Address::invalid(); 6256 llvm::Value *NumOfElements; 6257 std::tie(NumOfElements, DependenciesArray) = 6258 emitDependClause(CGF, Data.Dependences, Loc); 6259 llvm::Value *DepWaitTaskArgs[6]; 6260 if (!Data.Dependences.empty()) { 6261 DepWaitTaskArgs[0] = UpLoc; 6262 DepWaitTaskArgs[1] = ThreadID; 6263 DepWaitTaskArgs[2] = NumOfElements; 6264 DepWaitTaskArgs[3] = DependenciesArray.getPointer(); 6265 DepWaitTaskArgs[4] = CGF.Builder.getInt32(0); 6266 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 6267 6268 CodeGenFunction::RunCleanupsScope LocalScope(CGF); 6269 6270 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 6271 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 6272 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info 6273 // is specified. 6274 CGF.EmitRuntimeCall( 6275 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_omp_wait_deps), 6276 DepWaitTaskArgs); 6277 6278 } else { 6279 6280 // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 6281 // global_tid); 6282 llvm::Value *Args[] = {UpLoc, ThreadID}; 6283 // Ignore return result until untied tasks are supported. 6284 CGF.EmitRuntimeCall( 6285 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_omp_taskwait), 6286 Args); 6287 } 6288 } 6289 6290 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 6291 Region->emitUntiedSwitch(CGF); 6292 } 6293 6294 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF, 6295 OpenMPDirectiveKind InnerKind, 6296 const RegionCodeGenTy &CodeGen, 6297 bool HasCancel) { 6298 if (!CGF.HaveInsertPoint()) 6299 return; 6300 InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel, 6301 InnerKind != OMPD_critical && 6302 InnerKind != OMPD_master && 6303 InnerKind != OMPD_masked); 6304 CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr); 6305 } 6306 6307 namespace { 6308 enum RTCancelKind { 6309 CancelNoreq = 0, 6310 CancelParallel = 1, 6311 CancelLoop = 2, 6312 CancelSections = 3, 6313 CancelTaskgroup = 4 6314 }; 6315 } // anonymous namespace 6316 6317 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) { 6318 RTCancelKind CancelKind = CancelNoreq; 6319 if (CancelRegion == OMPD_parallel) 6320 CancelKind = CancelParallel; 6321 else if (CancelRegion == OMPD_for) 6322 CancelKind = CancelLoop; 6323 else if (CancelRegion == OMPD_sections) 6324 CancelKind = CancelSections; 6325 else { 6326 assert(CancelRegion == OMPD_taskgroup); 6327 CancelKind = CancelTaskgroup; 6328 } 6329 return CancelKind; 6330 } 6331 6332 void CGOpenMPRuntime::emitCancellationPointCall( 6333 CodeGenFunction &CGF, SourceLocation Loc, 6334 OpenMPDirectiveKind CancelRegion) { 6335 if (!CGF.HaveInsertPoint()) 6336 return; 6337 // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 6338 // global_tid, kmp_int32 cncl_kind); 6339 if (auto *OMPRegionInfo = 6340 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6341 // For 'cancellation point taskgroup', the task region info may not have a 6342 // cancel. This may instead happen in another adjacent task. 6343 if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) { 6344 llvm::Value *Args[] = { 6345 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 6346 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6347 // Ignore return result until untied tasks are supported. 6348 llvm::Value *Result = CGF.EmitRuntimeCall( 6349 OMPBuilder.getOrCreateRuntimeFunction( 6350 CGM.getModule(), OMPRTL___kmpc_cancellationpoint), 6351 Args); 6352 // if (__kmpc_cancellationpoint()) { 6353 // call i32 @__kmpc_cancel_barrier( // for parallel cancellation only 6354 // exit from construct; 6355 // } 6356 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6357 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6358 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6359 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6360 CGF.EmitBlock(ExitBB); 6361 if (CancelRegion == OMPD_parallel) 6362 emitBarrierCall(CGF, Loc, OMPD_unknown, /*EmitChecks=*/false); 6363 // exit from construct; 6364 CodeGenFunction::JumpDest CancelDest = 6365 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6366 CGF.EmitBranchThroughCleanup(CancelDest); 6367 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6368 } 6369 } 6370 } 6371 6372 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc, 6373 const Expr *IfCond, 6374 OpenMPDirectiveKind CancelRegion) { 6375 if (!CGF.HaveInsertPoint()) 6376 return; 6377 // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 6378 // kmp_int32 cncl_kind); 6379 auto &M = CGM.getModule(); 6380 if (auto *OMPRegionInfo = 6381 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6382 auto &&ThenGen = [this, &M, Loc, CancelRegion, 6383 OMPRegionInfo](CodeGenFunction &CGF, PrePostActionTy &) { 6384 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6385 llvm::Value *Args[] = { 6386 RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc), 6387 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6388 // Ignore return result until untied tasks are supported. 6389 llvm::Value *Result = CGF.EmitRuntimeCall( 6390 OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_cancel), Args); 6391 // if (__kmpc_cancel()) { 6392 // call i32 @__kmpc_cancel_barrier( // for parallel cancellation only 6393 // exit from construct; 6394 // } 6395 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6396 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6397 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6398 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6399 CGF.EmitBlock(ExitBB); 6400 if (CancelRegion == OMPD_parallel) 6401 RT.emitBarrierCall(CGF, Loc, OMPD_unknown, /*EmitChecks=*/false); 6402 // exit from construct; 6403 CodeGenFunction::JumpDest CancelDest = 6404 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6405 CGF.EmitBranchThroughCleanup(CancelDest); 6406 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6407 }; 6408 if (IfCond) { 6409 emitIfClause(CGF, IfCond, ThenGen, 6410 [](CodeGenFunction &, PrePostActionTy &) {}); 6411 } else { 6412 RegionCodeGenTy ThenRCG(ThenGen); 6413 ThenRCG(CGF); 6414 } 6415 } 6416 } 6417 6418 namespace { 6419 /// Cleanup action for uses_allocators support. 6420 class OMPUsesAllocatorsActionTy final : public PrePostActionTy { 6421 ArrayRef<std::pair<const Expr *, const Expr *>> Allocators; 6422 6423 public: 6424 OMPUsesAllocatorsActionTy( 6425 ArrayRef<std::pair<const Expr *, const Expr *>> Allocators) 6426 : Allocators(Allocators) {} 6427 void Enter(CodeGenFunction &CGF) override { 6428 if (!CGF.HaveInsertPoint()) 6429 return; 6430 for (const auto &AllocatorData : Allocators) { 6431 CGF.CGM.getOpenMPRuntime().emitUsesAllocatorsInit( 6432 CGF, AllocatorData.first, AllocatorData.second); 6433 } 6434 } 6435 void Exit(CodeGenFunction &CGF) override { 6436 if (!CGF.HaveInsertPoint()) 6437 return; 6438 for (const auto &AllocatorData : Allocators) { 6439 CGF.CGM.getOpenMPRuntime().emitUsesAllocatorsFini(CGF, 6440 AllocatorData.first); 6441 } 6442 } 6443 }; 6444 } // namespace 6445 6446 void CGOpenMPRuntime::emitTargetOutlinedFunction( 6447 const OMPExecutableDirective &D, StringRef ParentName, 6448 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6449 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6450 assert(!ParentName.empty() && "Invalid target region parent name!"); 6451 HasEmittedTargetRegion = true; 6452 SmallVector<std::pair<const Expr *, const Expr *>, 4> Allocators; 6453 for (const auto *C : D.getClausesOfKind<OMPUsesAllocatorsClause>()) { 6454 for (unsigned I = 0, E = C->getNumberOfAllocators(); I < E; ++I) { 6455 const OMPUsesAllocatorsClause::Data D = C->getAllocatorData(I); 6456 if (!D.AllocatorTraits) 6457 continue; 6458 Allocators.emplace_back(D.Allocator, D.AllocatorTraits); 6459 } 6460 } 6461 OMPUsesAllocatorsActionTy UsesAllocatorAction(Allocators); 6462 CodeGen.setAction(UsesAllocatorAction); 6463 emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID, 6464 IsOffloadEntry, CodeGen); 6465 } 6466 6467 void CGOpenMPRuntime::emitUsesAllocatorsInit(CodeGenFunction &CGF, 6468 const Expr *Allocator, 6469 const Expr *AllocatorTraits) { 6470 llvm::Value *ThreadId = getThreadID(CGF, Allocator->getExprLoc()); 6471 ThreadId = CGF.Builder.CreateIntCast(ThreadId, CGF.IntTy, /*isSigned=*/true); 6472 // Use default memspace handle. 6473 llvm::Value *MemSpaceHandle = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 6474 llvm::Value *NumTraits = llvm::ConstantInt::get( 6475 CGF.IntTy, cast<ConstantArrayType>( 6476 AllocatorTraits->getType()->getAsArrayTypeUnsafe()) 6477 ->getSize() 6478 .getLimitedValue()); 6479 LValue AllocatorTraitsLVal = CGF.EmitLValue(AllocatorTraits); 6480 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6481 AllocatorTraitsLVal.getAddress(CGF), CGF.VoidPtrPtrTy, CGF.VoidPtrTy); 6482 AllocatorTraitsLVal = CGF.MakeAddrLValue(Addr, CGF.getContext().VoidPtrTy, 6483 AllocatorTraitsLVal.getBaseInfo(), 6484 AllocatorTraitsLVal.getTBAAInfo()); 6485 llvm::Value *Traits = 6486 CGF.EmitLoadOfScalar(AllocatorTraitsLVal, AllocatorTraits->getExprLoc()); 6487 6488 llvm::Value *AllocatorVal = 6489 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 6490 CGM.getModule(), OMPRTL___kmpc_init_allocator), 6491 {ThreadId, MemSpaceHandle, NumTraits, Traits}); 6492 // Store to allocator. 6493 CGF.EmitVarDecl(*cast<VarDecl>( 6494 cast<DeclRefExpr>(Allocator->IgnoreParenImpCasts())->getDecl())); 6495 LValue AllocatorLVal = CGF.EmitLValue(Allocator->IgnoreParenImpCasts()); 6496 AllocatorVal = 6497 CGF.EmitScalarConversion(AllocatorVal, CGF.getContext().VoidPtrTy, 6498 Allocator->getType(), Allocator->getExprLoc()); 6499 CGF.EmitStoreOfScalar(AllocatorVal, AllocatorLVal); 6500 } 6501 6502 void CGOpenMPRuntime::emitUsesAllocatorsFini(CodeGenFunction &CGF, 6503 const Expr *Allocator) { 6504 llvm::Value *ThreadId = getThreadID(CGF, Allocator->getExprLoc()); 6505 ThreadId = CGF.Builder.CreateIntCast(ThreadId, CGF.IntTy, /*isSigned=*/true); 6506 LValue AllocatorLVal = CGF.EmitLValue(Allocator->IgnoreParenImpCasts()); 6507 llvm::Value *AllocatorVal = 6508 CGF.EmitLoadOfScalar(AllocatorLVal, Allocator->getExprLoc()); 6509 AllocatorVal = CGF.EmitScalarConversion(AllocatorVal, Allocator->getType(), 6510 CGF.getContext().VoidPtrTy, 6511 Allocator->getExprLoc()); 6512 (void)CGF.EmitRuntimeCall( 6513 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 6514 OMPRTL___kmpc_destroy_allocator), 6515 {ThreadId, AllocatorVal}); 6516 } 6517 6518 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper( 6519 const OMPExecutableDirective &D, StringRef ParentName, 6520 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6521 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6522 // Create a unique name for the entry function using the source location 6523 // information of the current target region. The name will be something like: 6524 // 6525 // __omp_offloading_DD_FFFF_PP_lBB 6526 // 6527 // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the 6528 // mangled name of the function that encloses the target region and BB is the 6529 // line number of the target region. 6530 6531 const bool BuildOutlinedFn = CGM.getLangOpts().OpenMPIsDevice || 6532 !CGM.getLangOpts().OpenMPOffloadMandatory; 6533 unsigned DeviceID; 6534 unsigned FileID; 6535 unsigned Line; 6536 getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID, 6537 Line); 6538 SmallString<64> EntryFnName; 6539 { 6540 llvm::raw_svector_ostream OS(EntryFnName); 6541 OS << "__omp_offloading" << llvm::format("_%x", DeviceID) 6542 << llvm::format("_%x_", FileID) << ParentName << "_l" << Line; 6543 } 6544 6545 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 6546 6547 CodeGenFunction CGF(CGM, true); 6548 CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName); 6549 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6550 6551 if (BuildOutlinedFn) 6552 OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS, D.getBeginLoc()); 6553 6554 // If this target outline function is not an offload entry, we don't need to 6555 // register it. 6556 if (!IsOffloadEntry) 6557 return; 6558 6559 // The target region ID is used by the runtime library to identify the current 6560 // target region, so it only has to be unique and not necessarily point to 6561 // anything. It could be the pointer to the outlined function that implements 6562 // the target region, but we aren't using that so that the compiler doesn't 6563 // need to keep that, and could therefore inline the host function if proven 6564 // worthwhile during optimization. In the other hand, if emitting code for the 6565 // device, the ID has to be the function address so that it can retrieved from 6566 // the offloading entry and launched by the runtime library. We also mark the 6567 // outlined function to have external linkage in case we are emitting code for 6568 // the device, because these functions will be entry points to the device. 6569 6570 if (CGM.getLangOpts().OpenMPIsDevice) { 6571 OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy); 6572 OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 6573 OutlinedFn->setDSOLocal(false); 6574 if (CGM.getTriple().isAMDGCN()) 6575 OutlinedFn->setCallingConv(llvm::CallingConv::AMDGPU_KERNEL); 6576 } else { 6577 std::string Name = getName({EntryFnName, "region_id"}); 6578 OutlinedFnID = new llvm::GlobalVariable( 6579 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 6580 llvm::GlobalValue::WeakAnyLinkage, 6581 llvm::Constant::getNullValue(CGM.Int8Ty), Name); 6582 } 6583 6584 // If we do not allow host fallback we still need a named address to use. 6585 llvm::Constant *TargetRegionEntryAddr = OutlinedFn; 6586 if (!BuildOutlinedFn) { 6587 assert(!CGM.getModule().getGlobalVariable(EntryFnName, true) && 6588 "Named kernel already exists?"); 6589 TargetRegionEntryAddr = new llvm::GlobalVariable( 6590 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 6591 llvm::GlobalValue::InternalLinkage, 6592 llvm::Constant::getNullValue(CGM.Int8Ty), EntryFnName); 6593 } 6594 6595 // Register the information for the entry associated with this target region. 6596 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 6597 DeviceID, FileID, ParentName, Line, TargetRegionEntryAddr, OutlinedFnID, 6598 OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion); 6599 6600 // Add NumTeams and ThreadLimit attributes to the outlined GPU function 6601 int32_t DefaultValTeams = -1; 6602 getNumTeamsExprForTargetDirective(CGF, D, DefaultValTeams); 6603 if (DefaultValTeams > 0 && OutlinedFn) { 6604 OutlinedFn->addFnAttr("omp_target_num_teams", 6605 std::to_string(DefaultValTeams)); 6606 } 6607 int32_t DefaultValThreads = -1; 6608 getNumThreadsExprForTargetDirective(CGF, D, DefaultValThreads); 6609 if (DefaultValThreads > 0 && OutlinedFn) { 6610 OutlinedFn->addFnAttr("omp_target_thread_limit", 6611 std::to_string(DefaultValThreads)); 6612 } 6613 6614 if (BuildOutlinedFn) 6615 CGM.getTargetCodeGenInfo().setTargetAttributes(nullptr, OutlinedFn, CGM); 6616 } 6617 6618 /// Checks if the expression is constant or does not have non-trivial function 6619 /// calls. 6620 static bool isTrivial(ASTContext &Ctx, const Expr * E) { 6621 // We can skip constant expressions. 6622 // We can skip expressions with trivial calls or simple expressions. 6623 return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) || 6624 !E->hasNonTrivialCall(Ctx)) && 6625 !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true); 6626 } 6627 6628 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx, 6629 const Stmt *Body) { 6630 const Stmt *Child = Body->IgnoreContainers(); 6631 while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) { 6632 Child = nullptr; 6633 for (const Stmt *S : C->body()) { 6634 if (const auto *E = dyn_cast<Expr>(S)) { 6635 if (isTrivial(Ctx, E)) 6636 continue; 6637 } 6638 // Some of the statements can be ignored. 6639 if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) || 6640 isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S)) 6641 continue; 6642 // Analyze declarations. 6643 if (const auto *DS = dyn_cast<DeclStmt>(S)) { 6644 if (llvm::all_of(DS->decls(), [](const Decl *D) { 6645 if (isa<EmptyDecl>(D) || isa<DeclContext>(D) || 6646 isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) || 6647 isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) || 6648 isa<UsingDirectiveDecl>(D) || 6649 isa<OMPDeclareReductionDecl>(D) || 6650 isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D)) 6651 return true; 6652 const auto *VD = dyn_cast<VarDecl>(D); 6653 if (!VD) 6654 return false; 6655 return VD->hasGlobalStorage() || !VD->isUsed(); 6656 })) 6657 continue; 6658 } 6659 // Found multiple children - cannot get the one child only. 6660 if (Child) 6661 return nullptr; 6662 Child = S; 6663 } 6664 if (Child) 6665 Child = Child->IgnoreContainers(); 6666 } 6667 return Child; 6668 } 6669 6670 const Expr *CGOpenMPRuntime::getNumTeamsExprForTargetDirective( 6671 CodeGenFunction &CGF, const OMPExecutableDirective &D, 6672 int32_t &DefaultVal) { 6673 6674 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6675 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 6676 "Expected target-based executable directive."); 6677 switch (DirectiveKind) { 6678 case OMPD_target: { 6679 const auto *CS = D.getInnermostCapturedStmt(); 6680 const auto *Body = 6681 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 6682 const Stmt *ChildStmt = 6683 CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body); 6684 if (const auto *NestedDir = 6685 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 6686 if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) { 6687 if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) { 6688 const Expr *NumTeams = 6689 NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6690 if (NumTeams->isIntegerConstantExpr(CGF.getContext())) 6691 if (auto Constant = 6692 NumTeams->getIntegerConstantExpr(CGF.getContext())) 6693 DefaultVal = Constant->getExtValue(); 6694 return NumTeams; 6695 } 6696 DefaultVal = 0; 6697 return nullptr; 6698 } 6699 if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) || 6700 isOpenMPSimdDirective(NestedDir->getDirectiveKind())) { 6701 DefaultVal = 1; 6702 return nullptr; 6703 } 6704 DefaultVal = 1; 6705 return nullptr; 6706 } 6707 // A value of -1 is used to check if we need to emit no teams region 6708 DefaultVal = -1; 6709 return nullptr; 6710 } 6711 case OMPD_target_teams: 6712 case OMPD_target_teams_distribute: 6713 case OMPD_target_teams_distribute_simd: 6714 case OMPD_target_teams_distribute_parallel_for: 6715 case OMPD_target_teams_distribute_parallel_for_simd: { 6716 if (D.hasClausesOfKind<OMPNumTeamsClause>()) { 6717 const Expr *NumTeams = 6718 D.getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6719 if (NumTeams->isIntegerConstantExpr(CGF.getContext())) 6720 if (auto Constant = NumTeams->getIntegerConstantExpr(CGF.getContext())) 6721 DefaultVal = Constant->getExtValue(); 6722 return NumTeams; 6723 } 6724 DefaultVal = 0; 6725 return nullptr; 6726 } 6727 case OMPD_target_parallel: 6728 case OMPD_target_parallel_for: 6729 case OMPD_target_parallel_for_simd: 6730 case OMPD_target_simd: 6731 DefaultVal = 1; 6732 return nullptr; 6733 case OMPD_parallel: 6734 case OMPD_for: 6735 case OMPD_parallel_for: 6736 case OMPD_parallel_master: 6737 case OMPD_parallel_sections: 6738 case OMPD_for_simd: 6739 case OMPD_parallel_for_simd: 6740 case OMPD_cancel: 6741 case OMPD_cancellation_point: 6742 case OMPD_ordered: 6743 case OMPD_threadprivate: 6744 case OMPD_allocate: 6745 case OMPD_task: 6746 case OMPD_simd: 6747 case OMPD_tile: 6748 case OMPD_unroll: 6749 case OMPD_sections: 6750 case OMPD_section: 6751 case OMPD_single: 6752 case OMPD_master: 6753 case OMPD_critical: 6754 case OMPD_taskyield: 6755 case OMPD_barrier: 6756 case OMPD_taskwait: 6757 case OMPD_taskgroup: 6758 case OMPD_atomic: 6759 case OMPD_flush: 6760 case OMPD_depobj: 6761 case OMPD_scan: 6762 case OMPD_teams: 6763 case OMPD_target_data: 6764 case OMPD_target_exit_data: 6765 case OMPD_target_enter_data: 6766 case OMPD_distribute: 6767 case OMPD_distribute_simd: 6768 case OMPD_distribute_parallel_for: 6769 case OMPD_distribute_parallel_for_simd: 6770 case OMPD_teams_distribute: 6771 case OMPD_teams_distribute_simd: 6772 case OMPD_teams_distribute_parallel_for: 6773 case OMPD_teams_distribute_parallel_for_simd: 6774 case OMPD_target_update: 6775 case OMPD_declare_simd: 6776 case OMPD_declare_variant: 6777 case OMPD_begin_declare_variant: 6778 case OMPD_end_declare_variant: 6779 case OMPD_declare_target: 6780 case OMPD_end_declare_target: 6781 case OMPD_declare_reduction: 6782 case OMPD_declare_mapper: 6783 case OMPD_taskloop: 6784 case OMPD_taskloop_simd: 6785 case OMPD_master_taskloop: 6786 case OMPD_master_taskloop_simd: 6787 case OMPD_parallel_master_taskloop: 6788 case OMPD_parallel_master_taskloop_simd: 6789 case OMPD_requires: 6790 case OMPD_metadirective: 6791 case OMPD_unknown: 6792 break; 6793 default: 6794 break; 6795 } 6796 llvm_unreachable("Unexpected directive kind."); 6797 } 6798 6799 llvm::Value *CGOpenMPRuntime::emitNumTeamsForTargetDirective( 6800 CodeGenFunction &CGF, const OMPExecutableDirective &D) { 6801 assert(!CGF.getLangOpts().OpenMPIsDevice && 6802 "Clauses associated with the teams directive expected to be emitted " 6803 "only for the host!"); 6804 CGBuilderTy &Bld = CGF.Builder; 6805 int32_t DefaultNT = -1; 6806 const Expr *NumTeams = getNumTeamsExprForTargetDirective(CGF, D, DefaultNT); 6807 if (NumTeams != nullptr) { 6808 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6809 6810 switch (DirectiveKind) { 6811 case OMPD_target: { 6812 const auto *CS = D.getInnermostCapturedStmt(); 6813 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6814 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6815 llvm::Value *NumTeamsVal = CGF.EmitScalarExpr(NumTeams, 6816 /*IgnoreResultAssign*/ true); 6817 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6818 /*isSigned=*/true); 6819 } 6820 case OMPD_target_teams: 6821 case OMPD_target_teams_distribute: 6822 case OMPD_target_teams_distribute_simd: 6823 case OMPD_target_teams_distribute_parallel_for: 6824 case OMPD_target_teams_distribute_parallel_for_simd: { 6825 CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF); 6826 llvm::Value *NumTeamsVal = CGF.EmitScalarExpr(NumTeams, 6827 /*IgnoreResultAssign*/ true); 6828 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6829 /*isSigned=*/true); 6830 } 6831 default: 6832 break; 6833 } 6834 } else if (DefaultNT == -1) { 6835 return nullptr; 6836 } 6837 6838 return Bld.getInt32(DefaultNT); 6839 } 6840 6841 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS, 6842 llvm::Value *DefaultThreadLimitVal) { 6843 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6844 CGF.getContext(), CS->getCapturedStmt()); 6845 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6846 if (isOpenMPParallelDirective(Dir->getDirectiveKind())) { 6847 llvm::Value *NumThreads = nullptr; 6848 llvm::Value *CondVal = nullptr; 6849 // Handle if clause. If if clause present, the number of threads is 6850 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 6851 if (Dir->hasClausesOfKind<OMPIfClause>()) { 6852 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6853 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6854 const OMPIfClause *IfClause = nullptr; 6855 for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) { 6856 if (C->getNameModifier() == OMPD_unknown || 6857 C->getNameModifier() == OMPD_parallel) { 6858 IfClause = C; 6859 break; 6860 } 6861 } 6862 if (IfClause) { 6863 const Expr *Cond = IfClause->getCondition(); 6864 bool Result; 6865 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 6866 if (!Result) 6867 return CGF.Builder.getInt32(1); 6868 } else { 6869 CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange()); 6870 if (const auto *PreInit = 6871 cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) { 6872 for (const auto *I : PreInit->decls()) { 6873 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6874 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6875 } else { 6876 CodeGenFunction::AutoVarEmission Emission = 6877 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6878 CGF.EmitAutoVarCleanups(Emission); 6879 } 6880 } 6881 } 6882 CondVal = CGF.EvaluateExprAsBool(Cond); 6883 } 6884 } 6885 } 6886 // Check the value of num_threads clause iff if clause was not specified 6887 // or is not evaluated to false. 6888 if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) { 6889 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6890 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6891 const auto *NumThreadsClause = 6892 Dir->getSingleClause<OMPNumThreadsClause>(); 6893 CodeGenFunction::LexicalScope Scope( 6894 CGF, NumThreadsClause->getNumThreads()->getSourceRange()); 6895 if (const auto *PreInit = 6896 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) { 6897 for (const auto *I : PreInit->decls()) { 6898 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6899 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6900 } else { 6901 CodeGenFunction::AutoVarEmission Emission = 6902 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6903 CGF.EmitAutoVarCleanups(Emission); 6904 } 6905 } 6906 } 6907 NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads()); 6908 NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, 6909 /*isSigned=*/false); 6910 if (DefaultThreadLimitVal) 6911 NumThreads = CGF.Builder.CreateSelect( 6912 CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads), 6913 DefaultThreadLimitVal, NumThreads); 6914 } else { 6915 NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal 6916 : CGF.Builder.getInt32(0); 6917 } 6918 // Process condition of the if clause. 6919 if (CondVal) { 6920 NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads, 6921 CGF.Builder.getInt32(1)); 6922 } 6923 return NumThreads; 6924 } 6925 if (isOpenMPSimdDirective(Dir->getDirectiveKind())) 6926 return CGF.Builder.getInt32(1); 6927 return DefaultThreadLimitVal; 6928 } 6929 return DefaultThreadLimitVal ? DefaultThreadLimitVal 6930 : CGF.Builder.getInt32(0); 6931 } 6932 6933 const Expr *CGOpenMPRuntime::getNumThreadsExprForTargetDirective( 6934 CodeGenFunction &CGF, const OMPExecutableDirective &D, 6935 int32_t &DefaultVal) { 6936 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6937 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 6938 "Expected target-based executable directive."); 6939 6940 switch (DirectiveKind) { 6941 case OMPD_target: 6942 // Teams have no clause thread_limit 6943 return nullptr; 6944 case OMPD_target_teams: 6945 case OMPD_target_teams_distribute: 6946 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 6947 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 6948 const Expr *ThreadLimit = ThreadLimitClause->getThreadLimit(); 6949 if (ThreadLimit->isIntegerConstantExpr(CGF.getContext())) 6950 if (auto Constant = 6951 ThreadLimit->getIntegerConstantExpr(CGF.getContext())) 6952 DefaultVal = Constant->getExtValue(); 6953 return ThreadLimit; 6954 } 6955 return nullptr; 6956 case OMPD_target_parallel: 6957 case OMPD_target_parallel_for: 6958 case OMPD_target_parallel_for_simd: 6959 case OMPD_target_teams_distribute_parallel_for: 6960 case OMPD_target_teams_distribute_parallel_for_simd: { 6961 Expr *ThreadLimit = nullptr; 6962 Expr *NumThreads = nullptr; 6963 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 6964 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 6965 ThreadLimit = ThreadLimitClause->getThreadLimit(); 6966 if (ThreadLimit->isIntegerConstantExpr(CGF.getContext())) 6967 if (auto Constant = 6968 ThreadLimit->getIntegerConstantExpr(CGF.getContext())) 6969 DefaultVal = Constant->getExtValue(); 6970 } 6971 if (D.hasClausesOfKind<OMPNumThreadsClause>()) { 6972 const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>(); 6973 NumThreads = NumThreadsClause->getNumThreads(); 6974 if (NumThreads->isIntegerConstantExpr(CGF.getContext())) { 6975 if (auto Constant = 6976 NumThreads->getIntegerConstantExpr(CGF.getContext())) { 6977 if (Constant->getExtValue() < DefaultVal) { 6978 DefaultVal = Constant->getExtValue(); 6979 ThreadLimit = NumThreads; 6980 } 6981 } 6982 } 6983 } 6984 return ThreadLimit; 6985 } 6986 case OMPD_target_teams_distribute_simd: 6987 case OMPD_target_simd: 6988 DefaultVal = 1; 6989 return nullptr; 6990 case OMPD_parallel: 6991 case OMPD_for: 6992 case OMPD_parallel_for: 6993 case OMPD_parallel_master: 6994 case OMPD_parallel_sections: 6995 case OMPD_for_simd: 6996 case OMPD_parallel_for_simd: 6997 case OMPD_cancel: 6998 case OMPD_cancellation_point: 6999 case OMPD_ordered: 7000 case OMPD_threadprivate: 7001 case OMPD_allocate: 7002 case OMPD_task: 7003 case OMPD_simd: 7004 case OMPD_tile: 7005 case OMPD_unroll: 7006 case OMPD_sections: 7007 case OMPD_section: 7008 case OMPD_single: 7009 case OMPD_master: 7010 case OMPD_critical: 7011 case OMPD_taskyield: 7012 case OMPD_barrier: 7013 case OMPD_taskwait: 7014 case OMPD_taskgroup: 7015 case OMPD_atomic: 7016 case OMPD_flush: 7017 case OMPD_depobj: 7018 case OMPD_scan: 7019 case OMPD_teams: 7020 case OMPD_target_data: 7021 case OMPD_target_exit_data: 7022 case OMPD_target_enter_data: 7023 case OMPD_distribute: 7024 case OMPD_distribute_simd: 7025 case OMPD_distribute_parallel_for: 7026 case OMPD_distribute_parallel_for_simd: 7027 case OMPD_teams_distribute: 7028 case OMPD_teams_distribute_simd: 7029 case OMPD_teams_distribute_parallel_for: 7030 case OMPD_teams_distribute_parallel_for_simd: 7031 case OMPD_target_update: 7032 case OMPD_declare_simd: 7033 case OMPD_declare_variant: 7034 case OMPD_begin_declare_variant: 7035 case OMPD_end_declare_variant: 7036 case OMPD_declare_target: 7037 case OMPD_end_declare_target: 7038 case OMPD_declare_reduction: 7039 case OMPD_declare_mapper: 7040 case OMPD_taskloop: 7041 case OMPD_taskloop_simd: 7042 case OMPD_master_taskloop: 7043 case OMPD_master_taskloop_simd: 7044 case OMPD_parallel_master_taskloop: 7045 case OMPD_parallel_master_taskloop_simd: 7046 case OMPD_requires: 7047 case OMPD_unknown: 7048 break; 7049 default: 7050 break; 7051 } 7052 llvm_unreachable("Unsupported directive kind."); 7053 } 7054 7055 llvm::Value *CGOpenMPRuntime::emitNumThreadsForTargetDirective( 7056 CodeGenFunction &CGF, const OMPExecutableDirective &D) { 7057 assert(!CGF.getLangOpts().OpenMPIsDevice && 7058 "Clauses associated with the teams directive expected to be emitted " 7059 "only for the host!"); 7060 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 7061 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 7062 "Expected target-based executable directive."); 7063 CGBuilderTy &Bld = CGF.Builder; 7064 llvm::Value *ThreadLimitVal = nullptr; 7065 llvm::Value *NumThreadsVal = nullptr; 7066 switch (DirectiveKind) { 7067 case OMPD_target: { 7068 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 7069 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7070 return NumThreads; 7071 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7072 CGF.getContext(), CS->getCapturedStmt()); 7073 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 7074 if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) { 7075 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 7076 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 7077 const auto *ThreadLimitClause = 7078 Dir->getSingleClause<OMPThreadLimitClause>(); 7079 CodeGenFunction::LexicalScope Scope( 7080 CGF, ThreadLimitClause->getThreadLimit()->getSourceRange()); 7081 if (const auto *PreInit = 7082 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) { 7083 for (const auto *I : PreInit->decls()) { 7084 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 7085 CGF.EmitVarDecl(cast<VarDecl>(*I)); 7086 } else { 7087 CodeGenFunction::AutoVarEmission Emission = 7088 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 7089 CGF.EmitAutoVarCleanups(Emission); 7090 } 7091 } 7092 } 7093 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7094 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7095 ThreadLimitVal = 7096 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7097 } 7098 if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) && 7099 !isOpenMPDistributeDirective(Dir->getDirectiveKind())) { 7100 CS = Dir->getInnermostCapturedStmt(); 7101 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7102 CGF.getContext(), CS->getCapturedStmt()); 7103 Dir = dyn_cast_or_null<OMPExecutableDirective>(Child); 7104 } 7105 if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) && 7106 !isOpenMPSimdDirective(Dir->getDirectiveKind())) { 7107 CS = Dir->getInnermostCapturedStmt(); 7108 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7109 return NumThreads; 7110 } 7111 if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind())) 7112 return Bld.getInt32(1); 7113 } 7114 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 7115 } 7116 case OMPD_target_teams: { 7117 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7118 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7119 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7120 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7121 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7122 ThreadLimitVal = 7123 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7124 } 7125 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 7126 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7127 return NumThreads; 7128 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7129 CGF.getContext(), CS->getCapturedStmt()); 7130 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 7131 if (Dir->getDirectiveKind() == OMPD_distribute) { 7132 CS = Dir->getInnermostCapturedStmt(); 7133 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7134 return NumThreads; 7135 } 7136 } 7137 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 7138 } 7139 case OMPD_target_teams_distribute: 7140 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7141 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7142 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7143 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7144 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7145 ThreadLimitVal = 7146 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7147 } 7148 return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal); 7149 case OMPD_target_parallel: 7150 case OMPD_target_parallel_for: 7151 case OMPD_target_parallel_for_simd: 7152 case OMPD_target_teams_distribute_parallel_for: 7153 case OMPD_target_teams_distribute_parallel_for_simd: { 7154 llvm::Value *CondVal = nullptr; 7155 // Handle if clause. If if clause present, the number of threads is 7156 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 7157 if (D.hasClausesOfKind<OMPIfClause>()) { 7158 const OMPIfClause *IfClause = nullptr; 7159 for (const auto *C : D.getClausesOfKind<OMPIfClause>()) { 7160 if (C->getNameModifier() == OMPD_unknown || 7161 C->getNameModifier() == OMPD_parallel) { 7162 IfClause = C; 7163 break; 7164 } 7165 } 7166 if (IfClause) { 7167 const Expr *Cond = IfClause->getCondition(); 7168 bool Result; 7169 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 7170 if (!Result) 7171 return Bld.getInt32(1); 7172 } else { 7173 CodeGenFunction::RunCleanupsScope Scope(CGF); 7174 CondVal = CGF.EvaluateExprAsBool(Cond); 7175 } 7176 } 7177 } 7178 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7179 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7180 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7181 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7182 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7183 ThreadLimitVal = 7184 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7185 } 7186 if (D.hasClausesOfKind<OMPNumThreadsClause>()) { 7187 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF); 7188 const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>(); 7189 llvm::Value *NumThreads = CGF.EmitScalarExpr( 7190 NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true); 7191 NumThreadsVal = 7192 Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false); 7193 ThreadLimitVal = ThreadLimitVal 7194 ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal, 7195 ThreadLimitVal), 7196 NumThreadsVal, ThreadLimitVal) 7197 : NumThreadsVal; 7198 } 7199 if (!ThreadLimitVal) 7200 ThreadLimitVal = Bld.getInt32(0); 7201 if (CondVal) 7202 return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1)); 7203 return ThreadLimitVal; 7204 } 7205 case OMPD_target_teams_distribute_simd: 7206 case OMPD_target_simd: 7207 return Bld.getInt32(1); 7208 case OMPD_parallel: 7209 case OMPD_for: 7210 case OMPD_parallel_for: 7211 case OMPD_parallel_master: 7212 case OMPD_parallel_sections: 7213 case OMPD_for_simd: 7214 case OMPD_parallel_for_simd: 7215 case OMPD_cancel: 7216 case OMPD_cancellation_point: 7217 case OMPD_ordered: 7218 case OMPD_threadprivate: 7219 case OMPD_allocate: 7220 case OMPD_task: 7221 case OMPD_simd: 7222 case OMPD_tile: 7223 case OMPD_unroll: 7224 case OMPD_sections: 7225 case OMPD_section: 7226 case OMPD_single: 7227 case OMPD_master: 7228 case OMPD_critical: 7229 case OMPD_taskyield: 7230 case OMPD_barrier: 7231 case OMPD_taskwait: 7232 case OMPD_taskgroup: 7233 case OMPD_atomic: 7234 case OMPD_flush: 7235 case OMPD_depobj: 7236 case OMPD_scan: 7237 case OMPD_teams: 7238 case OMPD_target_data: 7239 case OMPD_target_exit_data: 7240 case OMPD_target_enter_data: 7241 case OMPD_distribute: 7242 case OMPD_distribute_simd: 7243 case OMPD_distribute_parallel_for: 7244 case OMPD_distribute_parallel_for_simd: 7245 case OMPD_teams_distribute: 7246 case OMPD_teams_distribute_simd: 7247 case OMPD_teams_distribute_parallel_for: 7248 case OMPD_teams_distribute_parallel_for_simd: 7249 case OMPD_target_update: 7250 case OMPD_declare_simd: 7251 case OMPD_declare_variant: 7252 case OMPD_begin_declare_variant: 7253 case OMPD_end_declare_variant: 7254 case OMPD_declare_target: 7255 case OMPD_end_declare_target: 7256 case OMPD_declare_reduction: 7257 case OMPD_declare_mapper: 7258 case OMPD_taskloop: 7259 case OMPD_taskloop_simd: 7260 case OMPD_master_taskloop: 7261 case OMPD_master_taskloop_simd: 7262 case OMPD_parallel_master_taskloop: 7263 case OMPD_parallel_master_taskloop_simd: 7264 case OMPD_requires: 7265 case OMPD_metadirective: 7266 case OMPD_unknown: 7267 break; 7268 default: 7269 break; 7270 } 7271 llvm_unreachable("Unsupported directive kind."); 7272 } 7273 7274 namespace { 7275 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 7276 7277 // Utility to handle information from clauses associated with a given 7278 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause). 7279 // It provides a convenient interface to obtain the information and generate 7280 // code for that information. 7281 class MappableExprsHandler { 7282 public: 7283 /// Values for bit flags used to specify the mapping type for 7284 /// offloading. 7285 enum OpenMPOffloadMappingFlags : uint64_t { 7286 /// No flags 7287 OMP_MAP_NONE = 0x0, 7288 /// Allocate memory on the device and move data from host to device. 7289 OMP_MAP_TO = 0x01, 7290 /// Allocate memory on the device and move data from device to host. 7291 OMP_MAP_FROM = 0x02, 7292 /// Always perform the requested mapping action on the element, even 7293 /// if it was already mapped before. 7294 OMP_MAP_ALWAYS = 0x04, 7295 /// Delete the element from the device environment, ignoring the 7296 /// current reference count associated with the element. 7297 OMP_MAP_DELETE = 0x08, 7298 /// The element being mapped is a pointer-pointee pair; both the 7299 /// pointer and the pointee should be mapped. 7300 OMP_MAP_PTR_AND_OBJ = 0x10, 7301 /// This flags signals that the base address of an entry should be 7302 /// passed to the target kernel as an argument. 7303 OMP_MAP_TARGET_PARAM = 0x20, 7304 /// Signal that the runtime library has to return the device pointer 7305 /// in the current position for the data being mapped. Used when we have the 7306 /// use_device_ptr or use_device_addr clause. 7307 OMP_MAP_RETURN_PARAM = 0x40, 7308 /// This flag signals that the reference being passed is a pointer to 7309 /// private data. 7310 OMP_MAP_PRIVATE = 0x80, 7311 /// Pass the element to the device by value. 7312 OMP_MAP_LITERAL = 0x100, 7313 /// Implicit map 7314 OMP_MAP_IMPLICIT = 0x200, 7315 /// Close is a hint to the runtime to allocate memory close to 7316 /// the target device. 7317 OMP_MAP_CLOSE = 0x400, 7318 /// 0x800 is reserved for compatibility with XLC. 7319 /// Produce a runtime error if the data is not already allocated. 7320 OMP_MAP_PRESENT = 0x1000, 7321 // Increment and decrement a separate reference counter so that the data 7322 // cannot be unmapped within the associated region. Thus, this flag is 7323 // intended to be used on 'target' and 'target data' directives because they 7324 // are inherently structured. It is not intended to be used on 'target 7325 // enter data' and 'target exit data' directives because they are inherently 7326 // dynamic. 7327 // This is an OpenMP extension for the sake of OpenACC support. 7328 OMP_MAP_OMPX_HOLD = 0x2000, 7329 /// Signal that the runtime library should use args as an array of 7330 /// descriptor_dim pointers and use args_size as dims. Used when we have 7331 /// non-contiguous list items in target update directive 7332 OMP_MAP_NON_CONTIG = 0x100000000000, 7333 /// The 16 MSBs of the flags indicate whether the entry is member of some 7334 /// struct/class. 7335 OMP_MAP_MEMBER_OF = 0xffff000000000000, 7336 LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF), 7337 }; 7338 7339 /// Get the offset of the OMP_MAP_MEMBER_OF field. 7340 static unsigned getFlagMemberOffset() { 7341 unsigned Offset = 0; 7342 for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1); 7343 Remain = Remain >> 1) 7344 Offset++; 7345 return Offset; 7346 } 7347 7348 /// Class that holds debugging information for a data mapping to be passed to 7349 /// the runtime library. 7350 class MappingExprInfo { 7351 /// The variable declaration used for the data mapping. 7352 const ValueDecl *MapDecl = nullptr; 7353 /// The original expression used in the map clause, or null if there is 7354 /// none. 7355 const Expr *MapExpr = nullptr; 7356 7357 public: 7358 MappingExprInfo(const ValueDecl *MapDecl, const Expr *MapExpr = nullptr) 7359 : MapDecl(MapDecl), MapExpr(MapExpr) {} 7360 7361 const ValueDecl *getMapDecl() const { return MapDecl; } 7362 const Expr *getMapExpr() const { return MapExpr; } 7363 }; 7364 7365 /// Class that associates information with a base pointer to be passed to the 7366 /// runtime library. 7367 class BasePointerInfo { 7368 /// The base pointer. 7369 llvm::Value *Ptr = nullptr; 7370 /// The base declaration that refers to this device pointer, or null if 7371 /// there is none. 7372 const ValueDecl *DevPtrDecl = nullptr; 7373 7374 public: 7375 BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr) 7376 : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {} 7377 llvm::Value *operator*() const { return Ptr; } 7378 const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; } 7379 void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; } 7380 }; 7381 7382 using MapExprsArrayTy = SmallVector<MappingExprInfo, 4>; 7383 using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>; 7384 using MapValuesArrayTy = SmallVector<llvm::Value *, 4>; 7385 using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>; 7386 using MapMappersArrayTy = SmallVector<const ValueDecl *, 4>; 7387 using MapDimArrayTy = SmallVector<uint64_t, 4>; 7388 using MapNonContiguousArrayTy = SmallVector<MapValuesArrayTy, 4>; 7389 7390 /// This structure contains combined information generated for mappable 7391 /// clauses, including base pointers, pointers, sizes, map types, user-defined 7392 /// mappers, and non-contiguous information. 7393 struct MapCombinedInfoTy { 7394 struct StructNonContiguousInfo { 7395 bool IsNonContiguous = false; 7396 MapDimArrayTy Dims; 7397 MapNonContiguousArrayTy Offsets; 7398 MapNonContiguousArrayTy Counts; 7399 MapNonContiguousArrayTy Strides; 7400 }; 7401 MapExprsArrayTy Exprs; 7402 MapBaseValuesArrayTy BasePointers; 7403 MapValuesArrayTy Pointers; 7404 MapValuesArrayTy Sizes; 7405 MapFlagsArrayTy Types; 7406 MapMappersArrayTy Mappers; 7407 StructNonContiguousInfo NonContigInfo; 7408 7409 /// Append arrays in \a CurInfo. 7410 void append(MapCombinedInfoTy &CurInfo) { 7411 Exprs.append(CurInfo.Exprs.begin(), CurInfo.Exprs.end()); 7412 BasePointers.append(CurInfo.BasePointers.begin(), 7413 CurInfo.BasePointers.end()); 7414 Pointers.append(CurInfo.Pointers.begin(), CurInfo.Pointers.end()); 7415 Sizes.append(CurInfo.Sizes.begin(), CurInfo.Sizes.end()); 7416 Types.append(CurInfo.Types.begin(), CurInfo.Types.end()); 7417 Mappers.append(CurInfo.Mappers.begin(), CurInfo.Mappers.end()); 7418 NonContigInfo.Dims.append(CurInfo.NonContigInfo.Dims.begin(), 7419 CurInfo.NonContigInfo.Dims.end()); 7420 NonContigInfo.Offsets.append(CurInfo.NonContigInfo.Offsets.begin(), 7421 CurInfo.NonContigInfo.Offsets.end()); 7422 NonContigInfo.Counts.append(CurInfo.NonContigInfo.Counts.begin(), 7423 CurInfo.NonContigInfo.Counts.end()); 7424 NonContigInfo.Strides.append(CurInfo.NonContigInfo.Strides.begin(), 7425 CurInfo.NonContigInfo.Strides.end()); 7426 } 7427 }; 7428 7429 /// Map between a struct and the its lowest & highest elements which have been 7430 /// mapped. 7431 /// [ValueDecl *] --> {LE(FieldIndex, Pointer), 7432 /// HE(FieldIndex, Pointer)} 7433 struct StructRangeInfoTy { 7434 MapCombinedInfoTy PreliminaryMapData; 7435 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = { 7436 0, Address::invalid()}; 7437 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = { 7438 0, Address::invalid()}; 7439 Address Base = Address::invalid(); 7440 Address LB = Address::invalid(); 7441 bool IsArraySection = false; 7442 bool HasCompleteRecord = false; 7443 }; 7444 7445 private: 7446 /// Kind that defines how a device pointer has to be returned. 7447 struct MapInfo { 7448 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 7449 OpenMPMapClauseKind MapType = OMPC_MAP_unknown; 7450 ArrayRef<OpenMPMapModifierKind> MapModifiers; 7451 ArrayRef<OpenMPMotionModifierKind> MotionModifiers; 7452 bool ReturnDevicePointer = false; 7453 bool IsImplicit = false; 7454 const ValueDecl *Mapper = nullptr; 7455 const Expr *VarRef = nullptr; 7456 bool ForDeviceAddr = false; 7457 7458 MapInfo() = default; 7459 MapInfo( 7460 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7461 OpenMPMapClauseKind MapType, 7462 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7463 ArrayRef<OpenMPMotionModifierKind> MotionModifiers, 7464 bool ReturnDevicePointer, bool IsImplicit, 7465 const ValueDecl *Mapper = nullptr, const Expr *VarRef = nullptr, 7466 bool ForDeviceAddr = false) 7467 : Components(Components), MapType(MapType), MapModifiers(MapModifiers), 7468 MotionModifiers(MotionModifiers), 7469 ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit), 7470 Mapper(Mapper), VarRef(VarRef), ForDeviceAddr(ForDeviceAddr) {} 7471 }; 7472 7473 /// If use_device_ptr or use_device_addr is used on a decl which is a struct 7474 /// member and there is no map information about it, then emission of that 7475 /// entry is deferred until the whole struct has been processed. 7476 struct DeferredDevicePtrEntryTy { 7477 const Expr *IE = nullptr; 7478 const ValueDecl *VD = nullptr; 7479 bool ForDeviceAddr = false; 7480 7481 DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD, 7482 bool ForDeviceAddr) 7483 : IE(IE), VD(VD), ForDeviceAddr(ForDeviceAddr) {} 7484 }; 7485 7486 /// The target directive from where the mappable clauses were extracted. It 7487 /// is either a executable directive or a user-defined mapper directive. 7488 llvm::PointerUnion<const OMPExecutableDirective *, 7489 const OMPDeclareMapperDecl *> 7490 CurDir; 7491 7492 /// Function the directive is being generated for. 7493 CodeGenFunction &CGF; 7494 7495 /// Set of all first private variables in the current directive. 7496 /// bool data is set to true if the variable is implicitly marked as 7497 /// firstprivate, false otherwise. 7498 llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls; 7499 7500 /// Map between device pointer declarations and their expression components. 7501 /// The key value for declarations in 'this' is null. 7502 llvm::DenseMap< 7503 const ValueDecl *, 7504 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>> 7505 DevPointersMap; 7506 7507 /// Map between lambda declarations and their map type. 7508 llvm::DenseMap<const ValueDecl *, const OMPMapClause *> LambdasMap; 7509 7510 llvm::Value *getExprTypeSize(const Expr *E) const { 7511 QualType ExprTy = E->getType().getCanonicalType(); 7512 7513 // Calculate the size for array shaping expression. 7514 if (const auto *OAE = dyn_cast<OMPArrayShapingExpr>(E)) { 7515 llvm::Value *Size = 7516 CGF.getTypeSize(OAE->getBase()->getType()->getPointeeType()); 7517 for (const Expr *SE : OAE->getDimensions()) { 7518 llvm::Value *Sz = CGF.EmitScalarExpr(SE); 7519 Sz = CGF.EmitScalarConversion(Sz, SE->getType(), 7520 CGF.getContext().getSizeType(), 7521 SE->getExprLoc()); 7522 Size = CGF.Builder.CreateNUWMul(Size, Sz); 7523 } 7524 return Size; 7525 } 7526 7527 // Reference types are ignored for mapping purposes. 7528 if (const auto *RefTy = ExprTy->getAs<ReferenceType>()) 7529 ExprTy = RefTy->getPointeeType().getCanonicalType(); 7530 7531 // Given that an array section is considered a built-in type, we need to 7532 // do the calculation based on the length of the section instead of relying 7533 // on CGF.getTypeSize(E->getType()). 7534 if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) { 7535 QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType( 7536 OAE->getBase()->IgnoreParenImpCasts()) 7537 .getCanonicalType(); 7538 7539 // If there is no length associated with the expression and lower bound is 7540 // not specified too, that means we are using the whole length of the 7541 // base. 7542 if (!OAE->getLength() && OAE->getColonLocFirst().isValid() && 7543 !OAE->getLowerBound()) 7544 return CGF.getTypeSize(BaseTy); 7545 7546 llvm::Value *ElemSize; 7547 if (const auto *PTy = BaseTy->getAs<PointerType>()) { 7548 ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType()); 7549 } else { 7550 const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr()); 7551 assert(ATy && "Expecting array type if not a pointer type."); 7552 ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType()); 7553 } 7554 7555 // If we don't have a length at this point, that is because we have an 7556 // array section with a single element. 7557 if (!OAE->getLength() && OAE->getColonLocFirst().isInvalid()) 7558 return ElemSize; 7559 7560 if (const Expr *LenExpr = OAE->getLength()) { 7561 llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr); 7562 LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(), 7563 CGF.getContext().getSizeType(), 7564 LenExpr->getExprLoc()); 7565 return CGF.Builder.CreateNUWMul(LengthVal, ElemSize); 7566 } 7567 assert(!OAE->getLength() && OAE->getColonLocFirst().isValid() && 7568 OAE->getLowerBound() && "expected array_section[lb:]."); 7569 // Size = sizetype - lb * elemtype; 7570 llvm::Value *LengthVal = CGF.getTypeSize(BaseTy); 7571 llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound()); 7572 LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(), 7573 CGF.getContext().getSizeType(), 7574 OAE->getLowerBound()->getExprLoc()); 7575 LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize); 7576 llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal); 7577 llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal); 7578 LengthVal = CGF.Builder.CreateSelect( 7579 Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0)); 7580 return LengthVal; 7581 } 7582 return CGF.getTypeSize(ExprTy); 7583 } 7584 7585 /// Return the corresponding bits for a given map clause modifier. Add 7586 /// a flag marking the map as a pointer if requested. Add a flag marking the 7587 /// map as the first one of a series of maps that relate to the same map 7588 /// expression. 7589 OpenMPOffloadMappingFlags getMapTypeBits( 7590 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers, 7591 ArrayRef<OpenMPMotionModifierKind> MotionModifiers, bool IsImplicit, 7592 bool AddPtrFlag, bool AddIsTargetParamFlag, bool IsNonContiguous) const { 7593 OpenMPOffloadMappingFlags Bits = 7594 IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE; 7595 switch (MapType) { 7596 case OMPC_MAP_alloc: 7597 case OMPC_MAP_release: 7598 // alloc and release is the default behavior in the runtime library, i.e. 7599 // if we don't pass any bits alloc/release that is what the runtime is 7600 // going to do. Therefore, we don't need to signal anything for these two 7601 // type modifiers. 7602 break; 7603 case OMPC_MAP_to: 7604 Bits |= OMP_MAP_TO; 7605 break; 7606 case OMPC_MAP_from: 7607 Bits |= OMP_MAP_FROM; 7608 break; 7609 case OMPC_MAP_tofrom: 7610 Bits |= OMP_MAP_TO | OMP_MAP_FROM; 7611 break; 7612 case OMPC_MAP_delete: 7613 Bits |= OMP_MAP_DELETE; 7614 break; 7615 case OMPC_MAP_unknown: 7616 llvm_unreachable("Unexpected map type!"); 7617 } 7618 if (AddPtrFlag) 7619 Bits |= OMP_MAP_PTR_AND_OBJ; 7620 if (AddIsTargetParamFlag) 7621 Bits |= OMP_MAP_TARGET_PARAM; 7622 if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_always)) 7623 Bits |= OMP_MAP_ALWAYS; 7624 if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_close)) 7625 Bits |= OMP_MAP_CLOSE; 7626 if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_present) || 7627 llvm::is_contained(MotionModifiers, OMPC_MOTION_MODIFIER_present)) 7628 Bits |= OMP_MAP_PRESENT; 7629 if (llvm::is_contained(MapModifiers, OMPC_MAP_MODIFIER_ompx_hold)) 7630 Bits |= OMP_MAP_OMPX_HOLD; 7631 if (IsNonContiguous) 7632 Bits |= OMP_MAP_NON_CONTIG; 7633 return Bits; 7634 } 7635 7636 /// Return true if the provided expression is a final array section. A 7637 /// final array section, is one whose length can't be proved to be one. 7638 bool isFinalArraySectionExpression(const Expr *E) const { 7639 const auto *OASE = dyn_cast<OMPArraySectionExpr>(E); 7640 7641 // It is not an array section and therefore not a unity-size one. 7642 if (!OASE) 7643 return false; 7644 7645 // An array section with no colon always refer to a single element. 7646 if (OASE->getColonLocFirst().isInvalid()) 7647 return false; 7648 7649 const Expr *Length = OASE->getLength(); 7650 7651 // If we don't have a length we have to check if the array has size 1 7652 // for this dimension. Also, we should always expect a length if the 7653 // base type is pointer. 7654 if (!Length) { 7655 QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType( 7656 OASE->getBase()->IgnoreParenImpCasts()) 7657 .getCanonicalType(); 7658 if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr())) 7659 return ATy->getSize().getSExtValue() != 1; 7660 // If we don't have a constant dimension length, we have to consider 7661 // the current section as having any size, so it is not necessarily 7662 // unitary. If it happen to be unity size, that's user fault. 7663 return true; 7664 } 7665 7666 // Check if the length evaluates to 1. 7667 Expr::EvalResult Result; 7668 if (!Length->EvaluateAsInt(Result, CGF.getContext())) 7669 return true; // Can have more that size 1. 7670 7671 llvm::APSInt ConstLength = Result.Val.getInt(); 7672 return ConstLength.getSExtValue() != 1; 7673 } 7674 7675 /// Generate the base pointers, section pointers, sizes, map type bits, and 7676 /// user-defined mappers (all included in \a CombinedInfo) for the provided 7677 /// map type, map or motion modifiers, and expression components. 7678 /// \a IsFirstComponent should be set to true if the provided set of 7679 /// components is the first associated with a capture. 7680 void generateInfoForComponentList( 7681 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers, 7682 ArrayRef<OpenMPMotionModifierKind> MotionModifiers, 7683 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7684 MapCombinedInfoTy &CombinedInfo, StructRangeInfoTy &PartialStruct, 7685 bool IsFirstComponentList, bool IsImplicit, 7686 const ValueDecl *Mapper = nullptr, bool ForDeviceAddr = false, 7687 const ValueDecl *BaseDecl = nullptr, const Expr *MapExpr = nullptr, 7688 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 7689 OverlappedElements = llvm::None) const { 7690 // The following summarizes what has to be generated for each map and the 7691 // types below. The generated information is expressed in this order: 7692 // base pointer, section pointer, size, flags 7693 // (to add to the ones that come from the map type and modifier). 7694 // 7695 // double d; 7696 // int i[100]; 7697 // float *p; 7698 // 7699 // struct S1 { 7700 // int i; 7701 // float f[50]; 7702 // } 7703 // struct S2 { 7704 // int i; 7705 // float f[50]; 7706 // S1 s; 7707 // double *p; 7708 // struct S2 *ps; 7709 // int &ref; 7710 // } 7711 // S2 s; 7712 // S2 *ps; 7713 // 7714 // map(d) 7715 // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM 7716 // 7717 // map(i) 7718 // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM 7719 // 7720 // map(i[1:23]) 7721 // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM 7722 // 7723 // map(p) 7724 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM 7725 // 7726 // map(p[1:24]) 7727 // &p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM | PTR_AND_OBJ 7728 // in unified shared memory mode or for local pointers 7729 // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM 7730 // 7731 // map(s) 7732 // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM 7733 // 7734 // map(s.i) 7735 // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM 7736 // 7737 // map(s.s.f) 7738 // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7739 // 7740 // map(s.p) 7741 // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM 7742 // 7743 // map(to: s.p[:22]) 7744 // &s, &(s.p), sizeof(double*), TARGET_PARAM (*) 7745 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**) 7746 // &(s.p), &(s.p[0]), 22*sizeof(double), 7747 // MEMBER_OF(1) | PTR_AND_OBJ | TO (***) 7748 // (*) alloc space for struct members, only this is a target parameter 7749 // (**) map the pointer (nothing to be mapped in this example) (the compiler 7750 // optimizes this entry out, same in the examples below) 7751 // (***) map the pointee (map: to) 7752 // 7753 // map(to: s.ref) 7754 // &s, &(s.ref), sizeof(int*), TARGET_PARAM (*) 7755 // &s, &(s.ref), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | TO (***) 7756 // (*) alloc space for struct members, only this is a target parameter 7757 // (**) map the pointer (nothing to be mapped in this example) (the compiler 7758 // optimizes this entry out, same in the examples below) 7759 // (***) map the pointee (map: to) 7760 // 7761 // map(s.ps) 7762 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7763 // 7764 // map(from: s.ps->s.i) 7765 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7766 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7767 // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7768 // 7769 // map(to: s.ps->ps) 7770 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7771 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7772 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | TO 7773 // 7774 // map(s.ps->ps->ps) 7775 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7776 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7777 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7778 // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7779 // 7780 // map(to: s.ps->ps->s.f[:22]) 7781 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7782 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7783 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7784 // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7785 // 7786 // map(ps) 7787 // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM 7788 // 7789 // map(ps->i) 7790 // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM 7791 // 7792 // map(ps->s.f) 7793 // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7794 // 7795 // map(from: ps->p) 7796 // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM 7797 // 7798 // map(to: ps->p[:22]) 7799 // ps, &(ps->p), sizeof(double*), TARGET_PARAM 7800 // ps, &(ps->p), sizeof(double*), MEMBER_OF(1) 7801 // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO 7802 // 7803 // map(ps->ps) 7804 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7805 // 7806 // map(from: ps->ps->s.i) 7807 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7808 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7809 // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7810 // 7811 // map(from: ps->ps->ps) 7812 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7813 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7814 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7815 // 7816 // map(ps->ps->ps->ps) 7817 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7818 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7819 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7820 // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7821 // 7822 // map(to: ps->ps->ps->s.f[:22]) 7823 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7824 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7825 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7826 // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7827 // 7828 // map(to: s.f[:22]) map(from: s.p[:33]) 7829 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) + 7830 // sizeof(double*) (**), TARGET_PARAM 7831 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO 7832 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) 7833 // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7834 // (*) allocate contiguous space needed to fit all mapped members even if 7835 // we allocate space for members not mapped (in this example, 7836 // s.f[22..49] and s.s are not mapped, yet we must allocate space for 7837 // them as well because they fall between &s.f[0] and &s.p) 7838 // 7839 // map(from: s.f[:22]) map(to: ps->p[:33]) 7840 // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM 7841 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7842 // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*) 7843 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO 7844 // (*) the struct this entry pertains to is the 2nd element in the list of 7845 // arguments, hence MEMBER_OF(2) 7846 // 7847 // map(from: s.f[:22], s.s) map(to: ps->p[:33]) 7848 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM 7849 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM 7850 // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM 7851 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7852 // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*) 7853 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO 7854 // (*) the struct this entry pertains to is the 4th element in the list 7855 // of arguments, hence MEMBER_OF(4) 7856 7857 // Track if the map information being generated is the first for a capture. 7858 bool IsCaptureFirstInfo = IsFirstComponentList; 7859 // When the variable is on a declare target link or in a to clause with 7860 // unified memory, a reference is needed to hold the host/device address 7861 // of the variable. 7862 bool RequiresReference = false; 7863 7864 // Scan the components from the base to the complete expression. 7865 auto CI = Components.rbegin(); 7866 auto CE = Components.rend(); 7867 auto I = CI; 7868 7869 // Track if the map information being generated is the first for a list of 7870 // components. 7871 bool IsExpressionFirstInfo = true; 7872 bool FirstPointerInComplexData = false; 7873 Address BP = Address::invalid(); 7874 const Expr *AssocExpr = I->getAssociatedExpression(); 7875 const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr); 7876 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 7877 const auto *OAShE = dyn_cast<OMPArrayShapingExpr>(AssocExpr); 7878 7879 if (isa<MemberExpr>(AssocExpr)) { 7880 // The base is the 'this' pointer. The content of the pointer is going 7881 // to be the base of the field being mapped. 7882 BP = CGF.LoadCXXThisAddress(); 7883 } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) || 7884 (OASE && 7885 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) { 7886 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF); 7887 } else if (OAShE && 7888 isa<CXXThisExpr>(OAShE->getBase()->IgnoreParenCasts())) { 7889 BP = Address::deprecated( 7890 CGF.EmitScalarExpr(OAShE->getBase()), 7891 CGF.getContext().getTypeAlignInChars(OAShE->getBase()->getType())); 7892 } else { 7893 // The base is the reference to the variable. 7894 // BP = &Var. 7895 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF); 7896 if (const auto *VD = 7897 dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) { 7898 if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 7899 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) { 7900 if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 7901 (*Res == OMPDeclareTargetDeclAttr::MT_To && 7902 CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) { 7903 RequiresReference = true; 7904 BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 7905 } 7906 } 7907 } 7908 7909 // If the variable is a pointer and is being dereferenced (i.e. is not 7910 // the last component), the base has to be the pointer itself, not its 7911 // reference. References are ignored for mapping purposes. 7912 QualType Ty = 7913 I->getAssociatedDeclaration()->getType().getNonReferenceType(); 7914 if (Ty->isAnyPointerType() && std::next(I) != CE) { 7915 // No need to generate individual map information for the pointer, it 7916 // can be associated with the combined storage if shared memory mode is 7917 // active or the base declaration is not global variable. 7918 const auto *VD = dyn_cast<VarDecl>(I->getAssociatedDeclaration()); 7919 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() || 7920 !VD || VD->hasLocalStorage()) 7921 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>()); 7922 else 7923 FirstPointerInComplexData = true; 7924 ++I; 7925 } 7926 } 7927 7928 // Track whether a component of the list should be marked as MEMBER_OF some 7929 // combined entry (for partial structs). Only the first PTR_AND_OBJ entry 7930 // in a component list should be marked as MEMBER_OF, all subsequent entries 7931 // do not belong to the base struct. E.g. 7932 // struct S2 s; 7933 // s.ps->ps->ps->f[:] 7934 // (1) (2) (3) (4) 7935 // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a 7936 // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3) 7937 // is the pointee of ps(2) which is not member of struct s, so it should not 7938 // be marked as such (it is still PTR_AND_OBJ). 7939 // The variable is initialized to false so that PTR_AND_OBJ entries which 7940 // are not struct members are not considered (e.g. array of pointers to 7941 // data). 7942 bool ShouldBeMemberOf = false; 7943 7944 // Variable keeping track of whether or not we have encountered a component 7945 // in the component list which is a member expression. Useful when we have a 7946 // pointer or a final array section, in which case it is the previous 7947 // component in the list which tells us whether we have a member expression. 7948 // E.g. X.f[:] 7949 // While processing the final array section "[:]" it is "f" which tells us 7950 // whether we are dealing with a member of a declared struct. 7951 const MemberExpr *EncounteredME = nullptr; 7952 7953 // Track for the total number of dimension. Start from one for the dummy 7954 // dimension. 7955 uint64_t DimSize = 1; 7956 7957 bool IsNonContiguous = CombinedInfo.NonContigInfo.IsNonContiguous; 7958 bool IsPrevMemberReference = false; 7959 7960 for (; I != CE; ++I) { 7961 // If the current component is member of a struct (parent struct) mark it. 7962 if (!EncounteredME) { 7963 EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression()); 7964 // If we encounter a PTR_AND_OBJ entry from now on it should be marked 7965 // as MEMBER_OF the parent struct. 7966 if (EncounteredME) { 7967 ShouldBeMemberOf = true; 7968 // Do not emit as complex pointer if this is actually not array-like 7969 // expression. 7970 if (FirstPointerInComplexData) { 7971 QualType Ty = std::prev(I) 7972 ->getAssociatedDeclaration() 7973 ->getType() 7974 .getNonReferenceType(); 7975 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>()); 7976 FirstPointerInComplexData = false; 7977 } 7978 } 7979 } 7980 7981 auto Next = std::next(I); 7982 7983 // We need to generate the addresses and sizes if this is the last 7984 // component, if the component is a pointer or if it is an array section 7985 // whose length can't be proved to be one. If this is a pointer, it 7986 // becomes the base address for the following components. 7987 7988 // A final array section, is one whose length can't be proved to be one. 7989 // If the map item is non-contiguous then we don't treat any array section 7990 // as final array section. 7991 bool IsFinalArraySection = 7992 !IsNonContiguous && 7993 isFinalArraySectionExpression(I->getAssociatedExpression()); 7994 7995 // If we have a declaration for the mapping use that, otherwise use 7996 // the base declaration of the map clause. 7997 const ValueDecl *MapDecl = (I->getAssociatedDeclaration()) 7998 ? I->getAssociatedDeclaration() 7999 : BaseDecl; 8000 MapExpr = (I->getAssociatedExpression()) ? I->getAssociatedExpression() 8001 : MapExpr; 8002 8003 // Get information on whether the element is a pointer. Have to do a 8004 // special treatment for array sections given that they are built-in 8005 // types. 8006 const auto *OASE = 8007 dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression()); 8008 const auto *OAShE = 8009 dyn_cast<OMPArrayShapingExpr>(I->getAssociatedExpression()); 8010 const auto *UO = dyn_cast<UnaryOperator>(I->getAssociatedExpression()); 8011 const auto *BO = dyn_cast<BinaryOperator>(I->getAssociatedExpression()); 8012 bool IsPointer = 8013 OAShE || 8014 (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE) 8015 .getCanonicalType() 8016 ->isAnyPointerType()) || 8017 I->getAssociatedExpression()->getType()->isAnyPointerType(); 8018 bool IsMemberReference = isa<MemberExpr>(I->getAssociatedExpression()) && 8019 MapDecl && 8020 MapDecl->getType()->isLValueReferenceType(); 8021 bool IsNonDerefPointer = IsPointer && !UO && !BO && !IsNonContiguous; 8022 8023 if (OASE) 8024 ++DimSize; 8025 8026 if (Next == CE || IsMemberReference || IsNonDerefPointer || 8027 IsFinalArraySection) { 8028 // If this is not the last component, we expect the pointer to be 8029 // associated with an array expression or member expression. 8030 assert((Next == CE || 8031 isa<MemberExpr>(Next->getAssociatedExpression()) || 8032 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) || 8033 isa<OMPArraySectionExpr>(Next->getAssociatedExpression()) || 8034 isa<OMPArrayShapingExpr>(Next->getAssociatedExpression()) || 8035 isa<UnaryOperator>(Next->getAssociatedExpression()) || 8036 isa<BinaryOperator>(Next->getAssociatedExpression())) && 8037 "Unexpected expression"); 8038 8039 Address LB = Address::invalid(); 8040 Address LowestElem = Address::invalid(); 8041 auto &&EmitMemberExprBase = [](CodeGenFunction &CGF, 8042 const MemberExpr *E) { 8043 const Expr *BaseExpr = E->getBase(); 8044 // If this is s.x, emit s as an lvalue. If it is s->x, emit s as a 8045 // scalar. 8046 LValue BaseLV; 8047 if (E->isArrow()) { 8048 LValueBaseInfo BaseInfo; 8049 TBAAAccessInfo TBAAInfo; 8050 Address Addr = 8051 CGF.EmitPointerWithAlignment(BaseExpr, &BaseInfo, &TBAAInfo); 8052 QualType PtrTy = BaseExpr->getType()->getPointeeType(); 8053 BaseLV = CGF.MakeAddrLValue(Addr, PtrTy, BaseInfo, TBAAInfo); 8054 } else { 8055 BaseLV = CGF.EmitOMPSharedLValue(BaseExpr); 8056 } 8057 return BaseLV; 8058 }; 8059 if (OAShE) { 8060 LowestElem = LB = 8061 Address::deprecated(CGF.EmitScalarExpr(OAShE->getBase()), 8062 CGF.getContext().getTypeAlignInChars( 8063 OAShE->getBase()->getType())); 8064 } else if (IsMemberReference) { 8065 const auto *ME = cast<MemberExpr>(I->getAssociatedExpression()); 8066 LValue BaseLVal = EmitMemberExprBase(CGF, ME); 8067 LowestElem = CGF.EmitLValueForFieldInitialization( 8068 BaseLVal, cast<FieldDecl>(MapDecl)) 8069 .getAddress(CGF); 8070 LB = CGF.EmitLoadOfReferenceLValue(LowestElem, MapDecl->getType()) 8071 .getAddress(CGF); 8072 } else { 8073 LowestElem = LB = 8074 CGF.EmitOMPSharedLValue(I->getAssociatedExpression()) 8075 .getAddress(CGF); 8076 } 8077 8078 // If this component is a pointer inside the base struct then we don't 8079 // need to create any entry for it - it will be combined with the object 8080 // it is pointing to into a single PTR_AND_OBJ entry. 8081 bool IsMemberPointerOrAddr = 8082 EncounteredME && 8083 (((IsPointer || ForDeviceAddr) && 8084 I->getAssociatedExpression() == EncounteredME) || 8085 (IsPrevMemberReference && !IsPointer) || 8086 (IsMemberReference && Next != CE && 8087 !Next->getAssociatedExpression()->getType()->isPointerType())); 8088 if (!OverlappedElements.empty() && Next == CE) { 8089 // Handle base element with the info for overlapped elements. 8090 assert(!PartialStruct.Base.isValid() && "The base element is set."); 8091 assert(!IsPointer && 8092 "Unexpected base element with the pointer type."); 8093 // Mark the whole struct as the struct that requires allocation on the 8094 // device. 8095 PartialStruct.LowestElem = {0, LowestElem}; 8096 CharUnits TypeSize = CGF.getContext().getTypeSizeInChars( 8097 I->getAssociatedExpression()->getType()); 8098 Address HB = CGF.Builder.CreateConstGEP( 8099 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8100 LowestElem, CGF.VoidPtrTy, CGF.Int8Ty), 8101 TypeSize.getQuantity() - 1); 8102 PartialStruct.HighestElem = { 8103 std::numeric_limits<decltype( 8104 PartialStruct.HighestElem.first)>::max(), 8105 HB}; 8106 PartialStruct.Base = BP; 8107 PartialStruct.LB = LB; 8108 assert( 8109 PartialStruct.PreliminaryMapData.BasePointers.empty() && 8110 "Overlapped elements must be used only once for the variable."); 8111 std::swap(PartialStruct.PreliminaryMapData, CombinedInfo); 8112 // Emit data for non-overlapped data. 8113 OpenMPOffloadMappingFlags Flags = 8114 OMP_MAP_MEMBER_OF | 8115 getMapTypeBits(MapType, MapModifiers, MotionModifiers, IsImplicit, 8116 /*AddPtrFlag=*/false, 8117 /*AddIsTargetParamFlag=*/false, IsNonContiguous); 8118 llvm::Value *Size = nullptr; 8119 // Do bitcopy of all non-overlapped structure elements. 8120 for (OMPClauseMappableExprCommon::MappableExprComponentListRef 8121 Component : OverlappedElements) { 8122 Address ComponentLB = Address::invalid(); 8123 for (const OMPClauseMappableExprCommon::MappableComponent &MC : 8124 Component) { 8125 if (const ValueDecl *VD = MC.getAssociatedDeclaration()) { 8126 const auto *FD = dyn_cast<FieldDecl>(VD); 8127 if (FD && FD->getType()->isLValueReferenceType()) { 8128 const auto *ME = 8129 cast<MemberExpr>(MC.getAssociatedExpression()); 8130 LValue BaseLVal = EmitMemberExprBase(CGF, ME); 8131 ComponentLB = 8132 CGF.EmitLValueForFieldInitialization(BaseLVal, FD) 8133 .getAddress(CGF); 8134 } else { 8135 ComponentLB = 8136 CGF.EmitOMPSharedLValue(MC.getAssociatedExpression()) 8137 .getAddress(CGF); 8138 } 8139 Size = CGF.Builder.CreatePtrDiff( 8140 CGF.Int8Ty, CGF.EmitCastToVoidPtr(ComponentLB.getPointer()), 8141 CGF.EmitCastToVoidPtr(LB.getPointer())); 8142 break; 8143 } 8144 } 8145 assert(Size && "Failed to determine structure size"); 8146 CombinedInfo.Exprs.emplace_back(MapDecl, MapExpr); 8147 CombinedInfo.BasePointers.push_back(BP.getPointer()); 8148 CombinedInfo.Pointers.push_back(LB.getPointer()); 8149 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 8150 Size, CGF.Int64Ty, /*isSigned=*/true)); 8151 CombinedInfo.Types.push_back(Flags); 8152 CombinedInfo.Mappers.push_back(nullptr); 8153 CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize 8154 : 1); 8155 LB = CGF.Builder.CreateConstGEP(ComponentLB, 1); 8156 } 8157 CombinedInfo.Exprs.emplace_back(MapDecl, MapExpr); 8158 CombinedInfo.BasePointers.push_back(BP.getPointer()); 8159 CombinedInfo.Pointers.push_back(LB.getPointer()); 8160 Size = CGF.Builder.CreatePtrDiff( 8161 CGF.Int8Ty, CGF.Builder.CreateConstGEP(HB, 1).getPointer(), 8162 CGF.EmitCastToVoidPtr(LB.getPointer())); 8163 CombinedInfo.Sizes.push_back( 8164 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 8165 CombinedInfo.Types.push_back(Flags); 8166 CombinedInfo.Mappers.push_back(nullptr); 8167 CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize 8168 : 1); 8169 break; 8170 } 8171 llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression()); 8172 if (!IsMemberPointerOrAddr || 8173 (Next == CE && MapType != OMPC_MAP_unknown)) { 8174 CombinedInfo.Exprs.emplace_back(MapDecl, MapExpr); 8175 CombinedInfo.BasePointers.push_back(BP.getPointer()); 8176 CombinedInfo.Pointers.push_back(LB.getPointer()); 8177 CombinedInfo.Sizes.push_back( 8178 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 8179 CombinedInfo.NonContigInfo.Dims.push_back(IsNonContiguous ? DimSize 8180 : 1); 8181 8182 // If Mapper is valid, the last component inherits the mapper. 8183 bool HasMapper = Mapper && Next == CE; 8184 CombinedInfo.Mappers.push_back(HasMapper ? Mapper : nullptr); 8185 8186 // We need to add a pointer flag for each map that comes from the 8187 // same expression except for the first one. We also need to signal 8188 // this map is the first one that relates with the current capture 8189 // (there is a set of entries for each capture). 8190 OpenMPOffloadMappingFlags Flags = getMapTypeBits( 8191 MapType, MapModifiers, MotionModifiers, IsImplicit, 8192 !IsExpressionFirstInfo || RequiresReference || 8193 FirstPointerInComplexData || IsMemberReference, 8194 IsCaptureFirstInfo && !RequiresReference, IsNonContiguous); 8195 8196 if (!IsExpressionFirstInfo || IsMemberReference) { 8197 // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well, 8198 // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags. 8199 if (IsPointer || (IsMemberReference && Next != CE)) 8200 Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS | 8201 OMP_MAP_DELETE | OMP_MAP_CLOSE); 8202 8203 if (ShouldBeMemberOf) { 8204 // Set placeholder value MEMBER_OF=FFFF to indicate that the flag 8205 // should be later updated with the correct value of MEMBER_OF. 8206 Flags |= OMP_MAP_MEMBER_OF; 8207 // From now on, all subsequent PTR_AND_OBJ entries should not be 8208 // marked as MEMBER_OF. 8209 ShouldBeMemberOf = false; 8210 } 8211 } 8212 8213 CombinedInfo.Types.push_back(Flags); 8214 } 8215 8216 // If we have encountered a member expression so far, keep track of the 8217 // mapped member. If the parent is "*this", then the value declaration 8218 // is nullptr. 8219 if (EncounteredME) { 8220 const auto *FD = cast<FieldDecl>(EncounteredME->getMemberDecl()); 8221 unsigned FieldIndex = FD->getFieldIndex(); 8222 8223 // Update info about the lowest and highest elements for this struct 8224 if (!PartialStruct.Base.isValid()) { 8225 PartialStruct.LowestElem = {FieldIndex, LowestElem}; 8226 if (IsFinalArraySection) { 8227 Address HB = 8228 CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false) 8229 .getAddress(CGF); 8230 PartialStruct.HighestElem = {FieldIndex, HB}; 8231 } else { 8232 PartialStruct.HighestElem = {FieldIndex, LowestElem}; 8233 } 8234 PartialStruct.Base = BP; 8235 PartialStruct.LB = BP; 8236 } else if (FieldIndex < PartialStruct.LowestElem.first) { 8237 PartialStruct.LowestElem = {FieldIndex, LowestElem}; 8238 } else if (FieldIndex > PartialStruct.HighestElem.first) { 8239 PartialStruct.HighestElem = {FieldIndex, LowestElem}; 8240 } 8241 } 8242 8243 // Need to emit combined struct for array sections. 8244 if (IsFinalArraySection || IsNonContiguous) 8245 PartialStruct.IsArraySection = true; 8246 8247 // If we have a final array section, we are done with this expression. 8248 if (IsFinalArraySection) 8249 break; 8250 8251 // The pointer becomes the base for the next element. 8252 if (Next != CE) 8253 BP = IsMemberReference ? LowestElem : LB; 8254 8255 IsExpressionFirstInfo = false; 8256 IsCaptureFirstInfo = false; 8257 FirstPointerInComplexData = false; 8258 IsPrevMemberReference = IsMemberReference; 8259 } else if (FirstPointerInComplexData) { 8260 QualType Ty = Components.rbegin() 8261 ->getAssociatedDeclaration() 8262 ->getType() 8263 .getNonReferenceType(); 8264 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>()); 8265 FirstPointerInComplexData = false; 8266 } 8267 } 8268 // If ran into the whole component - allocate the space for the whole 8269 // record. 8270 if (!EncounteredME) 8271 PartialStruct.HasCompleteRecord = true; 8272 8273 if (!IsNonContiguous) 8274 return; 8275 8276 const ASTContext &Context = CGF.getContext(); 8277 8278 // For supporting stride in array section, we need to initialize the first 8279 // dimension size as 1, first offset as 0, and first count as 1 8280 MapValuesArrayTy CurOffsets = {llvm::ConstantInt::get(CGF.CGM.Int64Ty, 0)}; 8281 MapValuesArrayTy CurCounts = {llvm::ConstantInt::get(CGF.CGM.Int64Ty, 1)}; 8282 MapValuesArrayTy CurStrides; 8283 MapValuesArrayTy DimSizes{llvm::ConstantInt::get(CGF.CGM.Int64Ty, 1)}; 8284 uint64_t ElementTypeSize; 8285 8286 // Collect Size information for each dimension and get the element size as 8287 // the first Stride. For example, for `int arr[10][10]`, the DimSizes 8288 // should be [10, 10] and the first stride is 4 btyes. 8289 for (const OMPClauseMappableExprCommon::MappableComponent &Component : 8290 Components) { 8291 const Expr *AssocExpr = Component.getAssociatedExpression(); 8292 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 8293 8294 if (!OASE) 8295 continue; 8296 8297 QualType Ty = OMPArraySectionExpr::getBaseOriginalType(OASE->getBase()); 8298 auto *CAT = Context.getAsConstantArrayType(Ty); 8299 auto *VAT = Context.getAsVariableArrayType(Ty); 8300 8301 // We need all the dimension size except for the last dimension. 8302 assert((VAT || CAT || &Component == &*Components.begin()) && 8303 "Should be either ConstantArray or VariableArray if not the " 8304 "first Component"); 8305 8306 // Get element size if CurStrides is empty. 8307 if (CurStrides.empty()) { 8308 const Type *ElementType = nullptr; 8309 if (CAT) 8310 ElementType = CAT->getElementType().getTypePtr(); 8311 else if (VAT) 8312 ElementType = VAT->getElementType().getTypePtr(); 8313 else 8314 assert(&Component == &*Components.begin() && 8315 "Only expect pointer (non CAT or VAT) when this is the " 8316 "first Component"); 8317 // If ElementType is null, then it means the base is a pointer 8318 // (neither CAT nor VAT) and we'll attempt to get ElementType again 8319 // for next iteration. 8320 if (ElementType) { 8321 // For the case that having pointer as base, we need to remove one 8322 // level of indirection. 8323 if (&Component != &*Components.begin()) 8324 ElementType = ElementType->getPointeeOrArrayElementType(); 8325 ElementTypeSize = 8326 Context.getTypeSizeInChars(ElementType).getQuantity(); 8327 CurStrides.push_back( 8328 llvm::ConstantInt::get(CGF.Int64Ty, ElementTypeSize)); 8329 } 8330 } 8331 // Get dimension value except for the last dimension since we don't need 8332 // it. 8333 if (DimSizes.size() < Components.size() - 1) { 8334 if (CAT) 8335 DimSizes.push_back(llvm::ConstantInt::get( 8336 CGF.Int64Ty, CAT->getSize().getZExtValue())); 8337 else if (VAT) 8338 DimSizes.push_back(CGF.Builder.CreateIntCast( 8339 CGF.EmitScalarExpr(VAT->getSizeExpr()), CGF.Int64Ty, 8340 /*IsSigned=*/false)); 8341 } 8342 } 8343 8344 // Skip the dummy dimension since we have already have its information. 8345 auto *DI = DimSizes.begin() + 1; 8346 // Product of dimension. 8347 llvm::Value *DimProd = 8348 llvm::ConstantInt::get(CGF.CGM.Int64Ty, ElementTypeSize); 8349 8350 // Collect info for non-contiguous. Notice that offset, count, and stride 8351 // are only meaningful for array-section, so we insert a null for anything 8352 // other than array-section. 8353 // Also, the size of offset, count, and stride are not the same as 8354 // pointers, base_pointers, sizes, or dims. Instead, the size of offset, 8355 // count, and stride are the same as the number of non-contiguous 8356 // declaration in target update to/from clause. 8357 for (const OMPClauseMappableExprCommon::MappableComponent &Component : 8358 Components) { 8359 const Expr *AssocExpr = Component.getAssociatedExpression(); 8360 8361 if (const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr)) { 8362 llvm::Value *Offset = CGF.Builder.CreateIntCast( 8363 CGF.EmitScalarExpr(AE->getIdx()), CGF.Int64Ty, 8364 /*isSigned=*/false); 8365 CurOffsets.push_back(Offset); 8366 CurCounts.push_back(llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/1)); 8367 CurStrides.push_back(CurStrides.back()); 8368 continue; 8369 } 8370 8371 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 8372 8373 if (!OASE) 8374 continue; 8375 8376 // Offset 8377 const Expr *OffsetExpr = OASE->getLowerBound(); 8378 llvm::Value *Offset = nullptr; 8379 if (!OffsetExpr) { 8380 // If offset is absent, then we just set it to zero. 8381 Offset = llvm::ConstantInt::get(CGF.Int64Ty, 0); 8382 } else { 8383 Offset = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(OffsetExpr), 8384 CGF.Int64Ty, 8385 /*isSigned=*/false); 8386 } 8387 CurOffsets.push_back(Offset); 8388 8389 // Count 8390 const Expr *CountExpr = OASE->getLength(); 8391 llvm::Value *Count = nullptr; 8392 if (!CountExpr) { 8393 // In Clang, once a high dimension is an array section, we construct all 8394 // the lower dimension as array section, however, for case like 8395 // arr[0:2][2], Clang construct the inner dimension as an array section 8396 // but it actually is not in an array section form according to spec. 8397 if (!OASE->getColonLocFirst().isValid() && 8398 !OASE->getColonLocSecond().isValid()) { 8399 Count = llvm::ConstantInt::get(CGF.Int64Ty, 1); 8400 } else { 8401 // OpenMP 5.0, 2.1.5 Array Sections, Description. 8402 // When the length is absent it defaults to ⌈(size − 8403 // lower-bound)/stride⌉, where size is the size of the array 8404 // dimension. 8405 const Expr *StrideExpr = OASE->getStride(); 8406 llvm::Value *Stride = 8407 StrideExpr 8408 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(StrideExpr), 8409 CGF.Int64Ty, /*isSigned=*/false) 8410 : nullptr; 8411 if (Stride) 8412 Count = CGF.Builder.CreateUDiv( 8413 CGF.Builder.CreateNUWSub(*DI, Offset), Stride); 8414 else 8415 Count = CGF.Builder.CreateNUWSub(*DI, Offset); 8416 } 8417 } else { 8418 Count = CGF.EmitScalarExpr(CountExpr); 8419 } 8420 Count = CGF.Builder.CreateIntCast(Count, CGF.Int64Ty, /*isSigned=*/false); 8421 CurCounts.push_back(Count); 8422 8423 // Stride_n' = Stride_n * (D_0 * D_1 ... * D_n-1) * Unit size 8424 // Take `int arr[5][5][5]` and `arr[0:2:2][1:2:1][0:2:2]` as an example: 8425 // Offset Count Stride 8426 // D0 0 1 4 (int) <- dummy dimension 8427 // D1 0 2 8 (2 * (1) * 4) 8428 // D2 1 2 20 (1 * (1 * 5) * 4) 8429 // D3 0 2 200 (2 * (1 * 5 * 4) * 4) 8430 const Expr *StrideExpr = OASE->getStride(); 8431 llvm::Value *Stride = 8432 StrideExpr 8433 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(StrideExpr), 8434 CGF.Int64Ty, /*isSigned=*/false) 8435 : nullptr; 8436 DimProd = CGF.Builder.CreateNUWMul(DimProd, *(DI - 1)); 8437 if (Stride) 8438 CurStrides.push_back(CGF.Builder.CreateNUWMul(DimProd, Stride)); 8439 else 8440 CurStrides.push_back(DimProd); 8441 if (DI != DimSizes.end()) 8442 ++DI; 8443 } 8444 8445 CombinedInfo.NonContigInfo.Offsets.push_back(CurOffsets); 8446 CombinedInfo.NonContigInfo.Counts.push_back(CurCounts); 8447 CombinedInfo.NonContigInfo.Strides.push_back(CurStrides); 8448 } 8449 8450 /// Return the adjusted map modifiers if the declaration a capture refers to 8451 /// appears in a first-private clause. This is expected to be used only with 8452 /// directives that start with 'target'. 8453 MappableExprsHandler::OpenMPOffloadMappingFlags 8454 getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const { 8455 assert(Cap.capturesVariable() && "Expected capture by reference only!"); 8456 8457 // A first private variable captured by reference will use only the 8458 // 'private ptr' and 'map to' flag. Return the right flags if the captured 8459 // declaration is known as first-private in this handler. 8460 if (FirstPrivateDecls.count(Cap.getCapturedVar())) { 8461 if (Cap.getCapturedVar()->getType()->isAnyPointerType()) 8462 return MappableExprsHandler::OMP_MAP_TO | 8463 MappableExprsHandler::OMP_MAP_PTR_AND_OBJ; 8464 return MappableExprsHandler::OMP_MAP_PRIVATE | 8465 MappableExprsHandler::OMP_MAP_TO; 8466 } 8467 auto I = LambdasMap.find(Cap.getCapturedVar()->getCanonicalDecl()); 8468 if (I != LambdasMap.end()) 8469 // for map(to: lambda): using user specified map type. 8470 return getMapTypeBits( 8471 I->getSecond()->getMapType(), I->getSecond()->getMapTypeModifiers(), 8472 /*MotionModifiers=*/llvm::None, I->getSecond()->isImplicit(), 8473 /*AddPtrFlag=*/false, 8474 /*AddIsTargetParamFlag=*/false, 8475 /*isNonContiguous=*/false); 8476 return MappableExprsHandler::OMP_MAP_TO | 8477 MappableExprsHandler::OMP_MAP_FROM; 8478 } 8479 8480 static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) { 8481 // Rotate by getFlagMemberOffset() bits. 8482 return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1) 8483 << getFlagMemberOffset()); 8484 } 8485 8486 static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags, 8487 OpenMPOffloadMappingFlags MemberOfFlag) { 8488 // If the entry is PTR_AND_OBJ but has not been marked with the special 8489 // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be 8490 // marked as MEMBER_OF. 8491 if ((Flags & OMP_MAP_PTR_AND_OBJ) && 8492 ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF)) 8493 return; 8494 8495 // Reset the placeholder value to prepare the flag for the assignment of the 8496 // proper MEMBER_OF value. 8497 Flags &= ~OMP_MAP_MEMBER_OF; 8498 Flags |= MemberOfFlag; 8499 } 8500 8501 void getPlainLayout(const CXXRecordDecl *RD, 8502 llvm::SmallVectorImpl<const FieldDecl *> &Layout, 8503 bool AsBase) const { 8504 const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD); 8505 8506 llvm::StructType *St = 8507 AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType(); 8508 8509 unsigned NumElements = St->getNumElements(); 8510 llvm::SmallVector< 8511 llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4> 8512 RecordLayout(NumElements); 8513 8514 // Fill bases. 8515 for (const auto &I : RD->bases()) { 8516 if (I.isVirtual()) 8517 continue; 8518 const auto *Base = I.getType()->getAsCXXRecordDecl(); 8519 // Ignore empty bases. 8520 if (Base->isEmpty() || CGF.getContext() 8521 .getASTRecordLayout(Base) 8522 .getNonVirtualSize() 8523 .isZero()) 8524 continue; 8525 8526 unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base); 8527 RecordLayout[FieldIndex] = Base; 8528 } 8529 // Fill in virtual bases. 8530 for (const auto &I : RD->vbases()) { 8531 const auto *Base = I.getType()->getAsCXXRecordDecl(); 8532 // Ignore empty bases. 8533 if (Base->isEmpty()) 8534 continue; 8535 unsigned FieldIndex = RL.getVirtualBaseIndex(Base); 8536 if (RecordLayout[FieldIndex]) 8537 continue; 8538 RecordLayout[FieldIndex] = Base; 8539 } 8540 // Fill in all the fields. 8541 assert(!RD->isUnion() && "Unexpected union."); 8542 for (const auto *Field : RD->fields()) { 8543 // Fill in non-bitfields. (Bitfields always use a zero pattern, which we 8544 // will fill in later.) 8545 if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) { 8546 unsigned FieldIndex = RL.getLLVMFieldNo(Field); 8547 RecordLayout[FieldIndex] = Field; 8548 } 8549 } 8550 for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *> 8551 &Data : RecordLayout) { 8552 if (Data.isNull()) 8553 continue; 8554 if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>()) 8555 getPlainLayout(Base, Layout, /*AsBase=*/true); 8556 else 8557 Layout.push_back(Data.get<const FieldDecl *>()); 8558 } 8559 } 8560 8561 /// Generate all the base pointers, section pointers, sizes, map types, and 8562 /// mappers for the extracted mappable expressions (all included in \a 8563 /// CombinedInfo). Also, for each item that relates with a device pointer, a 8564 /// pair of the relevant declaration and index where it occurs is appended to 8565 /// the device pointers info array. 8566 void generateAllInfoForClauses( 8567 ArrayRef<const OMPClause *> Clauses, MapCombinedInfoTy &CombinedInfo, 8568 const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkipVarSet = 8569 llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) const { 8570 // We have to process the component lists that relate with the same 8571 // declaration in a single chunk so that we can generate the map flags 8572 // correctly. Therefore, we organize all lists in a map. 8573 enum MapKind { Present, Allocs, Other, Total }; 8574 llvm::MapVector<CanonicalDeclPtr<const Decl>, 8575 SmallVector<SmallVector<MapInfo, 8>, 4>> 8576 Info; 8577 8578 // Helper function to fill the information map for the different supported 8579 // clauses. 8580 auto &&InfoGen = 8581 [&Info, &SkipVarSet]( 8582 const ValueDecl *D, MapKind Kind, 8583 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 8584 OpenMPMapClauseKind MapType, 8585 ArrayRef<OpenMPMapModifierKind> MapModifiers, 8586 ArrayRef<OpenMPMotionModifierKind> MotionModifiers, 8587 bool ReturnDevicePointer, bool IsImplicit, const ValueDecl *Mapper, 8588 const Expr *VarRef = nullptr, bool ForDeviceAddr = false) { 8589 if (SkipVarSet.contains(D)) 8590 return; 8591 auto It = Info.find(D); 8592 if (It == Info.end()) 8593 It = Info 8594 .insert(std::make_pair( 8595 D, SmallVector<SmallVector<MapInfo, 8>, 4>(Total))) 8596 .first; 8597 It->second[Kind].emplace_back( 8598 L, MapType, MapModifiers, MotionModifiers, ReturnDevicePointer, 8599 IsImplicit, Mapper, VarRef, ForDeviceAddr); 8600 }; 8601 8602 for (const auto *Cl : Clauses) { 8603 const auto *C = dyn_cast<OMPMapClause>(Cl); 8604 if (!C) 8605 continue; 8606 MapKind Kind = Other; 8607 if (llvm::is_contained(C->getMapTypeModifiers(), 8608 OMPC_MAP_MODIFIER_present)) 8609 Kind = Present; 8610 else if (C->getMapType() == OMPC_MAP_alloc) 8611 Kind = Allocs; 8612 const auto *EI = C->getVarRefs().begin(); 8613 for (const auto L : C->component_lists()) { 8614 const Expr *E = (C->getMapLoc().isValid()) ? *EI : nullptr; 8615 InfoGen(std::get<0>(L), Kind, std::get<1>(L), C->getMapType(), 8616 C->getMapTypeModifiers(), llvm::None, 8617 /*ReturnDevicePointer=*/false, C->isImplicit(), std::get<2>(L), 8618 E); 8619 ++EI; 8620 } 8621 } 8622 for (const auto *Cl : Clauses) { 8623 const auto *C = dyn_cast<OMPToClause>(Cl); 8624 if (!C) 8625 continue; 8626 MapKind Kind = Other; 8627 if (llvm::is_contained(C->getMotionModifiers(), 8628 OMPC_MOTION_MODIFIER_present)) 8629 Kind = Present; 8630 const auto *EI = C->getVarRefs().begin(); 8631 for (const auto L : C->component_lists()) { 8632 InfoGen(std::get<0>(L), Kind, std::get<1>(L), OMPC_MAP_to, llvm::None, 8633 C->getMotionModifiers(), /*ReturnDevicePointer=*/false, 8634 C->isImplicit(), std::get<2>(L), *EI); 8635 ++EI; 8636 } 8637 } 8638 for (const auto *Cl : Clauses) { 8639 const auto *C = dyn_cast<OMPFromClause>(Cl); 8640 if (!C) 8641 continue; 8642 MapKind Kind = Other; 8643 if (llvm::is_contained(C->getMotionModifiers(), 8644 OMPC_MOTION_MODIFIER_present)) 8645 Kind = Present; 8646 const auto *EI = C->getVarRefs().begin(); 8647 for (const auto L : C->component_lists()) { 8648 InfoGen(std::get<0>(L), Kind, std::get<1>(L), OMPC_MAP_from, llvm::None, 8649 C->getMotionModifiers(), /*ReturnDevicePointer=*/false, 8650 C->isImplicit(), std::get<2>(L), *EI); 8651 ++EI; 8652 } 8653 } 8654 8655 // Look at the use_device_ptr clause information and mark the existing map 8656 // entries as such. If there is no map information for an entry in the 8657 // use_device_ptr list, we create one with map type 'alloc' and zero size 8658 // section. It is the user fault if that was not mapped before. If there is 8659 // no map information and the pointer is a struct member, then we defer the 8660 // emission of that entry until the whole struct has been processed. 8661 llvm::MapVector<CanonicalDeclPtr<const Decl>, 8662 SmallVector<DeferredDevicePtrEntryTy, 4>> 8663 DeferredInfo; 8664 MapCombinedInfoTy UseDevicePtrCombinedInfo; 8665 8666 for (const auto *Cl : Clauses) { 8667 const auto *C = dyn_cast<OMPUseDevicePtrClause>(Cl); 8668 if (!C) 8669 continue; 8670 for (const auto L : C->component_lists()) { 8671 OMPClauseMappableExprCommon::MappableExprComponentListRef Components = 8672 std::get<1>(L); 8673 assert(!Components.empty() && 8674 "Not expecting empty list of components!"); 8675 const ValueDecl *VD = Components.back().getAssociatedDeclaration(); 8676 VD = cast<ValueDecl>(VD->getCanonicalDecl()); 8677 const Expr *IE = Components.back().getAssociatedExpression(); 8678 // If the first component is a member expression, we have to look into 8679 // 'this', which maps to null in the map of map information. Otherwise 8680 // look directly for the information. 8681 auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD); 8682 8683 // We potentially have map information for this declaration already. 8684 // Look for the first set of components that refer to it. 8685 if (It != Info.end()) { 8686 bool Found = false; 8687 for (auto &Data : It->second) { 8688 auto *CI = llvm::find_if(Data, [VD](const MapInfo &MI) { 8689 return MI.Components.back().getAssociatedDeclaration() == VD; 8690 }); 8691 // If we found a map entry, signal that the pointer has to be 8692 // returned and move on to the next declaration. Exclude cases where 8693 // the base pointer is mapped as array subscript, array section or 8694 // array shaping. The base address is passed as a pointer to base in 8695 // this case and cannot be used as a base for use_device_ptr list 8696 // item. 8697 if (CI != Data.end()) { 8698 auto PrevCI = std::next(CI->Components.rbegin()); 8699 const auto *VarD = dyn_cast<VarDecl>(VD); 8700 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() || 8701 isa<MemberExpr>(IE) || 8702 !VD->getType().getNonReferenceType()->isPointerType() || 8703 PrevCI == CI->Components.rend() || 8704 isa<MemberExpr>(PrevCI->getAssociatedExpression()) || !VarD || 8705 VarD->hasLocalStorage()) { 8706 CI->ReturnDevicePointer = true; 8707 Found = true; 8708 break; 8709 } 8710 } 8711 } 8712 if (Found) 8713 continue; 8714 } 8715 8716 // We didn't find any match in our map information - generate a zero 8717 // size array section - if the pointer is a struct member we defer this 8718 // action until the whole struct has been processed. 8719 if (isa<MemberExpr>(IE)) { 8720 // Insert the pointer into Info to be processed by 8721 // generateInfoForComponentList. Because it is a member pointer 8722 // without a pointee, no entry will be generated for it, therefore 8723 // we need to generate one after the whole struct has been processed. 8724 // Nonetheless, generateInfoForComponentList must be called to take 8725 // the pointer into account for the calculation of the range of the 8726 // partial struct. 8727 InfoGen(nullptr, Other, Components, OMPC_MAP_unknown, llvm::None, 8728 llvm::None, /*ReturnDevicePointer=*/false, C->isImplicit(), 8729 nullptr); 8730 DeferredInfo[nullptr].emplace_back(IE, VD, /*ForDeviceAddr=*/false); 8731 } else { 8732 llvm::Value *Ptr = 8733 CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc()); 8734 UseDevicePtrCombinedInfo.Exprs.push_back(VD); 8735 UseDevicePtrCombinedInfo.BasePointers.emplace_back(Ptr, VD); 8736 UseDevicePtrCombinedInfo.Pointers.push_back(Ptr); 8737 UseDevicePtrCombinedInfo.Sizes.push_back( 8738 llvm::Constant::getNullValue(CGF.Int64Ty)); 8739 UseDevicePtrCombinedInfo.Types.push_back(OMP_MAP_RETURN_PARAM); 8740 UseDevicePtrCombinedInfo.Mappers.push_back(nullptr); 8741 } 8742 } 8743 } 8744 8745 // Look at the use_device_addr clause information and mark the existing map 8746 // entries as such. If there is no map information for an entry in the 8747 // use_device_addr list, we create one with map type 'alloc' and zero size 8748 // section. It is the user fault if that was not mapped before. If there is 8749 // no map information and the pointer is a struct member, then we defer the 8750 // emission of that entry until the whole struct has been processed. 8751 llvm::SmallDenseSet<CanonicalDeclPtr<const Decl>, 4> Processed; 8752 for (const auto *Cl : Clauses) { 8753 const auto *C = dyn_cast<OMPUseDeviceAddrClause>(Cl); 8754 if (!C) 8755 continue; 8756 for (const auto L : C->component_lists()) { 8757 assert(!std::get<1>(L).empty() && 8758 "Not expecting empty list of components!"); 8759 const ValueDecl *VD = std::get<1>(L).back().getAssociatedDeclaration(); 8760 if (!Processed.insert(VD).second) 8761 continue; 8762 VD = cast<ValueDecl>(VD->getCanonicalDecl()); 8763 const Expr *IE = std::get<1>(L).back().getAssociatedExpression(); 8764 // If the first component is a member expression, we have to look into 8765 // 'this', which maps to null in the map of map information. Otherwise 8766 // look directly for the information. 8767 auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD); 8768 8769 // We potentially have map information for this declaration already. 8770 // Look for the first set of components that refer to it. 8771 if (It != Info.end()) { 8772 bool Found = false; 8773 for (auto &Data : It->second) { 8774 auto *CI = llvm::find_if(Data, [VD](const MapInfo &MI) { 8775 return MI.Components.back().getAssociatedDeclaration() == VD; 8776 }); 8777 // If we found a map entry, signal that the pointer has to be 8778 // returned and move on to the next declaration. 8779 if (CI != Data.end()) { 8780 CI->ReturnDevicePointer = true; 8781 Found = true; 8782 break; 8783 } 8784 } 8785 if (Found) 8786 continue; 8787 } 8788 8789 // We didn't find any match in our map information - generate a zero 8790 // size array section - if the pointer is a struct member we defer this 8791 // action until the whole struct has been processed. 8792 if (isa<MemberExpr>(IE)) { 8793 // Insert the pointer into Info to be processed by 8794 // generateInfoForComponentList. Because it is a member pointer 8795 // without a pointee, no entry will be generated for it, therefore 8796 // we need to generate one after the whole struct has been processed. 8797 // Nonetheless, generateInfoForComponentList must be called to take 8798 // the pointer into account for the calculation of the range of the 8799 // partial struct. 8800 InfoGen(nullptr, Other, std::get<1>(L), OMPC_MAP_unknown, llvm::None, 8801 llvm::None, /*ReturnDevicePointer=*/false, C->isImplicit(), 8802 nullptr, nullptr, /*ForDeviceAddr=*/true); 8803 DeferredInfo[nullptr].emplace_back(IE, VD, /*ForDeviceAddr=*/true); 8804 } else { 8805 llvm::Value *Ptr; 8806 if (IE->isGLValue()) 8807 Ptr = CGF.EmitLValue(IE).getPointer(CGF); 8808 else 8809 Ptr = CGF.EmitScalarExpr(IE); 8810 CombinedInfo.Exprs.push_back(VD); 8811 CombinedInfo.BasePointers.emplace_back(Ptr, VD); 8812 CombinedInfo.Pointers.push_back(Ptr); 8813 CombinedInfo.Sizes.push_back( 8814 llvm::Constant::getNullValue(CGF.Int64Ty)); 8815 CombinedInfo.Types.push_back(OMP_MAP_RETURN_PARAM); 8816 CombinedInfo.Mappers.push_back(nullptr); 8817 } 8818 } 8819 } 8820 8821 for (const auto &Data : Info) { 8822 StructRangeInfoTy PartialStruct; 8823 // Temporary generated information. 8824 MapCombinedInfoTy CurInfo; 8825 const Decl *D = Data.first; 8826 const ValueDecl *VD = cast_or_null<ValueDecl>(D); 8827 for (const auto &M : Data.second) { 8828 for (const MapInfo &L : M) { 8829 assert(!L.Components.empty() && 8830 "Not expecting declaration with no component lists."); 8831 8832 // Remember the current base pointer index. 8833 unsigned CurrentBasePointersIdx = CurInfo.BasePointers.size(); 8834 CurInfo.NonContigInfo.IsNonContiguous = 8835 L.Components.back().isNonContiguous(); 8836 generateInfoForComponentList( 8837 L.MapType, L.MapModifiers, L.MotionModifiers, L.Components, 8838 CurInfo, PartialStruct, /*IsFirstComponentList=*/false, 8839 L.IsImplicit, L.Mapper, L.ForDeviceAddr, VD, L.VarRef); 8840 8841 // If this entry relates with a device pointer, set the relevant 8842 // declaration and add the 'return pointer' flag. 8843 if (L.ReturnDevicePointer) { 8844 assert(CurInfo.BasePointers.size() > CurrentBasePointersIdx && 8845 "Unexpected number of mapped base pointers."); 8846 8847 const ValueDecl *RelevantVD = 8848 L.Components.back().getAssociatedDeclaration(); 8849 assert(RelevantVD && 8850 "No relevant declaration related with device pointer??"); 8851 8852 CurInfo.BasePointers[CurrentBasePointersIdx].setDevicePtrDecl( 8853 RelevantVD); 8854 CurInfo.Types[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM; 8855 } 8856 } 8857 } 8858 8859 // Append any pending zero-length pointers which are struct members and 8860 // used with use_device_ptr or use_device_addr. 8861 auto CI = DeferredInfo.find(Data.first); 8862 if (CI != DeferredInfo.end()) { 8863 for (const DeferredDevicePtrEntryTy &L : CI->second) { 8864 llvm::Value *BasePtr; 8865 llvm::Value *Ptr; 8866 if (L.ForDeviceAddr) { 8867 if (L.IE->isGLValue()) 8868 Ptr = this->CGF.EmitLValue(L.IE).getPointer(CGF); 8869 else 8870 Ptr = this->CGF.EmitScalarExpr(L.IE); 8871 BasePtr = Ptr; 8872 // Entry is RETURN_PARAM. Also, set the placeholder value 8873 // MEMBER_OF=FFFF so that the entry is later updated with the 8874 // correct value of MEMBER_OF. 8875 CurInfo.Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_MEMBER_OF); 8876 } else { 8877 BasePtr = this->CGF.EmitLValue(L.IE).getPointer(CGF); 8878 Ptr = this->CGF.EmitLoadOfScalar(this->CGF.EmitLValue(L.IE), 8879 L.IE->getExprLoc()); 8880 // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the 8881 // placeholder value MEMBER_OF=FFFF so that the entry is later 8882 // updated with the correct value of MEMBER_OF. 8883 CurInfo.Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM | 8884 OMP_MAP_MEMBER_OF); 8885 } 8886 CurInfo.Exprs.push_back(L.VD); 8887 CurInfo.BasePointers.emplace_back(BasePtr, L.VD); 8888 CurInfo.Pointers.push_back(Ptr); 8889 CurInfo.Sizes.push_back( 8890 llvm::Constant::getNullValue(this->CGF.Int64Ty)); 8891 CurInfo.Mappers.push_back(nullptr); 8892 } 8893 } 8894 // If there is an entry in PartialStruct it means we have a struct with 8895 // individual members mapped. Emit an extra combined entry. 8896 if (PartialStruct.Base.isValid()) { 8897 CurInfo.NonContigInfo.Dims.push_back(0); 8898 emitCombinedEntry(CombinedInfo, CurInfo.Types, PartialStruct, VD); 8899 } 8900 8901 // We need to append the results of this capture to what we already 8902 // have. 8903 CombinedInfo.append(CurInfo); 8904 } 8905 // Append data for use_device_ptr clauses. 8906 CombinedInfo.append(UseDevicePtrCombinedInfo); 8907 } 8908 8909 public: 8910 MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF) 8911 : CurDir(&Dir), CGF(CGF) { 8912 // Extract firstprivate clause information. 8913 for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>()) 8914 for (const auto *D : C->varlists()) 8915 FirstPrivateDecls.try_emplace( 8916 cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit()); 8917 // Extract implicit firstprivates from uses_allocators clauses. 8918 for (const auto *C : Dir.getClausesOfKind<OMPUsesAllocatorsClause>()) { 8919 for (unsigned I = 0, E = C->getNumberOfAllocators(); I < E; ++I) { 8920 OMPUsesAllocatorsClause::Data D = C->getAllocatorData(I); 8921 if (const auto *DRE = dyn_cast_or_null<DeclRefExpr>(D.AllocatorTraits)) 8922 FirstPrivateDecls.try_emplace(cast<VarDecl>(DRE->getDecl()), 8923 /*Implicit=*/true); 8924 else if (const auto *VD = dyn_cast<VarDecl>( 8925 cast<DeclRefExpr>(D.Allocator->IgnoreParenImpCasts()) 8926 ->getDecl())) 8927 FirstPrivateDecls.try_emplace(VD, /*Implicit=*/true); 8928 } 8929 } 8930 // Extract device pointer clause information. 8931 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>()) 8932 for (auto L : C->component_lists()) 8933 DevPointersMap[std::get<0>(L)].push_back(std::get<1>(L)); 8934 // Extract map information. 8935 for (const auto *C : Dir.getClausesOfKind<OMPMapClause>()) { 8936 if (C->getMapType() != OMPC_MAP_to) 8937 continue; 8938 for (auto L : C->component_lists()) { 8939 const ValueDecl *VD = std::get<0>(L); 8940 const auto *RD = VD ? VD->getType() 8941 .getCanonicalType() 8942 .getNonReferenceType() 8943 ->getAsCXXRecordDecl() 8944 : nullptr; 8945 if (RD && RD->isLambda()) 8946 LambdasMap.try_emplace(std::get<0>(L), C); 8947 } 8948 } 8949 } 8950 8951 /// Constructor for the declare mapper directive. 8952 MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF) 8953 : CurDir(&Dir), CGF(CGF) {} 8954 8955 /// Generate code for the combined entry if we have a partially mapped struct 8956 /// and take care of the mapping flags of the arguments corresponding to 8957 /// individual struct members. 8958 void emitCombinedEntry(MapCombinedInfoTy &CombinedInfo, 8959 MapFlagsArrayTy &CurTypes, 8960 const StructRangeInfoTy &PartialStruct, 8961 const ValueDecl *VD = nullptr, 8962 bool NotTargetParams = true) const { 8963 if (CurTypes.size() == 1 && 8964 ((CurTypes.back() & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF) && 8965 !PartialStruct.IsArraySection) 8966 return; 8967 Address LBAddr = PartialStruct.LowestElem.second; 8968 Address HBAddr = PartialStruct.HighestElem.second; 8969 if (PartialStruct.HasCompleteRecord) { 8970 LBAddr = PartialStruct.LB; 8971 HBAddr = PartialStruct.LB; 8972 } 8973 CombinedInfo.Exprs.push_back(VD); 8974 // Base is the base of the struct 8975 CombinedInfo.BasePointers.push_back(PartialStruct.Base.getPointer()); 8976 // Pointer is the address of the lowest element 8977 llvm::Value *LB = LBAddr.getPointer(); 8978 CombinedInfo.Pointers.push_back(LB); 8979 // There should not be a mapper for a combined entry. 8980 CombinedInfo.Mappers.push_back(nullptr); 8981 // Size is (addr of {highest+1} element) - (addr of lowest element) 8982 llvm::Value *HB = HBAddr.getPointer(); 8983 llvm::Value *HAddr = 8984 CGF.Builder.CreateConstGEP1_32(HBAddr.getElementType(), HB, /*Idx0=*/1); 8985 llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy); 8986 llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy); 8987 llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CGF.Int8Ty, CHAddr, CLAddr); 8988 llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty, 8989 /*isSigned=*/false); 8990 CombinedInfo.Sizes.push_back(Size); 8991 // Map type is always TARGET_PARAM, if generate info for captures. 8992 CombinedInfo.Types.push_back(NotTargetParams ? OMP_MAP_NONE 8993 : OMP_MAP_TARGET_PARAM); 8994 // If any element has the present modifier, then make sure the runtime 8995 // doesn't attempt to allocate the struct. 8996 if (CurTypes.end() != 8997 llvm::find_if(CurTypes, [](OpenMPOffloadMappingFlags Type) { 8998 return Type & OMP_MAP_PRESENT; 8999 })) 9000 CombinedInfo.Types.back() |= OMP_MAP_PRESENT; 9001 // Remove TARGET_PARAM flag from the first element 9002 (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM; 9003 // If any element has the ompx_hold modifier, then make sure the runtime 9004 // uses the hold reference count for the struct as a whole so that it won't 9005 // be unmapped by an extra dynamic reference count decrement. Add it to all 9006 // elements as well so the runtime knows which reference count to check 9007 // when determining whether it's time for device-to-host transfers of 9008 // individual elements. 9009 if (CurTypes.end() != 9010 llvm::find_if(CurTypes, [](OpenMPOffloadMappingFlags Type) { 9011 return Type & OMP_MAP_OMPX_HOLD; 9012 })) { 9013 CombinedInfo.Types.back() |= OMP_MAP_OMPX_HOLD; 9014 for (auto &M : CurTypes) 9015 M |= OMP_MAP_OMPX_HOLD; 9016 } 9017 9018 // All other current entries will be MEMBER_OF the combined entry 9019 // (except for PTR_AND_OBJ entries which do not have a placeholder value 9020 // 0xFFFF in the MEMBER_OF field). 9021 OpenMPOffloadMappingFlags MemberOfFlag = 9022 getMemberOfFlag(CombinedInfo.BasePointers.size() - 1); 9023 for (auto &M : CurTypes) 9024 setCorrectMemberOfFlag(M, MemberOfFlag); 9025 } 9026 9027 /// Generate all the base pointers, section pointers, sizes, map types, and 9028 /// mappers for the extracted mappable expressions (all included in \a 9029 /// CombinedInfo). Also, for each item that relates with a device pointer, a 9030 /// pair of the relevant declaration and index where it occurs is appended to 9031 /// the device pointers info array. 9032 void generateAllInfo( 9033 MapCombinedInfoTy &CombinedInfo, 9034 const llvm::DenseSet<CanonicalDeclPtr<const Decl>> &SkipVarSet = 9035 llvm::DenseSet<CanonicalDeclPtr<const Decl>>()) const { 9036 assert(CurDir.is<const OMPExecutableDirective *>() && 9037 "Expect a executable directive"); 9038 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 9039 generateAllInfoForClauses(CurExecDir->clauses(), CombinedInfo, SkipVarSet); 9040 } 9041 9042 /// Generate all the base pointers, section pointers, sizes, map types, and 9043 /// mappers for the extracted map clauses of user-defined mapper (all included 9044 /// in \a CombinedInfo). 9045 void generateAllInfoForMapper(MapCombinedInfoTy &CombinedInfo) const { 9046 assert(CurDir.is<const OMPDeclareMapperDecl *>() && 9047 "Expect a declare mapper directive"); 9048 const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>(); 9049 generateAllInfoForClauses(CurMapperDir->clauses(), CombinedInfo); 9050 } 9051 9052 /// Emit capture info for lambdas for variables captured by reference. 9053 void generateInfoForLambdaCaptures( 9054 const ValueDecl *VD, llvm::Value *Arg, MapCombinedInfoTy &CombinedInfo, 9055 llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const { 9056 QualType VDType = VD->getType().getCanonicalType().getNonReferenceType(); 9057 const auto *RD = VDType->getAsCXXRecordDecl(); 9058 if (!RD || !RD->isLambda()) 9059 return; 9060 Address VDAddr(Arg, CGF.ConvertTypeForMem(VDType), 9061 CGF.getContext().getDeclAlign(VD)); 9062 LValue VDLVal = CGF.MakeAddrLValue(VDAddr, VDType); 9063 llvm::DenseMap<const VarDecl *, FieldDecl *> Captures; 9064 FieldDecl *ThisCapture = nullptr; 9065 RD->getCaptureFields(Captures, ThisCapture); 9066 if (ThisCapture) { 9067 LValue ThisLVal = 9068 CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture); 9069 LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture); 9070 LambdaPointers.try_emplace(ThisLVal.getPointer(CGF), 9071 VDLVal.getPointer(CGF)); 9072 CombinedInfo.Exprs.push_back(VD); 9073 CombinedInfo.BasePointers.push_back(ThisLVal.getPointer(CGF)); 9074 CombinedInfo.Pointers.push_back(ThisLValVal.getPointer(CGF)); 9075 CombinedInfo.Sizes.push_back( 9076 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy), 9077 CGF.Int64Ty, /*isSigned=*/true)); 9078 CombinedInfo.Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 9079 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 9080 CombinedInfo.Mappers.push_back(nullptr); 9081 } 9082 for (const LambdaCapture &LC : RD->captures()) { 9083 if (!LC.capturesVariable()) 9084 continue; 9085 const VarDecl *VD = LC.getCapturedVar(); 9086 if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType()) 9087 continue; 9088 auto It = Captures.find(VD); 9089 assert(It != Captures.end() && "Found lambda capture without field."); 9090 LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second); 9091 if (LC.getCaptureKind() == LCK_ByRef) { 9092 LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second); 9093 LambdaPointers.try_emplace(VarLVal.getPointer(CGF), 9094 VDLVal.getPointer(CGF)); 9095 CombinedInfo.Exprs.push_back(VD); 9096 CombinedInfo.BasePointers.push_back(VarLVal.getPointer(CGF)); 9097 CombinedInfo.Pointers.push_back(VarLValVal.getPointer(CGF)); 9098 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 9099 CGF.getTypeSize( 9100 VD->getType().getCanonicalType().getNonReferenceType()), 9101 CGF.Int64Ty, /*isSigned=*/true)); 9102 } else { 9103 RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation()); 9104 LambdaPointers.try_emplace(VarLVal.getPointer(CGF), 9105 VDLVal.getPointer(CGF)); 9106 CombinedInfo.Exprs.push_back(VD); 9107 CombinedInfo.BasePointers.push_back(VarLVal.getPointer(CGF)); 9108 CombinedInfo.Pointers.push_back(VarRVal.getScalarVal()); 9109 CombinedInfo.Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0)); 9110 } 9111 CombinedInfo.Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 9112 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 9113 CombinedInfo.Mappers.push_back(nullptr); 9114 } 9115 } 9116 9117 /// Set correct indices for lambdas captures. 9118 void adjustMemberOfForLambdaCaptures( 9119 const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers, 9120 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 9121 MapFlagsArrayTy &Types) const { 9122 for (unsigned I = 0, E = Types.size(); I < E; ++I) { 9123 // Set correct member_of idx for all implicit lambda captures. 9124 if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 9125 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT)) 9126 continue; 9127 llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]); 9128 assert(BasePtr && "Unable to find base lambda address."); 9129 int TgtIdx = -1; 9130 for (unsigned J = I; J > 0; --J) { 9131 unsigned Idx = J - 1; 9132 if (Pointers[Idx] != BasePtr) 9133 continue; 9134 TgtIdx = Idx; 9135 break; 9136 } 9137 assert(TgtIdx != -1 && "Unable to find parent lambda."); 9138 // All other current entries will be MEMBER_OF the combined entry 9139 // (except for PTR_AND_OBJ entries which do not have a placeholder value 9140 // 0xFFFF in the MEMBER_OF field). 9141 OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx); 9142 setCorrectMemberOfFlag(Types[I], MemberOfFlag); 9143 } 9144 } 9145 9146 /// Generate the base pointers, section pointers, sizes, map types, and 9147 /// mappers associated to a given capture (all included in \a CombinedInfo). 9148 void generateInfoForCapture(const CapturedStmt::Capture *Cap, 9149 llvm::Value *Arg, MapCombinedInfoTy &CombinedInfo, 9150 StructRangeInfoTy &PartialStruct) const { 9151 assert(!Cap->capturesVariableArrayType() && 9152 "Not expecting to generate map info for a variable array type!"); 9153 9154 // We need to know when we generating information for the first component 9155 const ValueDecl *VD = Cap->capturesThis() 9156 ? nullptr 9157 : Cap->getCapturedVar()->getCanonicalDecl(); 9158 9159 // for map(to: lambda): skip here, processing it in 9160 // generateDefaultMapInfo 9161 if (LambdasMap.count(VD)) 9162 return; 9163 9164 // If this declaration appears in a is_device_ptr clause we just have to 9165 // pass the pointer by value. If it is a reference to a declaration, we just 9166 // pass its value. 9167 if (DevPointersMap.count(VD)) { 9168 CombinedInfo.Exprs.push_back(VD); 9169 CombinedInfo.BasePointers.emplace_back(Arg, VD); 9170 CombinedInfo.Pointers.push_back(Arg); 9171 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 9172 CGF.getTypeSize(CGF.getContext().VoidPtrTy), CGF.Int64Ty, 9173 /*isSigned=*/true)); 9174 CombinedInfo.Types.push_back( 9175 (Cap->capturesVariable() ? OMP_MAP_TO : OMP_MAP_LITERAL) | 9176 OMP_MAP_TARGET_PARAM); 9177 CombinedInfo.Mappers.push_back(nullptr); 9178 return; 9179 } 9180 9181 using MapData = 9182 std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef, 9183 OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool, 9184 const ValueDecl *, const Expr *>; 9185 SmallVector<MapData, 4> DeclComponentLists; 9186 assert(CurDir.is<const OMPExecutableDirective *>() && 9187 "Expect a executable directive"); 9188 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 9189 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) { 9190 const auto *EI = C->getVarRefs().begin(); 9191 for (const auto L : C->decl_component_lists(VD)) { 9192 const ValueDecl *VDecl, *Mapper; 9193 // The Expression is not correct if the mapping is implicit 9194 const Expr *E = (C->getMapLoc().isValid()) ? *EI : nullptr; 9195 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 9196 std::tie(VDecl, Components, Mapper) = L; 9197 assert(VDecl == VD && "We got information for the wrong declaration??"); 9198 assert(!Components.empty() && 9199 "Not expecting declaration with no component lists."); 9200 DeclComponentLists.emplace_back(Components, C->getMapType(), 9201 C->getMapTypeModifiers(), 9202 C->isImplicit(), Mapper, E); 9203 ++EI; 9204 } 9205 } 9206 llvm::stable_sort(DeclComponentLists, [](const MapData &LHS, 9207 const MapData &RHS) { 9208 ArrayRef<OpenMPMapModifierKind> MapModifiers = std::get<2>(LHS); 9209 OpenMPMapClauseKind MapType = std::get<1>(RHS); 9210 bool HasPresent = 9211 llvm::is_contained(MapModifiers, clang::OMPC_MAP_MODIFIER_present); 9212 bool HasAllocs = MapType == OMPC_MAP_alloc; 9213 MapModifiers = std::get<2>(RHS); 9214 MapType = std::get<1>(LHS); 9215 bool HasPresentR = 9216 llvm::is_contained(MapModifiers, clang::OMPC_MAP_MODIFIER_present); 9217 bool HasAllocsR = MapType == OMPC_MAP_alloc; 9218 return (HasPresent && !HasPresentR) || (HasAllocs && !HasAllocsR); 9219 }); 9220 9221 // Find overlapping elements (including the offset from the base element). 9222 llvm::SmallDenseMap< 9223 const MapData *, 9224 llvm::SmallVector< 9225 OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>, 9226 4> 9227 OverlappedData; 9228 size_t Count = 0; 9229 for (const MapData &L : DeclComponentLists) { 9230 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 9231 OpenMPMapClauseKind MapType; 9232 ArrayRef<OpenMPMapModifierKind> MapModifiers; 9233 bool IsImplicit; 9234 const ValueDecl *Mapper; 9235 const Expr *VarRef; 9236 std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper, VarRef) = 9237 L; 9238 ++Count; 9239 for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) { 9240 OMPClauseMappableExprCommon::MappableExprComponentListRef Components1; 9241 std::tie(Components1, MapType, MapModifiers, IsImplicit, Mapper, 9242 VarRef) = L1; 9243 auto CI = Components.rbegin(); 9244 auto CE = Components.rend(); 9245 auto SI = Components1.rbegin(); 9246 auto SE = Components1.rend(); 9247 for (; CI != CE && SI != SE; ++CI, ++SI) { 9248 if (CI->getAssociatedExpression()->getStmtClass() != 9249 SI->getAssociatedExpression()->getStmtClass()) 9250 break; 9251 // Are we dealing with different variables/fields? 9252 if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration()) 9253 break; 9254 } 9255 // Found overlapping if, at least for one component, reached the head 9256 // of the components list. 9257 if (CI == CE || SI == SE) { 9258 // Ignore it if it is the same component. 9259 if (CI == CE && SI == SE) 9260 continue; 9261 const auto It = (SI == SE) ? CI : SI; 9262 // If one component is a pointer and another one is a kind of 9263 // dereference of this pointer (array subscript, section, dereference, 9264 // etc.), it is not an overlapping. 9265 // Same, if one component is a base and another component is a 9266 // dereferenced pointer memberexpr with the same base. 9267 if (!isa<MemberExpr>(It->getAssociatedExpression()) || 9268 (std::prev(It)->getAssociatedDeclaration() && 9269 std::prev(It) 9270 ->getAssociatedDeclaration() 9271 ->getType() 9272 ->isPointerType()) || 9273 (It->getAssociatedDeclaration() && 9274 It->getAssociatedDeclaration()->getType()->isPointerType() && 9275 std::next(It) != CE && std::next(It) != SE)) 9276 continue; 9277 const MapData &BaseData = CI == CE ? L : L1; 9278 OMPClauseMappableExprCommon::MappableExprComponentListRef SubData = 9279 SI == SE ? Components : Components1; 9280 auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData); 9281 OverlappedElements.getSecond().push_back(SubData); 9282 } 9283 } 9284 } 9285 // Sort the overlapped elements for each item. 9286 llvm::SmallVector<const FieldDecl *, 4> Layout; 9287 if (!OverlappedData.empty()) { 9288 const Type *BaseType = VD->getType().getCanonicalType().getTypePtr(); 9289 const Type *OrigType = BaseType->getPointeeOrArrayElementType(); 9290 while (BaseType != OrigType) { 9291 BaseType = OrigType->getCanonicalTypeInternal().getTypePtr(); 9292 OrigType = BaseType->getPointeeOrArrayElementType(); 9293 } 9294 9295 if (const auto *CRD = BaseType->getAsCXXRecordDecl()) 9296 getPlainLayout(CRD, Layout, /*AsBase=*/false); 9297 else { 9298 const auto *RD = BaseType->getAsRecordDecl(); 9299 Layout.append(RD->field_begin(), RD->field_end()); 9300 } 9301 } 9302 for (auto &Pair : OverlappedData) { 9303 llvm::stable_sort( 9304 Pair.getSecond(), 9305 [&Layout]( 9306 OMPClauseMappableExprCommon::MappableExprComponentListRef First, 9307 OMPClauseMappableExprCommon::MappableExprComponentListRef 9308 Second) { 9309 auto CI = First.rbegin(); 9310 auto CE = First.rend(); 9311 auto SI = Second.rbegin(); 9312 auto SE = Second.rend(); 9313 for (; CI != CE && SI != SE; ++CI, ++SI) { 9314 if (CI->getAssociatedExpression()->getStmtClass() != 9315 SI->getAssociatedExpression()->getStmtClass()) 9316 break; 9317 // Are we dealing with different variables/fields? 9318 if (CI->getAssociatedDeclaration() != 9319 SI->getAssociatedDeclaration()) 9320 break; 9321 } 9322 9323 // Lists contain the same elements. 9324 if (CI == CE && SI == SE) 9325 return false; 9326 9327 // List with less elements is less than list with more elements. 9328 if (CI == CE || SI == SE) 9329 return CI == CE; 9330 9331 const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration()); 9332 const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration()); 9333 if (FD1->getParent() == FD2->getParent()) 9334 return FD1->getFieldIndex() < FD2->getFieldIndex(); 9335 const auto *It = 9336 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) { 9337 return FD == FD1 || FD == FD2; 9338 }); 9339 return *It == FD1; 9340 }); 9341 } 9342 9343 // Associated with a capture, because the mapping flags depend on it. 9344 // Go through all of the elements with the overlapped elements. 9345 bool IsFirstComponentList = true; 9346 for (const auto &Pair : OverlappedData) { 9347 const MapData &L = *Pair.getFirst(); 9348 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 9349 OpenMPMapClauseKind MapType; 9350 ArrayRef<OpenMPMapModifierKind> MapModifiers; 9351 bool IsImplicit; 9352 const ValueDecl *Mapper; 9353 const Expr *VarRef; 9354 std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper, VarRef) = 9355 L; 9356 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 9357 OverlappedComponents = Pair.getSecond(); 9358 generateInfoForComponentList( 9359 MapType, MapModifiers, llvm::None, Components, CombinedInfo, 9360 PartialStruct, IsFirstComponentList, IsImplicit, Mapper, 9361 /*ForDeviceAddr=*/false, VD, VarRef, OverlappedComponents); 9362 IsFirstComponentList = false; 9363 } 9364 // Go through other elements without overlapped elements. 9365 for (const MapData &L : DeclComponentLists) { 9366 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 9367 OpenMPMapClauseKind MapType; 9368 ArrayRef<OpenMPMapModifierKind> MapModifiers; 9369 bool IsImplicit; 9370 const ValueDecl *Mapper; 9371 const Expr *VarRef; 9372 std::tie(Components, MapType, MapModifiers, IsImplicit, Mapper, VarRef) = 9373 L; 9374 auto It = OverlappedData.find(&L); 9375 if (It == OverlappedData.end()) 9376 generateInfoForComponentList(MapType, MapModifiers, llvm::None, 9377 Components, CombinedInfo, PartialStruct, 9378 IsFirstComponentList, IsImplicit, Mapper, 9379 /*ForDeviceAddr=*/false, VD, VarRef); 9380 IsFirstComponentList = false; 9381 } 9382 } 9383 9384 /// Generate the default map information for a given capture \a CI, 9385 /// record field declaration \a RI and captured value \a CV. 9386 void generateDefaultMapInfo(const CapturedStmt::Capture &CI, 9387 const FieldDecl &RI, llvm::Value *CV, 9388 MapCombinedInfoTy &CombinedInfo) const { 9389 bool IsImplicit = true; 9390 // Do the default mapping. 9391 if (CI.capturesThis()) { 9392 CombinedInfo.Exprs.push_back(nullptr); 9393 CombinedInfo.BasePointers.push_back(CV); 9394 CombinedInfo.Pointers.push_back(CV); 9395 const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr()); 9396 CombinedInfo.Sizes.push_back( 9397 CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()), 9398 CGF.Int64Ty, /*isSigned=*/true)); 9399 // Default map type. 9400 CombinedInfo.Types.push_back(OMP_MAP_TO | OMP_MAP_FROM); 9401 } else if (CI.capturesVariableByCopy()) { 9402 const VarDecl *VD = CI.getCapturedVar(); 9403 CombinedInfo.Exprs.push_back(VD->getCanonicalDecl()); 9404 CombinedInfo.BasePointers.push_back(CV); 9405 CombinedInfo.Pointers.push_back(CV); 9406 if (!RI.getType()->isAnyPointerType()) { 9407 // We have to signal to the runtime captures passed by value that are 9408 // not pointers. 9409 CombinedInfo.Types.push_back(OMP_MAP_LITERAL); 9410 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 9411 CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true)); 9412 } else { 9413 // Pointers are implicitly mapped with a zero size and no flags 9414 // (other than first map that is added for all implicit maps). 9415 CombinedInfo.Types.push_back(OMP_MAP_NONE); 9416 CombinedInfo.Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty)); 9417 } 9418 auto I = FirstPrivateDecls.find(VD); 9419 if (I != FirstPrivateDecls.end()) 9420 IsImplicit = I->getSecond(); 9421 } else { 9422 assert(CI.capturesVariable() && "Expected captured reference."); 9423 const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr()); 9424 QualType ElementType = PtrTy->getPointeeType(); 9425 CombinedInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 9426 CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true)); 9427 // The default map type for a scalar/complex type is 'to' because by 9428 // default the value doesn't have to be retrieved. For an aggregate 9429 // type, the default is 'tofrom'. 9430 CombinedInfo.Types.push_back(getMapModifiersForPrivateClauses(CI)); 9431 const VarDecl *VD = CI.getCapturedVar(); 9432 auto I = FirstPrivateDecls.find(VD); 9433 CombinedInfo.Exprs.push_back(VD->getCanonicalDecl()); 9434 CombinedInfo.BasePointers.push_back(CV); 9435 if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) { 9436 Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue( 9437 CV, ElementType, CGF.getContext().getDeclAlign(VD), 9438 AlignmentSource::Decl)); 9439 CombinedInfo.Pointers.push_back(PtrAddr.getPointer()); 9440 } else { 9441 CombinedInfo.Pointers.push_back(CV); 9442 } 9443 if (I != FirstPrivateDecls.end()) 9444 IsImplicit = I->getSecond(); 9445 } 9446 // Every default map produces a single argument which is a target parameter. 9447 CombinedInfo.Types.back() |= OMP_MAP_TARGET_PARAM; 9448 9449 // Add flag stating this is an implicit map. 9450 if (IsImplicit) 9451 CombinedInfo.Types.back() |= OMP_MAP_IMPLICIT; 9452 9453 // No user-defined mapper for default mapping. 9454 CombinedInfo.Mappers.push_back(nullptr); 9455 } 9456 }; 9457 } // anonymous namespace 9458 9459 static void emitNonContiguousDescriptor( 9460 CodeGenFunction &CGF, MappableExprsHandler::MapCombinedInfoTy &CombinedInfo, 9461 CGOpenMPRuntime::TargetDataInfo &Info) { 9462 CodeGenModule &CGM = CGF.CGM; 9463 MappableExprsHandler::MapCombinedInfoTy::StructNonContiguousInfo 9464 &NonContigInfo = CombinedInfo.NonContigInfo; 9465 9466 // Build an array of struct descriptor_dim and then assign it to 9467 // offload_args. 9468 // 9469 // struct descriptor_dim { 9470 // uint64_t offset; 9471 // uint64_t count; 9472 // uint64_t stride 9473 // }; 9474 ASTContext &C = CGF.getContext(); 9475 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0); 9476 RecordDecl *RD; 9477 RD = C.buildImplicitRecord("descriptor_dim"); 9478 RD->startDefinition(); 9479 addFieldToRecordDecl(C, RD, Int64Ty); 9480 addFieldToRecordDecl(C, RD, Int64Ty); 9481 addFieldToRecordDecl(C, RD, Int64Ty); 9482 RD->completeDefinition(); 9483 QualType DimTy = C.getRecordType(RD); 9484 9485 enum { OffsetFD = 0, CountFD, StrideFD }; 9486 // We need two index variable here since the size of "Dims" is the same as the 9487 // size of Components, however, the size of offset, count, and stride is equal 9488 // to the size of base declaration that is non-contiguous. 9489 for (unsigned I = 0, L = 0, E = NonContigInfo.Dims.size(); I < E; ++I) { 9490 // Skip emitting ir if dimension size is 1 since it cannot be 9491 // non-contiguous. 9492 if (NonContigInfo.Dims[I] == 1) 9493 continue; 9494 llvm::APInt Size(/*numBits=*/32, NonContigInfo.Dims[I]); 9495 QualType ArrayTy = 9496 C.getConstantArrayType(DimTy, Size, nullptr, ArrayType::Normal, 0); 9497 Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims"); 9498 for (unsigned II = 0, EE = NonContigInfo.Dims[I]; II < EE; ++II) { 9499 unsigned RevIdx = EE - II - 1; 9500 LValue DimsLVal = CGF.MakeAddrLValue( 9501 CGF.Builder.CreateConstArrayGEP(DimsAddr, II), DimTy); 9502 // Offset 9503 LValue OffsetLVal = CGF.EmitLValueForField( 9504 DimsLVal, *std::next(RD->field_begin(), OffsetFD)); 9505 CGF.EmitStoreOfScalar(NonContigInfo.Offsets[L][RevIdx], OffsetLVal); 9506 // Count 9507 LValue CountLVal = CGF.EmitLValueForField( 9508 DimsLVal, *std::next(RD->field_begin(), CountFD)); 9509 CGF.EmitStoreOfScalar(NonContigInfo.Counts[L][RevIdx], CountLVal); 9510 // Stride 9511 LValue StrideLVal = CGF.EmitLValueForField( 9512 DimsLVal, *std::next(RD->field_begin(), StrideFD)); 9513 CGF.EmitStoreOfScalar(NonContigInfo.Strides[L][RevIdx], StrideLVal); 9514 } 9515 // args[I] = &dims 9516 Address DAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 9517 DimsAddr, CGM.Int8PtrTy, CGM.Int8Ty); 9518 llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32( 9519 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9520 Info.PointersArray, 0, I); 9521 Address PAddr(P, CGM.VoidPtrTy, CGF.getPointerAlign()); 9522 CGF.Builder.CreateStore(DAddr.getPointer(), PAddr); 9523 ++L; 9524 } 9525 } 9526 9527 // Try to extract the base declaration from a `this->x` expression if possible. 9528 static ValueDecl *getDeclFromThisExpr(const Expr *E) { 9529 if (!E) 9530 return nullptr; 9531 9532 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E->IgnoreParenCasts())) 9533 if (const MemberExpr *ME = 9534 dyn_cast<MemberExpr>(OASE->getBase()->IgnoreParenImpCasts())) 9535 return ME->getMemberDecl(); 9536 return nullptr; 9537 } 9538 9539 /// Emit a string constant containing the names of the values mapped to the 9540 /// offloading runtime library. 9541 llvm::Constant * 9542 emitMappingInformation(CodeGenFunction &CGF, llvm::OpenMPIRBuilder &OMPBuilder, 9543 MappableExprsHandler::MappingExprInfo &MapExprs) { 9544 9545 uint32_t SrcLocStrSize; 9546 if (!MapExprs.getMapDecl() && !MapExprs.getMapExpr()) 9547 return OMPBuilder.getOrCreateDefaultSrcLocStr(SrcLocStrSize); 9548 9549 SourceLocation Loc; 9550 if (!MapExprs.getMapDecl() && MapExprs.getMapExpr()) { 9551 if (const ValueDecl *VD = getDeclFromThisExpr(MapExprs.getMapExpr())) 9552 Loc = VD->getLocation(); 9553 else 9554 Loc = MapExprs.getMapExpr()->getExprLoc(); 9555 } else { 9556 Loc = MapExprs.getMapDecl()->getLocation(); 9557 } 9558 9559 std::string ExprName; 9560 if (MapExprs.getMapExpr()) { 9561 PrintingPolicy P(CGF.getContext().getLangOpts()); 9562 llvm::raw_string_ostream OS(ExprName); 9563 MapExprs.getMapExpr()->printPretty(OS, nullptr, P); 9564 OS.flush(); 9565 } else { 9566 ExprName = MapExprs.getMapDecl()->getNameAsString(); 9567 } 9568 9569 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc); 9570 return OMPBuilder.getOrCreateSrcLocStr(PLoc.getFilename(), ExprName, 9571 PLoc.getLine(), PLoc.getColumn(), 9572 SrcLocStrSize); 9573 } 9574 9575 /// Emit the arrays used to pass the captures and map information to the 9576 /// offloading runtime library. If there is no map or capture information, 9577 /// return nullptr by reference. 9578 static void emitOffloadingArrays( 9579 CodeGenFunction &CGF, MappableExprsHandler::MapCombinedInfoTy &CombinedInfo, 9580 CGOpenMPRuntime::TargetDataInfo &Info, llvm::OpenMPIRBuilder &OMPBuilder, 9581 bool IsNonContiguous = false) { 9582 CodeGenModule &CGM = CGF.CGM; 9583 ASTContext &Ctx = CGF.getContext(); 9584 9585 // Reset the array information. 9586 Info.clearArrayInfo(); 9587 Info.NumberOfPtrs = CombinedInfo.BasePointers.size(); 9588 9589 if (Info.NumberOfPtrs) { 9590 // Detect if we have any capture size requiring runtime evaluation of the 9591 // size so that a constant array could be eventually used. 9592 9593 llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true); 9594 QualType PointerArrayType = Ctx.getConstantArrayType( 9595 Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal, 9596 /*IndexTypeQuals=*/0); 9597 9598 Info.BasePointersArray = 9599 CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer(); 9600 Info.PointersArray = 9601 CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer(); 9602 Address MappersArray = 9603 CGF.CreateMemTemp(PointerArrayType, ".offload_mappers"); 9604 Info.MappersArray = MappersArray.getPointer(); 9605 9606 // If we don't have any VLA types or other types that require runtime 9607 // evaluation, we can use a constant array for the map sizes, otherwise we 9608 // need to fill up the arrays as we do for the pointers. 9609 QualType Int64Ty = 9610 Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 9611 SmallVector<llvm::Constant *> ConstSizes( 9612 CombinedInfo.Sizes.size(), llvm::ConstantInt::get(CGF.Int64Ty, 0)); 9613 llvm::SmallBitVector RuntimeSizes(CombinedInfo.Sizes.size()); 9614 for (unsigned I = 0, E = CombinedInfo.Sizes.size(); I < E; ++I) { 9615 if (auto *CI = dyn_cast<llvm::Constant>(CombinedInfo.Sizes[I])) { 9616 if (!isa<llvm::ConstantExpr>(CI) && !isa<llvm::GlobalValue>(CI)) { 9617 if (IsNonContiguous && (CombinedInfo.Types[I] & 9618 MappableExprsHandler::OMP_MAP_NON_CONTIG)) 9619 ConstSizes[I] = llvm::ConstantInt::get( 9620 CGF.Int64Ty, CombinedInfo.NonContigInfo.Dims[I]); 9621 else 9622 ConstSizes[I] = CI; 9623 continue; 9624 } 9625 } 9626 RuntimeSizes.set(I); 9627 } 9628 9629 if (RuntimeSizes.all()) { 9630 QualType SizeArrayType = Ctx.getConstantArrayType( 9631 Int64Ty, PointerNumAP, nullptr, ArrayType::Normal, 9632 /*IndexTypeQuals=*/0); 9633 Info.SizesArray = 9634 CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer(); 9635 } else { 9636 auto *SizesArrayInit = llvm::ConstantArray::get( 9637 llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes); 9638 std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"}); 9639 auto *SizesArrayGbl = new llvm::GlobalVariable( 9640 CGM.getModule(), SizesArrayInit->getType(), /*isConstant=*/true, 9641 llvm::GlobalValue::PrivateLinkage, SizesArrayInit, Name); 9642 SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 9643 if (RuntimeSizes.any()) { 9644 QualType SizeArrayType = Ctx.getConstantArrayType( 9645 Int64Ty, PointerNumAP, nullptr, ArrayType::Normal, 9646 /*IndexTypeQuals=*/0); 9647 Address Buffer = CGF.CreateMemTemp(SizeArrayType, ".offload_sizes"); 9648 llvm::Value *GblConstPtr = 9649 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 9650 SizesArrayGbl, CGM.Int64Ty->getPointerTo()); 9651 CGF.Builder.CreateMemCpy( 9652 Buffer, 9653 Address(GblConstPtr, CGM.Int64Ty, 9654 CGM.getNaturalTypeAlignment(Ctx.getIntTypeForBitwidth( 9655 /*DestWidth=*/64, /*Signed=*/false))), 9656 CGF.getTypeSize(SizeArrayType)); 9657 Info.SizesArray = Buffer.getPointer(); 9658 } else { 9659 Info.SizesArray = SizesArrayGbl; 9660 } 9661 } 9662 9663 // The map types are always constant so we don't need to generate code to 9664 // fill arrays. Instead, we create an array constant. 9665 SmallVector<uint64_t, 4> Mapping(CombinedInfo.Types.size(), 0); 9666 llvm::copy(CombinedInfo.Types, Mapping.begin()); 9667 std::string MaptypesName = 9668 CGM.getOpenMPRuntime().getName({"offload_maptypes"}); 9669 auto *MapTypesArrayGbl = 9670 OMPBuilder.createOffloadMaptypes(Mapping, MaptypesName); 9671 Info.MapTypesArray = MapTypesArrayGbl; 9672 9673 // The information types are only built if there is debug information 9674 // requested. 9675 if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo) { 9676 Info.MapNamesArray = llvm::Constant::getNullValue( 9677 llvm::Type::getInt8Ty(CGF.Builder.getContext())->getPointerTo()); 9678 } else { 9679 auto fillInfoMap = [&](MappableExprsHandler::MappingExprInfo &MapExpr) { 9680 return emitMappingInformation(CGF, OMPBuilder, MapExpr); 9681 }; 9682 SmallVector<llvm::Constant *, 4> InfoMap(CombinedInfo.Exprs.size()); 9683 llvm::transform(CombinedInfo.Exprs, InfoMap.begin(), fillInfoMap); 9684 std::string MapnamesName = 9685 CGM.getOpenMPRuntime().getName({"offload_mapnames"}); 9686 auto *MapNamesArrayGbl = 9687 OMPBuilder.createOffloadMapnames(InfoMap, MapnamesName); 9688 Info.MapNamesArray = MapNamesArrayGbl; 9689 } 9690 9691 // If there's a present map type modifier, it must not be applied to the end 9692 // of a region, so generate a separate map type array in that case. 9693 if (Info.separateBeginEndCalls()) { 9694 bool EndMapTypesDiffer = false; 9695 for (uint64_t &Type : Mapping) { 9696 if (Type & MappableExprsHandler::OMP_MAP_PRESENT) { 9697 Type &= ~MappableExprsHandler::OMP_MAP_PRESENT; 9698 EndMapTypesDiffer = true; 9699 } 9700 } 9701 if (EndMapTypesDiffer) { 9702 MapTypesArrayGbl = 9703 OMPBuilder.createOffloadMaptypes(Mapping, MaptypesName); 9704 Info.MapTypesArrayEnd = MapTypesArrayGbl; 9705 } 9706 } 9707 9708 for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) { 9709 llvm::Value *BPVal = *CombinedInfo.BasePointers[I]; 9710 llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32( 9711 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9712 Info.BasePointersArray, 0, I); 9713 BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 9714 BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0)); 9715 Address BPAddr(BP, BPVal->getType(), 9716 Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 9717 CGF.Builder.CreateStore(BPVal, BPAddr); 9718 9719 if (Info.requiresDevicePointerInfo()) 9720 if (const ValueDecl *DevVD = 9721 CombinedInfo.BasePointers[I].getDevicePtrDecl()) 9722 Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr); 9723 9724 llvm::Value *PVal = CombinedInfo.Pointers[I]; 9725 llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32( 9726 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9727 Info.PointersArray, 0, I); 9728 P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 9729 P, PVal->getType()->getPointerTo(/*AddrSpace=*/0)); 9730 Address PAddr(P, PVal->getType(), Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 9731 CGF.Builder.CreateStore(PVal, PAddr); 9732 9733 if (RuntimeSizes.test(I)) { 9734 llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32( 9735 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 9736 Info.SizesArray, 9737 /*Idx0=*/0, 9738 /*Idx1=*/I); 9739 Address SAddr(S, CGM.Int64Ty, Ctx.getTypeAlignInChars(Int64Ty)); 9740 CGF.Builder.CreateStore(CGF.Builder.CreateIntCast(CombinedInfo.Sizes[I], 9741 CGM.Int64Ty, 9742 /*isSigned=*/true), 9743 SAddr); 9744 } 9745 9746 // Fill up the mapper array. 9747 llvm::Value *MFunc = llvm::ConstantPointerNull::get(CGM.VoidPtrTy); 9748 if (CombinedInfo.Mappers[I]) { 9749 MFunc = CGM.getOpenMPRuntime().getOrCreateUserDefinedMapperFunc( 9750 cast<OMPDeclareMapperDecl>(CombinedInfo.Mappers[I])); 9751 MFunc = CGF.Builder.CreatePointerCast(MFunc, CGM.VoidPtrTy); 9752 Info.HasMapper = true; 9753 } 9754 Address MAddr = CGF.Builder.CreateConstArrayGEP(MappersArray, I); 9755 CGF.Builder.CreateStore(MFunc, MAddr); 9756 } 9757 } 9758 9759 if (!IsNonContiguous || CombinedInfo.NonContigInfo.Offsets.empty() || 9760 Info.NumberOfPtrs == 0) 9761 return; 9762 9763 emitNonContiguousDescriptor(CGF, CombinedInfo, Info); 9764 } 9765 9766 namespace { 9767 /// Additional arguments for emitOffloadingArraysArgument function. 9768 struct ArgumentsOptions { 9769 bool ForEndCall = false; 9770 ArgumentsOptions() = default; 9771 ArgumentsOptions(bool ForEndCall) : ForEndCall(ForEndCall) {} 9772 }; 9773 } // namespace 9774 9775 /// Emit the arguments to be passed to the runtime library based on the 9776 /// arrays of base pointers, pointers, sizes, map types, and mappers. If 9777 /// ForEndCall, emit map types to be passed for the end of the region instead of 9778 /// the beginning. 9779 static void emitOffloadingArraysArgument( 9780 CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg, 9781 llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg, 9782 llvm::Value *&MapTypesArrayArg, llvm::Value *&MapNamesArrayArg, 9783 llvm::Value *&MappersArrayArg, CGOpenMPRuntime::TargetDataInfo &Info, 9784 const ArgumentsOptions &Options = ArgumentsOptions()) { 9785 assert((!Options.ForEndCall || Info.separateBeginEndCalls()) && 9786 "expected region end call to runtime only when end call is separate"); 9787 CodeGenModule &CGM = CGF.CGM; 9788 if (Info.NumberOfPtrs) { 9789 BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 9790 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9791 Info.BasePointersArray, 9792 /*Idx0=*/0, /*Idx1=*/0); 9793 PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 9794 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9795 Info.PointersArray, 9796 /*Idx0=*/0, 9797 /*Idx1=*/0); 9798 SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 9799 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray, 9800 /*Idx0=*/0, /*Idx1=*/0); 9801 MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 9802 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 9803 Options.ForEndCall && Info.MapTypesArrayEnd ? Info.MapTypesArrayEnd 9804 : Info.MapTypesArray, 9805 /*Idx0=*/0, 9806 /*Idx1=*/0); 9807 9808 // Only emit the mapper information arrays if debug information is 9809 // requested. 9810 if (CGF.CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo) 9811 MapNamesArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 9812 else 9813 MapNamesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 9814 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 9815 Info.MapNamesArray, 9816 /*Idx0=*/0, 9817 /*Idx1=*/0); 9818 // If there is no user-defined mapper, set the mapper array to nullptr to 9819 // avoid an unnecessary data privatization 9820 if (!Info.HasMapper) 9821 MappersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 9822 else 9823 MappersArrayArg = 9824 CGF.Builder.CreatePointerCast(Info.MappersArray, CGM.VoidPtrPtrTy); 9825 } else { 9826 BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 9827 PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 9828 SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 9829 MapTypesArrayArg = 9830 llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 9831 MapNamesArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 9832 MappersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 9833 } 9834 } 9835 9836 /// Check for inner distribute directive. 9837 static const OMPExecutableDirective * 9838 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) { 9839 const auto *CS = D.getInnermostCapturedStmt(); 9840 const auto *Body = 9841 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 9842 const Stmt *ChildStmt = 9843 CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 9844 9845 if (const auto *NestedDir = 9846 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 9847 OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind(); 9848 switch (D.getDirectiveKind()) { 9849 case OMPD_target: 9850 if (isOpenMPDistributeDirective(DKind)) 9851 return NestedDir; 9852 if (DKind == OMPD_teams) { 9853 Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers( 9854 /*IgnoreCaptured=*/true); 9855 if (!Body) 9856 return nullptr; 9857 ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 9858 if (const auto *NND = 9859 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 9860 DKind = NND->getDirectiveKind(); 9861 if (isOpenMPDistributeDirective(DKind)) 9862 return NND; 9863 } 9864 } 9865 return nullptr; 9866 case OMPD_target_teams: 9867 if (isOpenMPDistributeDirective(DKind)) 9868 return NestedDir; 9869 return nullptr; 9870 case OMPD_target_parallel: 9871 case OMPD_target_simd: 9872 case OMPD_target_parallel_for: 9873 case OMPD_target_parallel_for_simd: 9874 return nullptr; 9875 case OMPD_target_teams_distribute: 9876 case OMPD_target_teams_distribute_simd: 9877 case OMPD_target_teams_distribute_parallel_for: 9878 case OMPD_target_teams_distribute_parallel_for_simd: 9879 case OMPD_parallel: 9880 case OMPD_for: 9881 case OMPD_parallel_for: 9882 case OMPD_parallel_master: 9883 case OMPD_parallel_sections: 9884 case OMPD_for_simd: 9885 case OMPD_parallel_for_simd: 9886 case OMPD_cancel: 9887 case OMPD_cancellation_point: 9888 case OMPD_ordered: 9889 case OMPD_threadprivate: 9890 case OMPD_allocate: 9891 case OMPD_task: 9892 case OMPD_simd: 9893 case OMPD_tile: 9894 case OMPD_unroll: 9895 case OMPD_sections: 9896 case OMPD_section: 9897 case OMPD_single: 9898 case OMPD_master: 9899 case OMPD_critical: 9900 case OMPD_taskyield: 9901 case OMPD_barrier: 9902 case OMPD_taskwait: 9903 case OMPD_taskgroup: 9904 case OMPD_atomic: 9905 case OMPD_flush: 9906 case OMPD_depobj: 9907 case OMPD_scan: 9908 case OMPD_teams: 9909 case OMPD_target_data: 9910 case OMPD_target_exit_data: 9911 case OMPD_target_enter_data: 9912 case OMPD_distribute: 9913 case OMPD_distribute_simd: 9914 case OMPD_distribute_parallel_for: 9915 case OMPD_distribute_parallel_for_simd: 9916 case OMPD_teams_distribute: 9917 case OMPD_teams_distribute_simd: 9918 case OMPD_teams_distribute_parallel_for: 9919 case OMPD_teams_distribute_parallel_for_simd: 9920 case OMPD_target_update: 9921 case OMPD_declare_simd: 9922 case OMPD_declare_variant: 9923 case OMPD_begin_declare_variant: 9924 case OMPD_end_declare_variant: 9925 case OMPD_declare_target: 9926 case OMPD_end_declare_target: 9927 case OMPD_declare_reduction: 9928 case OMPD_declare_mapper: 9929 case OMPD_taskloop: 9930 case OMPD_taskloop_simd: 9931 case OMPD_master_taskloop: 9932 case OMPD_master_taskloop_simd: 9933 case OMPD_parallel_master_taskloop: 9934 case OMPD_parallel_master_taskloop_simd: 9935 case OMPD_requires: 9936 case OMPD_metadirective: 9937 case OMPD_unknown: 9938 default: 9939 llvm_unreachable("Unexpected directive."); 9940 } 9941 } 9942 9943 return nullptr; 9944 } 9945 9946 /// Emit the user-defined mapper function. The code generation follows the 9947 /// pattern in the example below. 9948 /// \code 9949 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle, 9950 /// void *base, void *begin, 9951 /// int64_t size, int64_t type, 9952 /// void *name = nullptr) { 9953 /// // Allocate space for an array section first or add a base/begin for 9954 /// // pointer dereference. 9955 /// if ((size > 1 || (base != begin && maptype.IsPtrAndObj)) && 9956 /// !maptype.IsDelete) 9957 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 9958 /// size*sizeof(Ty), clearToFromMember(type)); 9959 /// // Map members. 9960 /// for (unsigned i = 0; i < size; i++) { 9961 /// // For each component specified by this mapper: 9962 /// for (auto c : begin[i]->all_components) { 9963 /// if (c.hasMapper()) 9964 /// (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size, 9965 /// c.arg_type, c.arg_name); 9966 /// else 9967 /// __tgt_push_mapper_component(rt_mapper_handle, c.arg_base, 9968 /// c.arg_begin, c.arg_size, c.arg_type, 9969 /// c.arg_name); 9970 /// } 9971 /// } 9972 /// // Delete the array section. 9973 /// if (size > 1 && maptype.IsDelete) 9974 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 9975 /// size*sizeof(Ty), clearToFromMember(type)); 9976 /// } 9977 /// \endcode 9978 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D, 9979 CodeGenFunction *CGF) { 9980 if (UDMMap.count(D) > 0) 9981 return; 9982 ASTContext &C = CGM.getContext(); 9983 QualType Ty = D->getType(); 9984 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 9985 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 9986 auto *MapperVarDecl = 9987 cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl()); 9988 SourceLocation Loc = D->getLocation(); 9989 CharUnits ElementSize = C.getTypeSizeInChars(Ty); 9990 llvm::Type *ElemTy = CGM.getTypes().ConvertTypeForMem(Ty); 9991 9992 // Prepare mapper function arguments and attributes. 9993 ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 9994 C.VoidPtrTy, ImplicitParamDecl::Other); 9995 ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 9996 ImplicitParamDecl::Other); 9997 ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 9998 C.VoidPtrTy, ImplicitParamDecl::Other); 9999 ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 10000 ImplicitParamDecl::Other); 10001 ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 10002 ImplicitParamDecl::Other); 10003 ImplicitParamDecl NameArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 10004 ImplicitParamDecl::Other); 10005 FunctionArgList Args; 10006 Args.push_back(&HandleArg); 10007 Args.push_back(&BaseArg); 10008 Args.push_back(&BeginArg); 10009 Args.push_back(&SizeArg); 10010 Args.push_back(&TypeArg); 10011 Args.push_back(&NameArg); 10012 const CGFunctionInfo &FnInfo = 10013 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 10014 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 10015 SmallString<64> TyStr; 10016 llvm::raw_svector_ostream Out(TyStr); 10017 CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out); 10018 std::string Name = getName({"omp_mapper", TyStr, D->getName()}); 10019 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 10020 Name, &CGM.getModule()); 10021 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 10022 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 10023 // Start the mapper function code generation. 10024 CodeGenFunction MapperCGF(CGM); 10025 MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 10026 // Compute the starting and end addresses of array elements. 10027 llvm::Value *Size = MapperCGF.EmitLoadOfScalar( 10028 MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false, 10029 C.getPointerType(Int64Ty), Loc); 10030 // Prepare common arguments for array initiation and deletion. 10031 llvm::Value *Handle = MapperCGF.EmitLoadOfScalar( 10032 MapperCGF.GetAddrOfLocalVar(&HandleArg), 10033 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 10034 llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar( 10035 MapperCGF.GetAddrOfLocalVar(&BaseArg), 10036 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 10037 llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar( 10038 MapperCGF.GetAddrOfLocalVar(&BeginArg), 10039 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 10040 // Convert the size in bytes into the number of array elements. 10041 Size = MapperCGF.Builder.CreateExactUDiv( 10042 Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity())); 10043 llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast( 10044 BeginIn, CGM.getTypes().ConvertTypeForMem(PtrTy)); 10045 llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(ElemTy, PtrBegin, Size); 10046 llvm::Value *MapType = MapperCGF.EmitLoadOfScalar( 10047 MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false, 10048 C.getPointerType(Int64Ty), Loc); 10049 llvm::Value *MapName = MapperCGF.EmitLoadOfScalar( 10050 MapperCGF.GetAddrOfLocalVar(&NameArg), 10051 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 10052 10053 // Emit array initiation if this is an array section and \p MapType indicates 10054 // that memory allocation is required. 10055 llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head"); 10056 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 10057 MapName, ElementSize, HeadBB, /*IsInit=*/true); 10058 10059 // Emit a for loop to iterate through SizeArg of elements and map all of them. 10060 10061 // Emit the loop header block. 10062 MapperCGF.EmitBlock(HeadBB); 10063 llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body"); 10064 llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done"); 10065 // Evaluate whether the initial condition is satisfied. 10066 llvm::Value *IsEmpty = 10067 MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty"); 10068 MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 10069 llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock(); 10070 10071 // Emit the loop body block. 10072 MapperCGF.EmitBlock(BodyBB); 10073 llvm::BasicBlock *LastBB = BodyBB; 10074 llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI( 10075 PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent"); 10076 PtrPHI->addIncoming(PtrBegin, EntryBB); 10077 Address PtrCurrent(PtrPHI, ElemTy, 10078 MapperCGF.GetAddrOfLocalVar(&BeginArg) 10079 .getAlignment() 10080 .alignmentOfArrayElement(ElementSize)); 10081 // Privatize the declared variable of mapper to be the current array element. 10082 CodeGenFunction::OMPPrivateScope Scope(MapperCGF); 10083 Scope.addPrivate(MapperVarDecl, PtrCurrent); 10084 (void)Scope.Privatize(); 10085 10086 // Get map clause information. Fill up the arrays with all mapped variables. 10087 MappableExprsHandler::MapCombinedInfoTy Info; 10088 MappableExprsHandler MEHandler(*D, MapperCGF); 10089 MEHandler.generateAllInfoForMapper(Info); 10090 10091 // Call the runtime API __tgt_mapper_num_components to get the number of 10092 // pre-existing components. 10093 llvm::Value *OffloadingArgs[] = {Handle}; 10094 llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall( 10095 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 10096 OMPRTL___tgt_mapper_num_components), 10097 OffloadingArgs); 10098 llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl( 10099 PreviousSize, 10100 MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset())); 10101 10102 // Fill up the runtime mapper handle for all components. 10103 for (unsigned I = 0; I < Info.BasePointers.size(); ++I) { 10104 llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast( 10105 *Info.BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 10106 llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast( 10107 Info.Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 10108 llvm::Value *CurSizeArg = Info.Sizes[I]; 10109 llvm::Value *CurNameArg = 10110 (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo) 10111 ? llvm::ConstantPointerNull::get(CGM.VoidPtrTy) 10112 : emitMappingInformation(MapperCGF, OMPBuilder, Info.Exprs[I]); 10113 10114 // Extract the MEMBER_OF field from the map type. 10115 llvm::Value *OriMapType = MapperCGF.Builder.getInt64(Info.Types[I]); 10116 llvm::Value *MemberMapType = 10117 MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize); 10118 10119 // Combine the map type inherited from user-defined mapper with that 10120 // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM 10121 // bits of the \a MapType, which is the input argument of the mapper 10122 // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM 10123 // bits of MemberMapType. 10124 // [OpenMP 5.0], 1.2.6. map-type decay. 10125 // | alloc | to | from | tofrom | release | delete 10126 // ---------------------------------------------------------- 10127 // alloc | alloc | alloc | alloc | alloc | release | delete 10128 // to | alloc | to | alloc | to | release | delete 10129 // from | alloc | alloc | from | from | release | delete 10130 // tofrom | alloc | to | from | tofrom | release | delete 10131 llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd( 10132 MapType, 10133 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO | 10134 MappableExprsHandler::OMP_MAP_FROM)); 10135 llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc"); 10136 llvm::BasicBlock *AllocElseBB = 10137 MapperCGF.createBasicBlock("omp.type.alloc.else"); 10138 llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to"); 10139 llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else"); 10140 llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from"); 10141 llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end"); 10142 llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom); 10143 MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB); 10144 // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM. 10145 MapperCGF.EmitBlock(AllocBB); 10146 llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd( 10147 MemberMapType, 10148 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 10149 MappableExprsHandler::OMP_MAP_FROM))); 10150 MapperCGF.Builder.CreateBr(EndBB); 10151 MapperCGF.EmitBlock(AllocElseBB); 10152 llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ( 10153 LeftToFrom, 10154 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO)); 10155 MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB); 10156 // In case of to, clear OMP_MAP_FROM. 10157 MapperCGF.EmitBlock(ToBB); 10158 llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd( 10159 MemberMapType, 10160 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM)); 10161 MapperCGF.Builder.CreateBr(EndBB); 10162 MapperCGF.EmitBlock(ToElseBB); 10163 llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ( 10164 LeftToFrom, 10165 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM)); 10166 MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB); 10167 // In case of from, clear OMP_MAP_TO. 10168 MapperCGF.EmitBlock(FromBB); 10169 llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd( 10170 MemberMapType, 10171 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO)); 10172 // In case of tofrom, do nothing. 10173 MapperCGF.EmitBlock(EndBB); 10174 LastBB = EndBB; 10175 llvm::PHINode *CurMapType = 10176 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype"); 10177 CurMapType->addIncoming(AllocMapType, AllocBB); 10178 CurMapType->addIncoming(ToMapType, ToBB); 10179 CurMapType->addIncoming(FromMapType, FromBB); 10180 CurMapType->addIncoming(MemberMapType, ToElseBB); 10181 10182 llvm::Value *OffloadingArgs[] = {Handle, CurBaseArg, CurBeginArg, 10183 CurSizeArg, CurMapType, CurNameArg}; 10184 if (Info.Mappers[I]) { 10185 // Call the corresponding mapper function. 10186 llvm::Function *MapperFunc = getOrCreateUserDefinedMapperFunc( 10187 cast<OMPDeclareMapperDecl>(Info.Mappers[I])); 10188 assert(MapperFunc && "Expect a valid mapper function is available."); 10189 MapperCGF.EmitNounwindRuntimeCall(MapperFunc, OffloadingArgs); 10190 } else { 10191 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 10192 // data structure. 10193 MapperCGF.EmitRuntimeCall( 10194 OMPBuilder.getOrCreateRuntimeFunction( 10195 CGM.getModule(), OMPRTL___tgt_push_mapper_component), 10196 OffloadingArgs); 10197 } 10198 } 10199 10200 // Update the pointer to point to the next element that needs to be mapped, 10201 // and check whether we have mapped all elements. 10202 llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32( 10203 ElemTy, PtrPHI, /*Idx0=*/1, "omp.arraymap.next"); 10204 PtrPHI->addIncoming(PtrNext, LastBB); 10205 llvm::Value *IsDone = 10206 MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone"); 10207 llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit"); 10208 MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB); 10209 10210 MapperCGF.EmitBlock(ExitBB); 10211 // Emit array deletion if this is an array section and \p MapType indicates 10212 // that deletion is required. 10213 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 10214 MapName, ElementSize, DoneBB, /*IsInit=*/false); 10215 10216 // Emit the function exit block. 10217 MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true); 10218 MapperCGF.FinishFunction(); 10219 UDMMap.try_emplace(D, Fn); 10220 if (CGF) { 10221 auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn); 10222 Decls.second.push_back(D); 10223 } 10224 } 10225 10226 /// Emit the array initialization or deletion portion for user-defined mapper 10227 /// code generation. First, it evaluates whether an array section is mapped and 10228 /// whether the \a MapType instructs to delete this section. If \a IsInit is 10229 /// true, and \a MapType indicates to not delete this array, array 10230 /// initialization code is generated. If \a IsInit is false, and \a MapType 10231 /// indicates to not this array, array deletion code is generated. 10232 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel( 10233 CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base, 10234 llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType, 10235 llvm::Value *MapName, CharUnits ElementSize, llvm::BasicBlock *ExitBB, 10236 bool IsInit) { 10237 StringRef Prefix = IsInit ? ".init" : ".del"; 10238 10239 // Evaluate if this is an array section. 10240 llvm::BasicBlock *BodyBB = 10241 MapperCGF.createBasicBlock(getName({"omp.array", Prefix})); 10242 llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGT( 10243 Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray"); 10244 llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd( 10245 MapType, 10246 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE)); 10247 llvm::Value *DeleteCond; 10248 llvm::Value *Cond; 10249 if (IsInit) { 10250 // base != begin? 10251 llvm::Value *BaseIsBegin = MapperCGF.Builder.CreateICmpNE(Base, Begin); 10252 // IsPtrAndObj? 10253 llvm::Value *PtrAndObjBit = MapperCGF.Builder.CreateAnd( 10254 MapType, 10255 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_PTR_AND_OBJ)); 10256 PtrAndObjBit = MapperCGF.Builder.CreateIsNotNull(PtrAndObjBit); 10257 BaseIsBegin = MapperCGF.Builder.CreateAnd(BaseIsBegin, PtrAndObjBit); 10258 Cond = MapperCGF.Builder.CreateOr(IsArray, BaseIsBegin); 10259 DeleteCond = MapperCGF.Builder.CreateIsNull( 10260 DeleteBit, getName({"omp.array", Prefix, ".delete"})); 10261 } else { 10262 Cond = IsArray; 10263 DeleteCond = MapperCGF.Builder.CreateIsNotNull( 10264 DeleteBit, getName({"omp.array", Prefix, ".delete"})); 10265 } 10266 Cond = MapperCGF.Builder.CreateAnd(Cond, DeleteCond); 10267 MapperCGF.Builder.CreateCondBr(Cond, BodyBB, ExitBB); 10268 10269 MapperCGF.EmitBlock(BodyBB); 10270 // Get the array size by multiplying element size and element number (i.e., \p 10271 // Size). 10272 llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul( 10273 Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity())); 10274 // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves 10275 // memory allocation/deletion purpose only. 10276 llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd( 10277 MapType, 10278 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 10279 MappableExprsHandler::OMP_MAP_FROM))); 10280 MapTypeArg = MapperCGF.Builder.CreateOr( 10281 MapTypeArg, 10282 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_IMPLICIT)); 10283 10284 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 10285 // data structure. 10286 llvm::Value *OffloadingArgs[] = {Handle, Base, Begin, 10287 ArraySize, MapTypeArg, MapName}; 10288 MapperCGF.EmitRuntimeCall( 10289 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 10290 OMPRTL___tgt_push_mapper_component), 10291 OffloadingArgs); 10292 } 10293 10294 llvm::Function *CGOpenMPRuntime::getOrCreateUserDefinedMapperFunc( 10295 const OMPDeclareMapperDecl *D) { 10296 auto I = UDMMap.find(D); 10297 if (I != UDMMap.end()) 10298 return I->second; 10299 emitUserDefinedMapper(D); 10300 return UDMMap.lookup(D); 10301 } 10302 10303 void CGOpenMPRuntime::emitTargetNumIterationsCall( 10304 CodeGenFunction &CGF, const OMPExecutableDirective &D, 10305 llvm::Value *DeviceID, 10306 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 10307 const OMPLoopDirective &D)> 10308 SizeEmitter) { 10309 OpenMPDirectiveKind Kind = D.getDirectiveKind(); 10310 const OMPExecutableDirective *TD = &D; 10311 // Get nested teams distribute kind directive, if any. 10312 if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind)) 10313 TD = getNestedDistributeDirective(CGM.getContext(), D); 10314 if (!TD) 10315 return; 10316 const auto *LD = cast<OMPLoopDirective>(TD); 10317 auto &&CodeGen = [LD, DeviceID, SizeEmitter, &D, this](CodeGenFunction &CGF, 10318 PrePostActionTy &) { 10319 if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) { 10320 llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 10321 llvm::Value *Args[] = {RTLoc, DeviceID, NumIterations}; 10322 CGF.EmitRuntimeCall( 10323 OMPBuilder.getOrCreateRuntimeFunction( 10324 CGM.getModule(), OMPRTL___kmpc_push_target_tripcount_mapper), 10325 Args); 10326 } 10327 }; 10328 emitInlinedDirective(CGF, OMPD_unknown, CodeGen); 10329 } 10330 10331 void CGOpenMPRuntime::emitTargetCall( 10332 CodeGenFunction &CGF, const OMPExecutableDirective &D, 10333 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 10334 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device, 10335 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 10336 const OMPLoopDirective &D)> 10337 SizeEmitter) { 10338 if (!CGF.HaveInsertPoint()) 10339 return; 10340 10341 const bool OffloadingMandatory = !CGM.getLangOpts().OpenMPIsDevice && 10342 CGM.getLangOpts().OpenMPOffloadMandatory; 10343 10344 assert((OffloadingMandatory || OutlinedFn) && "Invalid outlined function!"); 10345 10346 const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>() || 10347 D.hasClausesOfKind<OMPNowaitClause>(); 10348 llvm::SmallVector<llvm::Value *, 16> CapturedVars; 10349 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 10350 auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF, 10351 PrePostActionTy &) { 10352 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 10353 }; 10354 emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen); 10355 10356 CodeGenFunction::OMPTargetDataInfo InputInfo; 10357 llvm::Value *MapTypesArray = nullptr; 10358 llvm::Value *MapNamesArray = nullptr; 10359 // Generate code for the host fallback function. 10360 auto &&FallbackGen = [this, OutlinedFn, &D, &CapturedVars, RequiresOuterTask, 10361 &CS, OffloadingMandatory](CodeGenFunction &CGF) { 10362 if (OffloadingMandatory) { 10363 CGF.Builder.CreateUnreachable(); 10364 } else { 10365 if (RequiresOuterTask) { 10366 CapturedVars.clear(); 10367 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 10368 } 10369 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 10370 } 10371 }; 10372 // Fill up the pointer arrays and transfer execution to the device. 10373 auto &&ThenGen = [this, Device, OutlinedFnID, &D, &InputInfo, &MapTypesArray, 10374 &MapNamesArray, SizeEmitter, 10375 FallbackGen](CodeGenFunction &CGF, PrePostActionTy &) { 10376 if (Device.getInt() == OMPC_DEVICE_ancestor) { 10377 // Reverse offloading is not supported, so just execute on the host. 10378 FallbackGen(CGF); 10379 return; 10380 } 10381 10382 // On top of the arrays that were filled up, the target offloading call 10383 // takes as arguments the device id as well as the host pointer. The host 10384 // pointer is used by the runtime library to identify the current target 10385 // region, so it only has to be unique and not necessarily point to 10386 // anything. It could be the pointer to the outlined function that 10387 // implements the target region, but we aren't using that so that the 10388 // compiler doesn't need to keep that, and could therefore inline the host 10389 // function if proven worthwhile during optimization. 10390 10391 // From this point on, we need to have an ID of the target region defined. 10392 assert(OutlinedFnID && "Invalid outlined function ID!"); 10393 (void)OutlinedFnID; 10394 10395 // Emit device ID if any. 10396 llvm::Value *DeviceID; 10397 if (Device.getPointer()) { 10398 assert((Device.getInt() == OMPC_DEVICE_unknown || 10399 Device.getInt() == OMPC_DEVICE_device_num) && 10400 "Expected device_num modifier."); 10401 llvm::Value *DevVal = CGF.EmitScalarExpr(Device.getPointer()); 10402 DeviceID = 10403 CGF.Builder.CreateIntCast(DevVal, CGF.Int64Ty, /*isSigned=*/true); 10404 } else { 10405 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10406 } 10407 10408 // Emit the number of elements in the offloading arrays. 10409 llvm::Value *PointerNum = 10410 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 10411 10412 // Return value of the runtime offloading call. 10413 llvm::Value *Return; 10414 10415 llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D); 10416 llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D); 10417 10418 // Source location for the ident struct 10419 llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 10420 10421 // Emit tripcount for the target loop-based directive. 10422 emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter); 10423 10424 bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 10425 // The target region is an outlined function launched by the runtime 10426 // via calls __tgt_target() or __tgt_target_teams(). 10427 // 10428 // __tgt_target() launches a target region with one team and one thread, 10429 // executing a serial region. This master thread may in turn launch 10430 // more threads within its team upon encountering a parallel region, 10431 // however, no additional teams can be launched on the device. 10432 // 10433 // __tgt_target_teams() launches a target region with one or more teams, 10434 // each with one or more threads. This call is required for target 10435 // constructs such as: 10436 // 'target teams' 10437 // 'target' / 'teams' 10438 // 'target teams distribute parallel for' 10439 // 'target parallel' 10440 // and so on. 10441 // 10442 // Note that on the host and CPU targets, the runtime implementation of 10443 // these calls simply call the outlined function without forking threads. 10444 // The outlined functions themselves have runtime calls to 10445 // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by 10446 // the compiler in emitTeamsCall() and emitParallelCall(). 10447 // 10448 // In contrast, on the NVPTX target, the implementation of 10449 // __tgt_target_teams() launches a GPU kernel with the requested number 10450 // of teams and threads so no additional calls to the runtime are required. 10451 if (NumTeams) { 10452 // If we have NumTeams defined this means that we have an enclosed teams 10453 // region. Therefore we also expect to have NumThreads defined. These two 10454 // values should be defined in the presence of a teams directive, 10455 // regardless of having any clauses associated. If the user is using teams 10456 // but no clauses, these two values will be the default that should be 10457 // passed to the runtime library - a 32-bit integer with the value zero. 10458 assert(NumThreads && "Thread limit expression should be available along " 10459 "with number of teams."); 10460 SmallVector<llvm::Value *> OffloadingArgs = { 10461 RTLoc, 10462 DeviceID, 10463 OutlinedFnID, 10464 PointerNum, 10465 InputInfo.BasePointersArray.getPointer(), 10466 InputInfo.PointersArray.getPointer(), 10467 InputInfo.SizesArray.getPointer(), 10468 MapTypesArray, 10469 MapNamesArray, 10470 InputInfo.MappersArray.getPointer(), 10471 NumTeams, 10472 NumThreads}; 10473 if (HasNowait) { 10474 // Add int32_t depNum = 0, void *depList = nullptr, int32_t 10475 // noAliasDepNum = 0, void *noAliasDepList = nullptr. 10476 OffloadingArgs.push_back(CGF.Builder.getInt32(0)); 10477 OffloadingArgs.push_back(llvm::ConstantPointerNull::get(CGM.VoidPtrTy)); 10478 OffloadingArgs.push_back(CGF.Builder.getInt32(0)); 10479 OffloadingArgs.push_back(llvm::ConstantPointerNull::get(CGM.VoidPtrTy)); 10480 } 10481 Return = CGF.EmitRuntimeCall( 10482 OMPBuilder.getOrCreateRuntimeFunction( 10483 CGM.getModule(), HasNowait 10484 ? OMPRTL___tgt_target_teams_nowait_mapper 10485 : OMPRTL___tgt_target_teams_mapper), 10486 OffloadingArgs); 10487 } else { 10488 SmallVector<llvm::Value *> OffloadingArgs = { 10489 RTLoc, 10490 DeviceID, 10491 OutlinedFnID, 10492 PointerNum, 10493 InputInfo.BasePointersArray.getPointer(), 10494 InputInfo.PointersArray.getPointer(), 10495 InputInfo.SizesArray.getPointer(), 10496 MapTypesArray, 10497 MapNamesArray, 10498 InputInfo.MappersArray.getPointer()}; 10499 if (HasNowait) { 10500 // Add int32_t depNum = 0, void *depList = nullptr, int32_t 10501 // noAliasDepNum = 0, void *noAliasDepList = nullptr. 10502 OffloadingArgs.push_back(CGF.Builder.getInt32(0)); 10503 OffloadingArgs.push_back(llvm::ConstantPointerNull::get(CGM.VoidPtrTy)); 10504 OffloadingArgs.push_back(CGF.Builder.getInt32(0)); 10505 OffloadingArgs.push_back(llvm::ConstantPointerNull::get(CGM.VoidPtrTy)); 10506 } 10507 Return = CGF.EmitRuntimeCall( 10508 OMPBuilder.getOrCreateRuntimeFunction( 10509 CGM.getModule(), HasNowait ? OMPRTL___tgt_target_nowait_mapper 10510 : OMPRTL___tgt_target_mapper), 10511 OffloadingArgs); 10512 } 10513 10514 // Check the error code and execute the host version if required. 10515 llvm::BasicBlock *OffloadFailedBlock = 10516 CGF.createBasicBlock("omp_offload.failed"); 10517 llvm::BasicBlock *OffloadContBlock = 10518 CGF.createBasicBlock("omp_offload.cont"); 10519 llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return); 10520 CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock); 10521 10522 CGF.EmitBlock(OffloadFailedBlock); 10523 FallbackGen(CGF); 10524 10525 CGF.EmitBranch(OffloadContBlock); 10526 10527 CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true); 10528 }; 10529 10530 // Notify that the host version must be executed. 10531 auto &&ElseGen = [FallbackGen](CodeGenFunction &CGF, PrePostActionTy &) { 10532 FallbackGen(CGF); 10533 }; 10534 10535 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray, 10536 &MapNamesArray, &CapturedVars, RequiresOuterTask, 10537 &CS](CodeGenFunction &CGF, PrePostActionTy &) { 10538 // Fill up the arrays with all the captured variables. 10539 MappableExprsHandler::MapCombinedInfoTy CombinedInfo; 10540 10541 // Get mappable expression information. 10542 MappableExprsHandler MEHandler(D, CGF); 10543 llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers; 10544 llvm::DenseSet<CanonicalDeclPtr<const Decl>> MappedVarSet; 10545 10546 auto RI = CS.getCapturedRecordDecl()->field_begin(); 10547 auto *CV = CapturedVars.begin(); 10548 for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(), 10549 CE = CS.capture_end(); 10550 CI != CE; ++CI, ++RI, ++CV) { 10551 MappableExprsHandler::MapCombinedInfoTy CurInfo; 10552 MappableExprsHandler::StructRangeInfoTy PartialStruct; 10553 10554 // VLA sizes are passed to the outlined region by copy and do not have map 10555 // information associated. 10556 if (CI->capturesVariableArrayType()) { 10557 CurInfo.Exprs.push_back(nullptr); 10558 CurInfo.BasePointers.push_back(*CV); 10559 CurInfo.Pointers.push_back(*CV); 10560 CurInfo.Sizes.push_back(CGF.Builder.CreateIntCast( 10561 CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true)); 10562 // Copy to the device as an argument. No need to retrieve it. 10563 CurInfo.Types.push_back(MappableExprsHandler::OMP_MAP_LITERAL | 10564 MappableExprsHandler::OMP_MAP_TARGET_PARAM | 10565 MappableExprsHandler::OMP_MAP_IMPLICIT); 10566 CurInfo.Mappers.push_back(nullptr); 10567 } else { 10568 // If we have any information in the map clause, we use it, otherwise we 10569 // just do a default mapping. 10570 MEHandler.generateInfoForCapture(CI, *CV, CurInfo, PartialStruct); 10571 if (!CI->capturesThis()) 10572 MappedVarSet.insert(CI->getCapturedVar()); 10573 else 10574 MappedVarSet.insert(nullptr); 10575 if (CurInfo.BasePointers.empty() && !PartialStruct.Base.isValid()) 10576 MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurInfo); 10577 // Generate correct mapping for variables captured by reference in 10578 // lambdas. 10579 if (CI->capturesVariable()) 10580 MEHandler.generateInfoForLambdaCaptures(CI->getCapturedVar(), *CV, 10581 CurInfo, LambdaPointers); 10582 } 10583 // We expect to have at least an element of information for this capture. 10584 assert((!CurInfo.BasePointers.empty() || PartialStruct.Base.isValid()) && 10585 "Non-existing map pointer for capture!"); 10586 assert(CurInfo.BasePointers.size() == CurInfo.Pointers.size() && 10587 CurInfo.BasePointers.size() == CurInfo.Sizes.size() && 10588 CurInfo.BasePointers.size() == CurInfo.Types.size() && 10589 CurInfo.BasePointers.size() == CurInfo.Mappers.size() && 10590 "Inconsistent map information sizes!"); 10591 10592 // If there is an entry in PartialStruct it means we have a struct with 10593 // individual members mapped. Emit an extra combined entry. 10594 if (PartialStruct.Base.isValid()) { 10595 CombinedInfo.append(PartialStruct.PreliminaryMapData); 10596 MEHandler.emitCombinedEntry( 10597 CombinedInfo, CurInfo.Types, PartialStruct, nullptr, 10598 !PartialStruct.PreliminaryMapData.BasePointers.empty()); 10599 } 10600 10601 // We need to append the results of this capture to what we already have. 10602 CombinedInfo.append(CurInfo); 10603 } 10604 // Adjust MEMBER_OF flags for the lambdas captures. 10605 MEHandler.adjustMemberOfForLambdaCaptures( 10606 LambdaPointers, CombinedInfo.BasePointers, CombinedInfo.Pointers, 10607 CombinedInfo.Types); 10608 // Map any list items in a map clause that were not captures because they 10609 // weren't referenced within the construct. 10610 MEHandler.generateAllInfo(CombinedInfo, MappedVarSet); 10611 10612 TargetDataInfo Info; 10613 // Fill up the arrays and create the arguments. 10614 emitOffloadingArrays(CGF, CombinedInfo, Info, OMPBuilder); 10615 emitOffloadingArraysArgument( 10616 CGF, Info.BasePointersArray, Info.PointersArray, Info.SizesArray, 10617 Info.MapTypesArray, Info.MapNamesArray, Info.MappersArray, Info, 10618 {/*ForEndCall=*/false}); 10619 10620 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 10621 InputInfo.BasePointersArray = 10622 Address(Info.BasePointersArray, CGF.VoidPtrTy, CGM.getPointerAlign()); 10623 InputInfo.PointersArray = 10624 Address(Info.PointersArray, CGF.VoidPtrTy, CGM.getPointerAlign()); 10625 InputInfo.SizesArray = 10626 Address(Info.SizesArray, CGF.Int64Ty, CGM.getPointerAlign()); 10627 InputInfo.MappersArray = 10628 Address(Info.MappersArray, CGF.VoidPtrTy, CGM.getPointerAlign()); 10629 MapTypesArray = Info.MapTypesArray; 10630 MapNamesArray = Info.MapNamesArray; 10631 if (RequiresOuterTask) 10632 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 10633 else 10634 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 10635 }; 10636 10637 auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask]( 10638 CodeGenFunction &CGF, PrePostActionTy &) { 10639 if (RequiresOuterTask) { 10640 CodeGenFunction::OMPTargetDataInfo InputInfo; 10641 CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo); 10642 } else { 10643 emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen); 10644 } 10645 }; 10646 10647 // If we have a target function ID it means that we need to support 10648 // offloading, otherwise, just execute on the host. We need to execute on host 10649 // regardless of the conditional in the if clause if, e.g., the user do not 10650 // specify target triples. 10651 if (OutlinedFnID) { 10652 if (IfCond) { 10653 emitIfClause(CGF, IfCond, TargetThenGen, TargetElseGen); 10654 } else { 10655 RegionCodeGenTy ThenRCG(TargetThenGen); 10656 ThenRCG(CGF); 10657 } 10658 } else { 10659 RegionCodeGenTy ElseRCG(TargetElseGen); 10660 ElseRCG(CGF); 10661 } 10662 } 10663 10664 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S, 10665 StringRef ParentName) { 10666 if (!S) 10667 return; 10668 10669 // Codegen OMP target directives that offload compute to the device. 10670 bool RequiresDeviceCodegen = 10671 isa<OMPExecutableDirective>(S) && 10672 isOpenMPTargetExecutionDirective( 10673 cast<OMPExecutableDirective>(S)->getDirectiveKind()); 10674 10675 if (RequiresDeviceCodegen) { 10676 const auto &E = *cast<OMPExecutableDirective>(S); 10677 unsigned DeviceID; 10678 unsigned FileID; 10679 unsigned Line; 10680 getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID, 10681 FileID, Line); 10682 10683 // Is this a target region that should not be emitted as an entry point? If 10684 // so just signal we are done with this target region. 10685 if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID, 10686 ParentName, Line)) 10687 return; 10688 10689 switch (E.getDirectiveKind()) { 10690 case OMPD_target: 10691 CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName, 10692 cast<OMPTargetDirective>(E)); 10693 break; 10694 case OMPD_target_parallel: 10695 CodeGenFunction::EmitOMPTargetParallelDeviceFunction( 10696 CGM, ParentName, cast<OMPTargetParallelDirective>(E)); 10697 break; 10698 case OMPD_target_teams: 10699 CodeGenFunction::EmitOMPTargetTeamsDeviceFunction( 10700 CGM, ParentName, cast<OMPTargetTeamsDirective>(E)); 10701 break; 10702 case OMPD_target_teams_distribute: 10703 CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction( 10704 CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E)); 10705 break; 10706 case OMPD_target_teams_distribute_simd: 10707 CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction( 10708 CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E)); 10709 break; 10710 case OMPD_target_parallel_for: 10711 CodeGenFunction::EmitOMPTargetParallelForDeviceFunction( 10712 CGM, ParentName, cast<OMPTargetParallelForDirective>(E)); 10713 break; 10714 case OMPD_target_parallel_for_simd: 10715 CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction( 10716 CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E)); 10717 break; 10718 case OMPD_target_simd: 10719 CodeGenFunction::EmitOMPTargetSimdDeviceFunction( 10720 CGM, ParentName, cast<OMPTargetSimdDirective>(E)); 10721 break; 10722 case OMPD_target_teams_distribute_parallel_for: 10723 CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction( 10724 CGM, ParentName, 10725 cast<OMPTargetTeamsDistributeParallelForDirective>(E)); 10726 break; 10727 case OMPD_target_teams_distribute_parallel_for_simd: 10728 CodeGenFunction:: 10729 EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction( 10730 CGM, ParentName, 10731 cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E)); 10732 break; 10733 case OMPD_parallel: 10734 case OMPD_for: 10735 case OMPD_parallel_for: 10736 case OMPD_parallel_master: 10737 case OMPD_parallel_sections: 10738 case OMPD_for_simd: 10739 case OMPD_parallel_for_simd: 10740 case OMPD_cancel: 10741 case OMPD_cancellation_point: 10742 case OMPD_ordered: 10743 case OMPD_threadprivate: 10744 case OMPD_allocate: 10745 case OMPD_task: 10746 case OMPD_simd: 10747 case OMPD_tile: 10748 case OMPD_unroll: 10749 case OMPD_sections: 10750 case OMPD_section: 10751 case OMPD_single: 10752 case OMPD_master: 10753 case OMPD_critical: 10754 case OMPD_taskyield: 10755 case OMPD_barrier: 10756 case OMPD_taskwait: 10757 case OMPD_taskgroup: 10758 case OMPD_atomic: 10759 case OMPD_flush: 10760 case OMPD_depobj: 10761 case OMPD_scan: 10762 case OMPD_teams: 10763 case OMPD_target_data: 10764 case OMPD_target_exit_data: 10765 case OMPD_target_enter_data: 10766 case OMPD_distribute: 10767 case OMPD_distribute_simd: 10768 case OMPD_distribute_parallel_for: 10769 case OMPD_distribute_parallel_for_simd: 10770 case OMPD_teams_distribute: 10771 case OMPD_teams_distribute_simd: 10772 case OMPD_teams_distribute_parallel_for: 10773 case OMPD_teams_distribute_parallel_for_simd: 10774 case OMPD_target_update: 10775 case OMPD_declare_simd: 10776 case OMPD_declare_variant: 10777 case OMPD_begin_declare_variant: 10778 case OMPD_end_declare_variant: 10779 case OMPD_declare_target: 10780 case OMPD_end_declare_target: 10781 case OMPD_declare_reduction: 10782 case OMPD_declare_mapper: 10783 case OMPD_taskloop: 10784 case OMPD_taskloop_simd: 10785 case OMPD_master_taskloop: 10786 case OMPD_master_taskloop_simd: 10787 case OMPD_parallel_master_taskloop: 10788 case OMPD_parallel_master_taskloop_simd: 10789 case OMPD_requires: 10790 case OMPD_metadirective: 10791 case OMPD_unknown: 10792 default: 10793 llvm_unreachable("Unknown target directive for OpenMP device codegen."); 10794 } 10795 return; 10796 } 10797 10798 if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) { 10799 if (!E->hasAssociatedStmt() || !E->getAssociatedStmt()) 10800 return; 10801 10802 scanForTargetRegionsFunctions(E->getRawStmt(), ParentName); 10803 return; 10804 } 10805 10806 // If this is a lambda function, look into its body. 10807 if (const auto *L = dyn_cast<LambdaExpr>(S)) 10808 S = L->getBody(); 10809 10810 // Keep looking for target regions recursively. 10811 for (const Stmt *II : S->children()) 10812 scanForTargetRegionsFunctions(II, ParentName); 10813 } 10814 10815 static bool isAssumedToBeNotEmitted(const ValueDecl *VD, bool IsDevice) { 10816 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 10817 OMPDeclareTargetDeclAttr::getDeviceType(VD); 10818 if (!DevTy) 10819 return false; 10820 // Do not emit device_type(nohost) functions for the host. 10821 if (!IsDevice && DevTy == OMPDeclareTargetDeclAttr::DT_NoHost) 10822 return true; 10823 // Do not emit device_type(host) functions for the device. 10824 if (IsDevice && DevTy == OMPDeclareTargetDeclAttr::DT_Host) 10825 return true; 10826 return false; 10827 } 10828 10829 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) { 10830 // If emitting code for the host, we do not process FD here. Instead we do 10831 // the normal code generation. 10832 if (!CGM.getLangOpts().OpenMPIsDevice) { 10833 if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl())) 10834 if (isAssumedToBeNotEmitted(cast<ValueDecl>(FD), 10835 CGM.getLangOpts().OpenMPIsDevice)) 10836 return true; 10837 return false; 10838 } 10839 10840 const ValueDecl *VD = cast<ValueDecl>(GD.getDecl()); 10841 // Try to detect target regions in the function. 10842 if (const auto *FD = dyn_cast<FunctionDecl>(VD)) { 10843 StringRef Name = CGM.getMangledName(GD); 10844 scanForTargetRegionsFunctions(FD->getBody(), Name); 10845 if (isAssumedToBeNotEmitted(cast<ValueDecl>(FD), 10846 CGM.getLangOpts().OpenMPIsDevice)) 10847 return true; 10848 } 10849 10850 // Do not to emit function if it is not marked as declare target. 10851 return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) && 10852 AlreadyEmittedTargetDecls.count(VD) == 0; 10853 } 10854 10855 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 10856 if (isAssumedToBeNotEmitted(cast<ValueDecl>(GD.getDecl()), 10857 CGM.getLangOpts().OpenMPIsDevice)) 10858 return true; 10859 10860 if (!CGM.getLangOpts().OpenMPIsDevice) 10861 return false; 10862 10863 // Check if there are Ctors/Dtors in this declaration and look for target 10864 // regions in it. We use the complete variant to produce the kernel name 10865 // mangling. 10866 QualType RDTy = cast<VarDecl>(GD.getDecl())->getType(); 10867 if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) { 10868 for (const CXXConstructorDecl *Ctor : RD->ctors()) { 10869 StringRef ParentName = 10870 CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete)); 10871 scanForTargetRegionsFunctions(Ctor->getBody(), ParentName); 10872 } 10873 if (const CXXDestructorDecl *Dtor = RD->getDestructor()) { 10874 StringRef ParentName = 10875 CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete)); 10876 scanForTargetRegionsFunctions(Dtor->getBody(), ParentName); 10877 } 10878 } 10879 10880 // Do not to emit variable if it is not marked as declare target. 10881 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 10882 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration( 10883 cast<VarDecl>(GD.getDecl())); 10884 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 10885 (*Res == OMPDeclareTargetDeclAttr::MT_To && 10886 HasRequiresUnifiedSharedMemory)) { 10887 DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl())); 10888 return true; 10889 } 10890 return false; 10891 } 10892 10893 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD, 10894 llvm::Constant *Addr) { 10895 if (CGM.getLangOpts().OMPTargetTriples.empty() && 10896 !CGM.getLangOpts().OpenMPIsDevice) 10897 return; 10898 10899 // If we have host/nohost variables, they do not need to be registered. 10900 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 10901 OMPDeclareTargetDeclAttr::getDeviceType(VD); 10902 if (DevTy && DevTy.getValue() != OMPDeclareTargetDeclAttr::DT_Any) 10903 return; 10904 10905 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 10906 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 10907 if (!Res) { 10908 if (CGM.getLangOpts().OpenMPIsDevice) { 10909 // Register non-target variables being emitted in device code (debug info 10910 // may cause this). 10911 StringRef VarName = CGM.getMangledName(VD); 10912 EmittedNonTargetVariables.try_emplace(VarName, Addr); 10913 } 10914 return; 10915 } 10916 // Register declare target variables. 10917 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags; 10918 StringRef VarName; 10919 CharUnits VarSize; 10920 llvm::GlobalValue::LinkageTypes Linkage; 10921 10922 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 10923 !HasRequiresUnifiedSharedMemory) { 10924 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 10925 VarName = CGM.getMangledName(VD); 10926 if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) { 10927 VarSize = CGM.getContext().getTypeSizeInChars(VD->getType()); 10928 assert(!VarSize.isZero() && "Expected non-zero size of the variable"); 10929 } else { 10930 VarSize = CharUnits::Zero(); 10931 } 10932 Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false); 10933 // Temp solution to prevent optimizations of the internal variables. 10934 if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) { 10935 // Do not create a "ref-variable" if the original is not also available 10936 // on the host. 10937 if (!OffloadEntriesInfoManager.hasDeviceGlobalVarEntryInfo(VarName)) 10938 return; 10939 std::string RefName = getName({VarName, "ref"}); 10940 if (!CGM.GetGlobalValue(RefName)) { 10941 llvm::Constant *AddrRef = 10942 getOrCreateInternalVariable(Addr->getType(), RefName); 10943 auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef); 10944 GVAddrRef->setConstant(/*Val=*/true); 10945 GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage); 10946 GVAddrRef->setInitializer(Addr); 10947 CGM.addCompilerUsedGlobal(GVAddrRef); 10948 } 10949 } 10950 } else { 10951 assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 10952 (*Res == OMPDeclareTargetDeclAttr::MT_To && 10953 HasRequiresUnifiedSharedMemory)) && 10954 "Declare target attribute must link or to with unified memory."); 10955 if (*Res == OMPDeclareTargetDeclAttr::MT_Link) 10956 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink; 10957 else 10958 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 10959 10960 if (CGM.getLangOpts().OpenMPIsDevice) { 10961 VarName = Addr->getName(); 10962 Addr = nullptr; 10963 } else { 10964 VarName = getAddrOfDeclareTargetVar(VD).getName(); 10965 Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer()); 10966 } 10967 VarSize = CGM.getPointerSize(); 10968 Linkage = llvm::GlobalValue::WeakAnyLinkage; 10969 } 10970 10971 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 10972 VarName, Addr, VarSize, Flags, Linkage); 10973 } 10974 10975 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) { 10976 if (isa<FunctionDecl>(GD.getDecl()) || 10977 isa<OMPDeclareReductionDecl>(GD.getDecl())) 10978 return emitTargetFunctions(GD); 10979 10980 return emitTargetGlobalVariable(GD); 10981 } 10982 10983 void CGOpenMPRuntime::emitDeferredTargetDecls() const { 10984 for (const VarDecl *VD : DeferredGlobalVariables) { 10985 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 10986 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 10987 if (!Res) 10988 continue; 10989 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 10990 !HasRequiresUnifiedSharedMemory) { 10991 CGM.EmitGlobal(VD); 10992 } else { 10993 assert((*Res == OMPDeclareTargetDeclAttr::MT_Link || 10994 (*Res == OMPDeclareTargetDeclAttr::MT_To && 10995 HasRequiresUnifiedSharedMemory)) && 10996 "Expected link clause or to clause with unified memory."); 10997 (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 10998 } 10999 } 11000 } 11001 11002 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas( 11003 CodeGenFunction &CGF, const OMPExecutableDirective &D) const { 11004 assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) && 11005 " Expected target-based directive."); 11006 } 11007 11008 void CGOpenMPRuntime::processRequiresDirective(const OMPRequiresDecl *D) { 11009 for (const OMPClause *Clause : D->clauselists()) { 11010 if (Clause->getClauseKind() == OMPC_unified_shared_memory) { 11011 HasRequiresUnifiedSharedMemory = true; 11012 } else if (const auto *AC = 11013 dyn_cast<OMPAtomicDefaultMemOrderClause>(Clause)) { 11014 switch (AC->getAtomicDefaultMemOrderKind()) { 11015 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel: 11016 RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease; 11017 break; 11018 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst: 11019 RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent; 11020 break; 11021 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed: 11022 RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic; 11023 break; 11024 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown: 11025 break; 11026 } 11027 } 11028 } 11029 } 11030 11031 llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const { 11032 return RequiresAtomicOrdering; 11033 } 11034 11035 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD, 11036 LangAS &AS) { 11037 if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>()) 11038 return false; 11039 const auto *A = VD->getAttr<OMPAllocateDeclAttr>(); 11040 switch(A->getAllocatorType()) { 11041 case OMPAllocateDeclAttr::OMPNullMemAlloc: 11042 case OMPAllocateDeclAttr::OMPDefaultMemAlloc: 11043 // Not supported, fallback to the default mem space. 11044 case OMPAllocateDeclAttr::OMPLargeCapMemAlloc: 11045 case OMPAllocateDeclAttr::OMPCGroupMemAlloc: 11046 case OMPAllocateDeclAttr::OMPHighBWMemAlloc: 11047 case OMPAllocateDeclAttr::OMPLowLatMemAlloc: 11048 case OMPAllocateDeclAttr::OMPThreadMemAlloc: 11049 case OMPAllocateDeclAttr::OMPConstMemAlloc: 11050 case OMPAllocateDeclAttr::OMPPTeamMemAlloc: 11051 AS = LangAS::Default; 11052 return true; 11053 case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc: 11054 llvm_unreachable("Expected predefined allocator for the variables with the " 11055 "static storage."); 11056 } 11057 return false; 11058 } 11059 11060 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const { 11061 return HasRequiresUnifiedSharedMemory; 11062 } 11063 11064 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII( 11065 CodeGenModule &CGM) 11066 : CGM(CGM) { 11067 if (CGM.getLangOpts().OpenMPIsDevice) { 11068 SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal; 11069 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false; 11070 } 11071 } 11072 11073 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() { 11074 if (CGM.getLangOpts().OpenMPIsDevice) 11075 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal; 11076 } 11077 11078 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) { 11079 if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal) 11080 return true; 11081 11082 const auto *D = cast<FunctionDecl>(GD.getDecl()); 11083 // Do not to emit function if it is marked as declare target as it was already 11084 // emitted. 11085 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) { 11086 if (D->hasBody() && AlreadyEmittedTargetDecls.count(D) == 0) { 11087 if (auto *F = dyn_cast_or_null<llvm::Function>( 11088 CGM.GetGlobalValue(CGM.getMangledName(GD)))) 11089 return !F->isDeclaration(); 11090 return false; 11091 } 11092 return true; 11093 } 11094 11095 return !AlreadyEmittedTargetDecls.insert(D).second; 11096 } 11097 11098 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() { 11099 // If we don't have entries or if we are emitting code for the device, we 11100 // don't need to do anything. 11101 if (CGM.getLangOpts().OMPTargetTriples.empty() || 11102 CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice || 11103 (OffloadEntriesInfoManager.empty() && 11104 !HasEmittedDeclareTargetRegion && 11105 !HasEmittedTargetRegion)) 11106 return nullptr; 11107 11108 // Create and register the function that handles the requires directives. 11109 ASTContext &C = CGM.getContext(); 11110 11111 llvm::Function *RequiresRegFn; 11112 { 11113 CodeGenFunction CGF(CGM); 11114 const auto &FI = CGM.getTypes().arrangeNullaryFunction(); 11115 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 11116 std::string ReqName = getName({"omp_offloading", "requires_reg"}); 11117 RequiresRegFn = CGM.CreateGlobalInitOrCleanUpFunction(FTy, ReqName, FI); 11118 CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {}); 11119 OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE; 11120 // TODO: check for other requires clauses. 11121 // The requires directive takes effect only when a target region is 11122 // present in the compilation unit. Otherwise it is ignored and not 11123 // passed to the runtime. This avoids the runtime from throwing an error 11124 // for mismatching requires clauses across compilation units that don't 11125 // contain at least 1 target region. 11126 assert((HasEmittedTargetRegion || 11127 HasEmittedDeclareTargetRegion || 11128 !OffloadEntriesInfoManager.empty()) && 11129 "Target or declare target region expected."); 11130 if (HasRequiresUnifiedSharedMemory) 11131 Flags = OMP_REQ_UNIFIED_SHARED_MEMORY; 11132 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 11133 CGM.getModule(), OMPRTL___tgt_register_requires), 11134 llvm::ConstantInt::get(CGM.Int64Ty, Flags)); 11135 CGF.FinishFunction(); 11136 } 11137 return RequiresRegFn; 11138 } 11139 11140 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF, 11141 const OMPExecutableDirective &D, 11142 SourceLocation Loc, 11143 llvm::Function *OutlinedFn, 11144 ArrayRef<llvm::Value *> CapturedVars) { 11145 if (!CGF.HaveInsertPoint()) 11146 return; 11147 11148 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 11149 CodeGenFunction::RunCleanupsScope Scope(CGF); 11150 11151 // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn); 11152 llvm::Value *Args[] = { 11153 RTLoc, 11154 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 11155 CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())}; 11156 llvm::SmallVector<llvm::Value *, 16> RealArgs; 11157 RealArgs.append(std::begin(Args), std::end(Args)); 11158 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 11159 11160 llvm::FunctionCallee RTLFn = OMPBuilder.getOrCreateRuntimeFunction( 11161 CGM.getModule(), OMPRTL___kmpc_fork_teams); 11162 CGF.EmitRuntimeCall(RTLFn, RealArgs); 11163 } 11164 11165 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 11166 const Expr *NumTeams, 11167 const Expr *ThreadLimit, 11168 SourceLocation Loc) { 11169 if (!CGF.HaveInsertPoint()) 11170 return; 11171 11172 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 11173 11174 llvm::Value *NumTeamsVal = 11175 NumTeams 11176 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams), 11177 CGF.CGM.Int32Ty, /* isSigned = */ true) 11178 : CGF.Builder.getInt32(0); 11179 11180 llvm::Value *ThreadLimitVal = 11181 ThreadLimit 11182 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit), 11183 CGF.CGM.Int32Ty, /* isSigned = */ true) 11184 : CGF.Builder.getInt32(0); 11185 11186 // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit) 11187 llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal, 11188 ThreadLimitVal}; 11189 CGF.EmitRuntimeCall(OMPBuilder.getOrCreateRuntimeFunction( 11190 CGM.getModule(), OMPRTL___kmpc_push_num_teams), 11191 PushNumTeamsArgs); 11192 } 11193 11194 void CGOpenMPRuntime::emitTargetDataCalls( 11195 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 11196 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 11197 if (!CGF.HaveInsertPoint()) 11198 return; 11199 11200 // Action used to replace the default codegen action and turn privatization 11201 // off. 11202 PrePostActionTy NoPrivAction; 11203 11204 // Generate the code for the opening of the data environment. Capture all the 11205 // arguments of the runtime call by reference because they are used in the 11206 // closing of the region. 11207 auto &&BeginThenGen = [this, &D, Device, &Info, 11208 &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) { 11209 // Fill up the arrays with all the mapped variables. 11210 MappableExprsHandler::MapCombinedInfoTy CombinedInfo; 11211 11212 // Get map clause information. 11213 MappableExprsHandler MEHandler(D, CGF); 11214 MEHandler.generateAllInfo(CombinedInfo); 11215 11216 // Fill up the arrays and create the arguments. 11217 emitOffloadingArrays(CGF, CombinedInfo, Info, OMPBuilder, 11218 /*IsNonContiguous=*/true); 11219 11220 llvm::Value *BasePointersArrayArg = nullptr; 11221 llvm::Value *PointersArrayArg = nullptr; 11222 llvm::Value *SizesArrayArg = nullptr; 11223 llvm::Value *MapTypesArrayArg = nullptr; 11224 llvm::Value *MapNamesArrayArg = nullptr; 11225 llvm::Value *MappersArrayArg = nullptr; 11226 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 11227 SizesArrayArg, MapTypesArrayArg, 11228 MapNamesArrayArg, MappersArrayArg, Info); 11229 11230 // Emit device ID if any. 11231 llvm::Value *DeviceID = nullptr; 11232 if (Device) { 11233 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 11234 CGF.Int64Ty, /*isSigned=*/true); 11235 } else { 11236 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 11237 } 11238 11239 // Emit the number of elements in the offloading arrays. 11240 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 11241 // 11242 // Source location for the ident struct 11243 llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 11244 11245 llvm::Value *OffloadingArgs[] = {RTLoc, 11246 DeviceID, 11247 PointerNum, 11248 BasePointersArrayArg, 11249 PointersArrayArg, 11250 SizesArrayArg, 11251 MapTypesArrayArg, 11252 MapNamesArrayArg, 11253 MappersArrayArg}; 11254 CGF.EmitRuntimeCall( 11255 OMPBuilder.getOrCreateRuntimeFunction( 11256 CGM.getModule(), OMPRTL___tgt_target_data_begin_mapper), 11257 OffloadingArgs); 11258 11259 // If device pointer privatization is required, emit the body of the region 11260 // here. It will have to be duplicated: with and without privatization. 11261 if (!Info.CaptureDeviceAddrMap.empty()) 11262 CodeGen(CGF); 11263 }; 11264 11265 // Generate code for the closing of the data region. 11266 auto &&EndThenGen = [this, Device, &Info, &D](CodeGenFunction &CGF, 11267 PrePostActionTy &) { 11268 assert(Info.isValid() && "Invalid data environment closing arguments."); 11269 11270 llvm::Value *BasePointersArrayArg = nullptr; 11271 llvm::Value *PointersArrayArg = nullptr; 11272 llvm::Value *SizesArrayArg = nullptr; 11273 llvm::Value *MapTypesArrayArg = nullptr; 11274 llvm::Value *MapNamesArrayArg = nullptr; 11275 llvm::Value *MappersArrayArg = nullptr; 11276 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 11277 SizesArrayArg, MapTypesArrayArg, 11278 MapNamesArrayArg, MappersArrayArg, Info, 11279 {/*ForEndCall=*/true}); 11280 11281 // Emit device ID if any. 11282 llvm::Value *DeviceID = nullptr; 11283 if (Device) { 11284 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 11285 CGF.Int64Ty, /*isSigned=*/true); 11286 } else { 11287 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 11288 } 11289 11290 // Emit the number of elements in the offloading arrays. 11291 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 11292 11293 // Source location for the ident struct 11294 llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 11295 11296 llvm::Value *OffloadingArgs[] = {RTLoc, 11297 DeviceID, 11298 PointerNum, 11299 BasePointersArrayArg, 11300 PointersArrayArg, 11301 SizesArrayArg, 11302 MapTypesArrayArg, 11303 MapNamesArrayArg, 11304 MappersArrayArg}; 11305 CGF.EmitRuntimeCall( 11306 OMPBuilder.getOrCreateRuntimeFunction( 11307 CGM.getModule(), OMPRTL___tgt_target_data_end_mapper), 11308 OffloadingArgs); 11309 }; 11310 11311 // If we need device pointer privatization, we need to emit the body of the 11312 // region with no privatization in the 'else' branch of the conditional. 11313 // Otherwise, we don't have to do anything. 11314 auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF, 11315 PrePostActionTy &) { 11316 if (!Info.CaptureDeviceAddrMap.empty()) { 11317 CodeGen.setAction(NoPrivAction); 11318 CodeGen(CGF); 11319 } 11320 }; 11321 11322 // We don't have to do anything to close the region if the if clause evaluates 11323 // to false. 11324 auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {}; 11325 11326 if (IfCond) { 11327 emitIfClause(CGF, IfCond, BeginThenGen, BeginElseGen); 11328 } else { 11329 RegionCodeGenTy RCG(BeginThenGen); 11330 RCG(CGF); 11331 } 11332 11333 // If we don't require privatization of device pointers, we emit the body in 11334 // between the runtime calls. This avoids duplicating the body code. 11335 if (Info.CaptureDeviceAddrMap.empty()) { 11336 CodeGen.setAction(NoPrivAction); 11337 CodeGen(CGF); 11338 } 11339 11340 if (IfCond) { 11341 emitIfClause(CGF, IfCond, EndThenGen, EndElseGen); 11342 } else { 11343 RegionCodeGenTy RCG(EndThenGen); 11344 RCG(CGF); 11345 } 11346 } 11347 11348 void CGOpenMPRuntime::emitTargetDataStandAloneCall( 11349 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 11350 const Expr *Device) { 11351 if (!CGF.HaveInsertPoint()) 11352 return; 11353 11354 assert((isa<OMPTargetEnterDataDirective>(D) || 11355 isa<OMPTargetExitDataDirective>(D) || 11356 isa<OMPTargetUpdateDirective>(D)) && 11357 "Expecting either target enter, exit data, or update directives."); 11358 11359 CodeGenFunction::OMPTargetDataInfo InputInfo; 11360 llvm::Value *MapTypesArray = nullptr; 11361 llvm::Value *MapNamesArray = nullptr; 11362 // Generate the code for the opening of the data environment. 11363 auto &&ThenGen = [this, &D, Device, &InputInfo, &MapTypesArray, 11364 &MapNamesArray](CodeGenFunction &CGF, PrePostActionTy &) { 11365 // Emit device ID if any. 11366 llvm::Value *DeviceID = nullptr; 11367 if (Device) { 11368 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 11369 CGF.Int64Ty, /*isSigned=*/true); 11370 } else { 11371 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 11372 } 11373 11374 // Emit the number of elements in the offloading arrays. 11375 llvm::Constant *PointerNum = 11376 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 11377 11378 // Source location for the ident struct 11379 llvm::Value *RTLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 11380 11381 llvm::Value *OffloadingArgs[] = {RTLoc, 11382 DeviceID, 11383 PointerNum, 11384 InputInfo.BasePointersArray.getPointer(), 11385 InputInfo.PointersArray.getPointer(), 11386 InputInfo.SizesArray.getPointer(), 11387 MapTypesArray, 11388 MapNamesArray, 11389 InputInfo.MappersArray.getPointer()}; 11390 11391 // Select the right runtime function call for each standalone 11392 // directive. 11393 const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 11394 RuntimeFunction RTLFn; 11395 switch (D.getDirectiveKind()) { 11396 case OMPD_target_enter_data: 11397 RTLFn = HasNowait ? OMPRTL___tgt_target_data_begin_nowait_mapper 11398 : OMPRTL___tgt_target_data_begin_mapper; 11399 break; 11400 case OMPD_target_exit_data: 11401 RTLFn = HasNowait ? OMPRTL___tgt_target_data_end_nowait_mapper 11402 : OMPRTL___tgt_target_data_end_mapper; 11403 break; 11404 case OMPD_target_update: 11405 RTLFn = HasNowait ? OMPRTL___tgt_target_data_update_nowait_mapper 11406 : OMPRTL___tgt_target_data_update_mapper; 11407 break; 11408 case OMPD_parallel: 11409 case OMPD_for: 11410 case OMPD_parallel_for: 11411 case OMPD_parallel_master: 11412 case OMPD_parallel_sections: 11413 case OMPD_for_simd: 11414 case OMPD_parallel_for_simd: 11415 case OMPD_cancel: 11416 case OMPD_cancellation_point: 11417 case OMPD_ordered: 11418 case OMPD_threadprivate: 11419 case OMPD_allocate: 11420 case OMPD_task: 11421 case OMPD_simd: 11422 case OMPD_tile: 11423 case OMPD_unroll: 11424 case OMPD_sections: 11425 case OMPD_section: 11426 case OMPD_single: 11427 case OMPD_master: 11428 case OMPD_critical: 11429 case OMPD_taskyield: 11430 case OMPD_barrier: 11431 case OMPD_taskwait: 11432 case OMPD_taskgroup: 11433 case OMPD_atomic: 11434 case OMPD_flush: 11435 case OMPD_depobj: 11436 case OMPD_scan: 11437 case OMPD_teams: 11438 case OMPD_target_data: 11439 case OMPD_distribute: 11440 case OMPD_distribute_simd: 11441 case OMPD_distribute_parallel_for: 11442 case OMPD_distribute_parallel_for_simd: 11443 case OMPD_teams_distribute: 11444 case OMPD_teams_distribute_simd: 11445 case OMPD_teams_distribute_parallel_for: 11446 case OMPD_teams_distribute_parallel_for_simd: 11447 case OMPD_declare_simd: 11448 case OMPD_declare_variant: 11449 case OMPD_begin_declare_variant: 11450 case OMPD_end_declare_variant: 11451 case OMPD_declare_target: 11452 case OMPD_end_declare_target: 11453 case OMPD_declare_reduction: 11454 case OMPD_declare_mapper: 11455 case OMPD_taskloop: 11456 case OMPD_taskloop_simd: 11457 case OMPD_master_taskloop: 11458 case OMPD_master_taskloop_simd: 11459 case OMPD_parallel_master_taskloop: 11460 case OMPD_parallel_master_taskloop_simd: 11461 case OMPD_target: 11462 case OMPD_target_simd: 11463 case OMPD_target_teams_distribute: 11464 case OMPD_target_teams_distribute_simd: 11465 case OMPD_target_teams_distribute_parallel_for: 11466 case OMPD_target_teams_distribute_parallel_for_simd: 11467 case OMPD_target_teams: 11468 case OMPD_target_parallel: 11469 case OMPD_target_parallel_for: 11470 case OMPD_target_parallel_for_simd: 11471 case OMPD_requires: 11472 case OMPD_metadirective: 11473 case OMPD_unknown: 11474 default: 11475 llvm_unreachable("Unexpected standalone target data directive."); 11476 break; 11477 } 11478 CGF.EmitRuntimeCall( 11479 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), RTLFn), 11480 OffloadingArgs); 11481 }; 11482 11483 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray, 11484 &MapNamesArray](CodeGenFunction &CGF, 11485 PrePostActionTy &) { 11486 // Fill up the arrays with all the mapped variables. 11487 MappableExprsHandler::MapCombinedInfoTy CombinedInfo; 11488 11489 // Get map clause information. 11490 MappableExprsHandler MEHandler(D, CGF); 11491 MEHandler.generateAllInfo(CombinedInfo); 11492 11493 TargetDataInfo Info; 11494 // Fill up the arrays and create the arguments. 11495 emitOffloadingArrays(CGF, CombinedInfo, Info, OMPBuilder, 11496 /*IsNonContiguous=*/true); 11497 bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>() || 11498 D.hasClausesOfKind<OMPNowaitClause>(); 11499 emitOffloadingArraysArgument( 11500 CGF, Info.BasePointersArray, Info.PointersArray, Info.SizesArray, 11501 Info.MapTypesArray, Info.MapNamesArray, Info.MappersArray, Info, 11502 {/*ForEndCall=*/false}); 11503 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 11504 InputInfo.BasePointersArray = 11505 Address(Info.BasePointersArray, CGF.VoidPtrTy, CGM.getPointerAlign()); 11506 InputInfo.PointersArray = 11507 Address(Info.PointersArray, CGF.VoidPtrTy, CGM.getPointerAlign()); 11508 InputInfo.SizesArray = 11509 Address(Info.SizesArray, CGF.Int64Ty, CGM.getPointerAlign()); 11510 InputInfo.MappersArray = 11511 Address(Info.MappersArray, CGF.VoidPtrTy, CGM.getPointerAlign()); 11512 MapTypesArray = Info.MapTypesArray; 11513 MapNamesArray = Info.MapNamesArray; 11514 if (RequiresOuterTask) 11515 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 11516 else 11517 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 11518 }; 11519 11520 if (IfCond) { 11521 emitIfClause(CGF, IfCond, TargetThenGen, 11522 [](CodeGenFunction &CGF, PrePostActionTy &) {}); 11523 } else { 11524 RegionCodeGenTy ThenRCG(TargetThenGen); 11525 ThenRCG(CGF); 11526 } 11527 } 11528 11529 namespace { 11530 /// Kind of parameter in a function with 'declare simd' directive. 11531 enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector }; 11532 /// Attribute set of the parameter. 11533 struct ParamAttrTy { 11534 ParamKindTy Kind = Vector; 11535 llvm::APSInt StrideOrArg; 11536 llvm::APSInt Alignment; 11537 }; 11538 } // namespace 11539 11540 static unsigned evaluateCDTSize(const FunctionDecl *FD, 11541 ArrayRef<ParamAttrTy> ParamAttrs) { 11542 // Every vector variant of a SIMD-enabled function has a vector length (VLEN). 11543 // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument 11544 // of that clause. The VLEN value must be power of 2. 11545 // In other case the notion of the function`s "characteristic data type" (CDT) 11546 // is used to compute the vector length. 11547 // CDT is defined in the following order: 11548 // a) For non-void function, the CDT is the return type. 11549 // b) If the function has any non-uniform, non-linear parameters, then the 11550 // CDT is the type of the first such parameter. 11551 // c) If the CDT determined by a) or b) above is struct, union, or class 11552 // type which is pass-by-value (except for the type that maps to the 11553 // built-in complex data type), the characteristic data type is int. 11554 // d) If none of the above three cases is applicable, the CDT is int. 11555 // The VLEN is then determined based on the CDT and the size of vector 11556 // register of that ISA for which current vector version is generated. The 11557 // VLEN is computed using the formula below: 11558 // VLEN = sizeof(vector_register) / sizeof(CDT), 11559 // where vector register size specified in section 3.2.1 Registers and the 11560 // Stack Frame of original AMD64 ABI document. 11561 QualType RetType = FD->getReturnType(); 11562 if (RetType.isNull()) 11563 return 0; 11564 ASTContext &C = FD->getASTContext(); 11565 QualType CDT; 11566 if (!RetType.isNull() && !RetType->isVoidType()) { 11567 CDT = RetType; 11568 } else { 11569 unsigned Offset = 0; 11570 if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) { 11571 if (ParamAttrs[Offset].Kind == Vector) 11572 CDT = C.getPointerType(C.getRecordType(MD->getParent())); 11573 ++Offset; 11574 } 11575 if (CDT.isNull()) { 11576 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 11577 if (ParamAttrs[I + Offset].Kind == Vector) { 11578 CDT = FD->getParamDecl(I)->getType(); 11579 break; 11580 } 11581 } 11582 } 11583 } 11584 if (CDT.isNull()) 11585 CDT = C.IntTy; 11586 CDT = CDT->getCanonicalTypeUnqualified(); 11587 if (CDT->isRecordType() || CDT->isUnionType()) 11588 CDT = C.IntTy; 11589 return C.getTypeSize(CDT); 11590 } 11591 11592 static void 11593 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn, 11594 const llvm::APSInt &VLENVal, 11595 ArrayRef<ParamAttrTy> ParamAttrs, 11596 OMPDeclareSimdDeclAttr::BranchStateTy State) { 11597 struct ISADataTy { 11598 char ISA; 11599 unsigned VecRegSize; 11600 }; 11601 ISADataTy ISAData[] = { 11602 { 11603 'b', 128 11604 }, // SSE 11605 { 11606 'c', 256 11607 }, // AVX 11608 { 11609 'd', 256 11610 }, // AVX2 11611 { 11612 'e', 512 11613 }, // AVX512 11614 }; 11615 llvm::SmallVector<char, 2> Masked; 11616 switch (State) { 11617 case OMPDeclareSimdDeclAttr::BS_Undefined: 11618 Masked.push_back('N'); 11619 Masked.push_back('M'); 11620 break; 11621 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 11622 Masked.push_back('N'); 11623 break; 11624 case OMPDeclareSimdDeclAttr::BS_Inbranch: 11625 Masked.push_back('M'); 11626 break; 11627 } 11628 for (char Mask : Masked) { 11629 for (const ISADataTy &Data : ISAData) { 11630 SmallString<256> Buffer; 11631 llvm::raw_svector_ostream Out(Buffer); 11632 Out << "_ZGV" << Data.ISA << Mask; 11633 if (!VLENVal) { 11634 unsigned NumElts = evaluateCDTSize(FD, ParamAttrs); 11635 assert(NumElts && "Non-zero simdlen/cdtsize expected"); 11636 Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts); 11637 } else { 11638 Out << VLENVal; 11639 } 11640 for (const ParamAttrTy &ParamAttr : ParamAttrs) { 11641 switch (ParamAttr.Kind){ 11642 case LinearWithVarStride: 11643 Out << 's' << ParamAttr.StrideOrArg; 11644 break; 11645 case Linear: 11646 Out << 'l'; 11647 if (ParamAttr.StrideOrArg != 1) 11648 Out << ParamAttr.StrideOrArg; 11649 break; 11650 case Uniform: 11651 Out << 'u'; 11652 break; 11653 case Vector: 11654 Out << 'v'; 11655 break; 11656 } 11657 if (!!ParamAttr.Alignment) 11658 Out << 'a' << ParamAttr.Alignment; 11659 } 11660 Out << '_' << Fn->getName(); 11661 Fn->addFnAttr(Out.str()); 11662 } 11663 } 11664 } 11665 11666 // This are the Functions that are needed to mangle the name of the 11667 // vector functions generated by the compiler, according to the rules 11668 // defined in the "Vector Function ABI specifications for AArch64", 11669 // available at 11670 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi. 11671 11672 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI. 11673 /// 11674 /// TODO: Need to implement the behavior for reference marked with a 11675 /// var or no linear modifiers (1.b in the section). For this, we 11676 /// need to extend ParamKindTy to support the linear modifiers. 11677 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) { 11678 QT = QT.getCanonicalType(); 11679 11680 if (QT->isVoidType()) 11681 return false; 11682 11683 if (Kind == ParamKindTy::Uniform) 11684 return false; 11685 11686 if (Kind == ParamKindTy::Linear) 11687 return false; 11688 11689 // TODO: Handle linear references with modifiers 11690 11691 if (Kind == ParamKindTy::LinearWithVarStride) 11692 return false; 11693 11694 return true; 11695 } 11696 11697 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI. 11698 static bool getAArch64PBV(QualType QT, ASTContext &C) { 11699 QT = QT.getCanonicalType(); 11700 unsigned Size = C.getTypeSize(QT); 11701 11702 // Only scalars and complex within 16 bytes wide set PVB to true. 11703 if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128) 11704 return false; 11705 11706 if (QT->isFloatingType()) 11707 return true; 11708 11709 if (QT->isIntegerType()) 11710 return true; 11711 11712 if (QT->isPointerType()) 11713 return true; 11714 11715 // TODO: Add support for complex types (section 3.1.2, item 2). 11716 11717 return false; 11718 } 11719 11720 /// Computes the lane size (LS) of a return type or of an input parameter, 11721 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI. 11722 /// TODO: Add support for references, section 3.2.1, item 1. 11723 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) { 11724 if (!getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) { 11725 QualType PTy = QT.getCanonicalType()->getPointeeType(); 11726 if (getAArch64PBV(PTy, C)) 11727 return C.getTypeSize(PTy); 11728 } 11729 if (getAArch64PBV(QT, C)) 11730 return C.getTypeSize(QT); 11731 11732 return C.getTypeSize(C.getUIntPtrType()); 11733 } 11734 11735 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the 11736 // signature of the scalar function, as defined in 3.2.2 of the 11737 // AAVFABI. 11738 static std::tuple<unsigned, unsigned, bool> 11739 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) { 11740 QualType RetType = FD->getReturnType().getCanonicalType(); 11741 11742 ASTContext &C = FD->getASTContext(); 11743 11744 bool OutputBecomesInput = false; 11745 11746 llvm::SmallVector<unsigned, 8> Sizes; 11747 if (!RetType->isVoidType()) { 11748 Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C)); 11749 if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {})) 11750 OutputBecomesInput = true; 11751 } 11752 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 11753 QualType QT = FD->getParamDecl(I)->getType().getCanonicalType(); 11754 Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C)); 11755 } 11756 11757 assert(!Sizes.empty() && "Unable to determine NDS and WDS."); 11758 // The LS of a function parameter / return value can only be a power 11759 // of 2, starting from 8 bits, up to 128. 11760 assert(llvm::all_of(Sizes, 11761 [](unsigned Size) { 11762 return Size == 8 || Size == 16 || Size == 32 || 11763 Size == 64 || Size == 128; 11764 }) && 11765 "Invalid size"); 11766 11767 return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)), 11768 *std::max_element(std::begin(Sizes), std::end(Sizes)), 11769 OutputBecomesInput); 11770 } 11771 11772 /// Mangle the parameter part of the vector function name according to 11773 /// their OpenMP classification. The mangling function is defined in 11774 /// section 3.5 of the AAVFABI. 11775 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) { 11776 SmallString<256> Buffer; 11777 llvm::raw_svector_ostream Out(Buffer); 11778 for (const auto &ParamAttr : ParamAttrs) { 11779 switch (ParamAttr.Kind) { 11780 case LinearWithVarStride: 11781 Out << "ls" << ParamAttr.StrideOrArg; 11782 break; 11783 case Linear: 11784 Out << 'l'; 11785 // Don't print the step value if it is not present or if it is 11786 // equal to 1. 11787 if (ParamAttr.StrideOrArg != 1) 11788 Out << ParamAttr.StrideOrArg; 11789 break; 11790 case Uniform: 11791 Out << 'u'; 11792 break; 11793 case Vector: 11794 Out << 'v'; 11795 break; 11796 } 11797 11798 if (!!ParamAttr.Alignment) 11799 Out << 'a' << ParamAttr.Alignment; 11800 } 11801 11802 return std::string(Out.str()); 11803 } 11804 11805 // Function used to add the attribute. The parameter `VLEN` is 11806 // templated to allow the use of "x" when targeting scalable functions 11807 // for SVE. 11808 template <typename T> 11809 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix, 11810 char ISA, StringRef ParSeq, 11811 StringRef MangledName, bool OutputBecomesInput, 11812 llvm::Function *Fn) { 11813 SmallString<256> Buffer; 11814 llvm::raw_svector_ostream Out(Buffer); 11815 Out << Prefix << ISA << LMask << VLEN; 11816 if (OutputBecomesInput) 11817 Out << "v"; 11818 Out << ParSeq << "_" << MangledName; 11819 Fn->addFnAttr(Out.str()); 11820 } 11821 11822 // Helper function to generate the Advanced SIMD names depending on 11823 // the value of the NDS when simdlen is not present. 11824 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask, 11825 StringRef Prefix, char ISA, 11826 StringRef ParSeq, StringRef MangledName, 11827 bool OutputBecomesInput, 11828 llvm::Function *Fn) { 11829 switch (NDS) { 11830 case 8: 11831 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 11832 OutputBecomesInput, Fn); 11833 addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName, 11834 OutputBecomesInput, Fn); 11835 break; 11836 case 16: 11837 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 11838 OutputBecomesInput, Fn); 11839 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 11840 OutputBecomesInput, Fn); 11841 break; 11842 case 32: 11843 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 11844 OutputBecomesInput, Fn); 11845 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 11846 OutputBecomesInput, Fn); 11847 break; 11848 case 64: 11849 case 128: 11850 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 11851 OutputBecomesInput, Fn); 11852 break; 11853 default: 11854 llvm_unreachable("Scalar type is too wide."); 11855 } 11856 } 11857 11858 /// Emit vector function attributes for AArch64, as defined in the AAVFABI. 11859 static void emitAArch64DeclareSimdFunction( 11860 CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN, 11861 ArrayRef<ParamAttrTy> ParamAttrs, 11862 OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName, 11863 char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) { 11864 11865 // Get basic data for building the vector signature. 11866 const auto Data = getNDSWDS(FD, ParamAttrs); 11867 const unsigned NDS = std::get<0>(Data); 11868 const unsigned WDS = std::get<1>(Data); 11869 const bool OutputBecomesInput = std::get<2>(Data); 11870 11871 // Check the values provided via `simdlen` by the user. 11872 // 1. A `simdlen(1)` doesn't produce vector signatures, 11873 if (UserVLEN == 1) { 11874 unsigned DiagID = CGM.getDiags().getCustomDiagID( 11875 DiagnosticsEngine::Warning, 11876 "The clause simdlen(1) has no effect when targeting aarch64."); 11877 CGM.getDiags().Report(SLoc, DiagID); 11878 return; 11879 } 11880 11881 // 2. Section 3.3.1, item 1: user input must be a power of 2 for 11882 // Advanced SIMD output. 11883 if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) { 11884 unsigned DiagID = CGM.getDiags().getCustomDiagID( 11885 DiagnosticsEngine::Warning, "The value specified in simdlen must be a " 11886 "power of 2 when targeting Advanced SIMD."); 11887 CGM.getDiags().Report(SLoc, DiagID); 11888 return; 11889 } 11890 11891 // 3. Section 3.4.1. SVE fixed lengh must obey the architectural 11892 // limits. 11893 if (ISA == 's' && UserVLEN != 0) { 11894 if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) { 11895 unsigned DiagID = CGM.getDiags().getCustomDiagID( 11896 DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit " 11897 "lanes in the architectural constraints " 11898 "for SVE (min is 128-bit, max is " 11899 "2048-bit, by steps of 128-bit)"); 11900 CGM.getDiags().Report(SLoc, DiagID) << WDS; 11901 return; 11902 } 11903 } 11904 11905 // Sort out parameter sequence. 11906 const std::string ParSeq = mangleVectorParameters(ParamAttrs); 11907 StringRef Prefix = "_ZGV"; 11908 // Generate simdlen from user input (if any). 11909 if (UserVLEN) { 11910 if (ISA == 's') { 11911 // SVE generates only a masked function. 11912 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 11913 OutputBecomesInput, Fn); 11914 } else { 11915 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 11916 // Advanced SIMD generates one or two functions, depending on 11917 // the `[not]inbranch` clause. 11918 switch (State) { 11919 case OMPDeclareSimdDeclAttr::BS_Undefined: 11920 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 11921 OutputBecomesInput, Fn); 11922 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 11923 OutputBecomesInput, Fn); 11924 break; 11925 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 11926 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 11927 OutputBecomesInput, Fn); 11928 break; 11929 case OMPDeclareSimdDeclAttr::BS_Inbranch: 11930 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 11931 OutputBecomesInput, Fn); 11932 break; 11933 } 11934 } 11935 } else { 11936 // If no user simdlen is provided, follow the AAVFABI rules for 11937 // generating the vector length. 11938 if (ISA == 's') { 11939 // SVE, section 3.4.1, item 1. 11940 addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName, 11941 OutputBecomesInput, Fn); 11942 } else { 11943 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 11944 // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or 11945 // two vector names depending on the use of the clause 11946 // `[not]inbranch`. 11947 switch (State) { 11948 case OMPDeclareSimdDeclAttr::BS_Undefined: 11949 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 11950 OutputBecomesInput, Fn); 11951 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 11952 OutputBecomesInput, Fn); 11953 break; 11954 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 11955 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 11956 OutputBecomesInput, Fn); 11957 break; 11958 case OMPDeclareSimdDeclAttr::BS_Inbranch: 11959 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 11960 OutputBecomesInput, Fn); 11961 break; 11962 } 11963 } 11964 } 11965 } 11966 11967 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD, 11968 llvm::Function *Fn) { 11969 ASTContext &C = CGM.getContext(); 11970 FD = FD->getMostRecentDecl(); 11971 // Map params to their positions in function decl. 11972 llvm::DenseMap<const Decl *, unsigned> ParamPositions; 11973 if (isa<CXXMethodDecl>(FD)) 11974 ParamPositions.try_emplace(FD, 0); 11975 unsigned ParamPos = ParamPositions.size(); 11976 for (const ParmVarDecl *P : FD->parameters()) { 11977 ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos); 11978 ++ParamPos; 11979 } 11980 while (FD) { 11981 for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) { 11982 llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size()); 11983 // Mark uniform parameters. 11984 for (const Expr *E : Attr->uniforms()) { 11985 E = E->IgnoreParenImpCasts(); 11986 unsigned Pos; 11987 if (isa<CXXThisExpr>(E)) { 11988 Pos = ParamPositions[FD]; 11989 } else { 11990 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 11991 ->getCanonicalDecl(); 11992 Pos = ParamPositions[PVD]; 11993 } 11994 ParamAttrs[Pos].Kind = Uniform; 11995 } 11996 // Get alignment info. 11997 auto *NI = Attr->alignments_begin(); 11998 for (const Expr *E : Attr->aligneds()) { 11999 E = E->IgnoreParenImpCasts(); 12000 unsigned Pos; 12001 QualType ParmTy; 12002 if (isa<CXXThisExpr>(E)) { 12003 Pos = ParamPositions[FD]; 12004 ParmTy = E->getType(); 12005 } else { 12006 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 12007 ->getCanonicalDecl(); 12008 Pos = ParamPositions[PVD]; 12009 ParmTy = PVD->getType(); 12010 } 12011 ParamAttrs[Pos].Alignment = 12012 (*NI) 12013 ? (*NI)->EvaluateKnownConstInt(C) 12014 : llvm::APSInt::getUnsigned( 12015 C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy)) 12016 .getQuantity()); 12017 ++NI; 12018 } 12019 // Mark linear parameters. 12020 auto *SI = Attr->steps_begin(); 12021 auto *MI = Attr->modifiers_begin(); 12022 for (const Expr *E : Attr->linears()) { 12023 E = E->IgnoreParenImpCasts(); 12024 unsigned Pos; 12025 // Rescaling factor needed to compute the linear parameter 12026 // value in the mangled name. 12027 unsigned PtrRescalingFactor = 1; 12028 if (isa<CXXThisExpr>(E)) { 12029 Pos = ParamPositions[FD]; 12030 } else { 12031 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 12032 ->getCanonicalDecl(); 12033 Pos = ParamPositions[PVD]; 12034 if (auto *P = dyn_cast<PointerType>(PVD->getType())) 12035 PtrRescalingFactor = CGM.getContext() 12036 .getTypeSizeInChars(P->getPointeeType()) 12037 .getQuantity(); 12038 } 12039 ParamAttrTy &ParamAttr = ParamAttrs[Pos]; 12040 ParamAttr.Kind = Linear; 12041 // Assuming a stride of 1, for `linear` without modifiers. 12042 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned(1); 12043 if (*SI) { 12044 Expr::EvalResult Result; 12045 if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) { 12046 if (const auto *DRE = 12047 cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) { 12048 if (const auto *StridePVD = 12049 dyn_cast<ParmVarDecl>(DRE->getDecl())) { 12050 ParamAttr.Kind = LinearWithVarStride; 12051 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned( 12052 ParamPositions[StridePVD->getCanonicalDecl()]); 12053 } 12054 } 12055 } else { 12056 ParamAttr.StrideOrArg = Result.Val.getInt(); 12057 } 12058 } 12059 // If we are using a linear clause on a pointer, we need to 12060 // rescale the value of linear_step with the byte size of the 12061 // pointee type. 12062 if (Linear == ParamAttr.Kind) 12063 ParamAttr.StrideOrArg = ParamAttr.StrideOrArg * PtrRescalingFactor; 12064 ++SI; 12065 ++MI; 12066 } 12067 llvm::APSInt VLENVal; 12068 SourceLocation ExprLoc; 12069 const Expr *VLENExpr = Attr->getSimdlen(); 12070 if (VLENExpr) { 12071 VLENVal = VLENExpr->EvaluateKnownConstInt(C); 12072 ExprLoc = VLENExpr->getExprLoc(); 12073 } 12074 OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState(); 12075 if (CGM.getTriple().isX86()) { 12076 emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State); 12077 } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) { 12078 unsigned VLEN = VLENVal.getExtValue(); 12079 StringRef MangledName = Fn->getName(); 12080 if (CGM.getTarget().hasFeature("sve")) 12081 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 12082 MangledName, 's', 128, Fn, ExprLoc); 12083 if (CGM.getTarget().hasFeature("neon")) 12084 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 12085 MangledName, 'n', 128, Fn, ExprLoc); 12086 } 12087 } 12088 FD = FD->getPreviousDecl(); 12089 } 12090 } 12091 12092 namespace { 12093 /// Cleanup action for doacross support. 12094 class DoacrossCleanupTy final : public EHScopeStack::Cleanup { 12095 public: 12096 static const int DoacrossFinArgs = 2; 12097 12098 private: 12099 llvm::FunctionCallee RTLFn; 12100 llvm::Value *Args[DoacrossFinArgs]; 12101 12102 public: 12103 DoacrossCleanupTy(llvm::FunctionCallee RTLFn, 12104 ArrayRef<llvm::Value *> CallArgs) 12105 : RTLFn(RTLFn) { 12106 assert(CallArgs.size() == DoacrossFinArgs); 12107 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 12108 } 12109 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 12110 if (!CGF.HaveInsertPoint()) 12111 return; 12112 CGF.EmitRuntimeCall(RTLFn, Args); 12113 } 12114 }; 12115 } // namespace 12116 12117 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF, 12118 const OMPLoopDirective &D, 12119 ArrayRef<Expr *> NumIterations) { 12120 if (!CGF.HaveInsertPoint()) 12121 return; 12122 12123 ASTContext &C = CGM.getContext(); 12124 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 12125 RecordDecl *RD; 12126 if (KmpDimTy.isNull()) { 12127 // Build struct kmp_dim { // loop bounds info casted to kmp_int64 12128 // kmp_int64 lo; // lower 12129 // kmp_int64 up; // upper 12130 // kmp_int64 st; // stride 12131 // }; 12132 RD = C.buildImplicitRecord("kmp_dim"); 12133 RD->startDefinition(); 12134 addFieldToRecordDecl(C, RD, Int64Ty); 12135 addFieldToRecordDecl(C, RD, Int64Ty); 12136 addFieldToRecordDecl(C, RD, Int64Ty); 12137 RD->completeDefinition(); 12138 KmpDimTy = C.getRecordType(RD); 12139 } else { 12140 RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl()); 12141 } 12142 llvm::APInt Size(/*numBits=*/32, NumIterations.size()); 12143 QualType ArrayTy = 12144 C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0); 12145 12146 Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims"); 12147 CGF.EmitNullInitialization(DimsAddr, ArrayTy); 12148 enum { LowerFD = 0, UpperFD, StrideFD }; 12149 // Fill dims with data. 12150 for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) { 12151 LValue DimsLVal = CGF.MakeAddrLValue( 12152 CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy); 12153 // dims.upper = num_iterations; 12154 LValue UpperLVal = CGF.EmitLValueForField( 12155 DimsLVal, *std::next(RD->field_begin(), UpperFD)); 12156 llvm::Value *NumIterVal = CGF.EmitScalarConversion( 12157 CGF.EmitScalarExpr(NumIterations[I]), NumIterations[I]->getType(), 12158 Int64Ty, NumIterations[I]->getExprLoc()); 12159 CGF.EmitStoreOfScalar(NumIterVal, UpperLVal); 12160 // dims.stride = 1; 12161 LValue StrideLVal = CGF.EmitLValueForField( 12162 DimsLVal, *std::next(RD->field_begin(), StrideFD)); 12163 CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1), 12164 StrideLVal); 12165 } 12166 12167 // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, 12168 // kmp_int32 num_dims, struct kmp_dim * dims); 12169 llvm::Value *Args[] = { 12170 emitUpdateLocation(CGF, D.getBeginLoc()), 12171 getThreadID(CGF, D.getBeginLoc()), 12172 llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()), 12173 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 12174 CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(), 12175 CGM.VoidPtrTy)}; 12176 12177 llvm::FunctionCallee RTLFn = OMPBuilder.getOrCreateRuntimeFunction( 12178 CGM.getModule(), OMPRTL___kmpc_doacross_init); 12179 CGF.EmitRuntimeCall(RTLFn, Args); 12180 llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = { 12181 emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())}; 12182 llvm::FunctionCallee FiniRTLFn = OMPBuilder.getOrCreateRuntimeFunction( 12183 CGM.getModule(), OMPRTL___kmpc_doacross_fini); 12184 CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 12185 llvm::makeArrayRef(FiniArgs)); 12186 } 12187 12188 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 12189 const OMPDependClause *C) { 12190 QualType Int64Ty = 12191 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 12192 llvm::APInt Size(/*numBits=*/32, C->getNumLoops()); 12193 QualType ArrayTy = CGM.getContext().getConstantArrayType( 12194 Int64Ty, Size, nullptr, ArrayType::Normal, 0); 12195 Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr"); 12196 for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) { 12197 const Expr *CounterVal = C->getLoopData(I); 12198 assert(CounterVal); 12199 llvm::Value *CntVal = CGF.EmitScalarConversion( 12200 CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty, 12201 CounterVal->getExprLoc()); 12202 CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I), 12203 /*Volatile=*/false, Int64Ty); 12204 } 12205 llvm::Value *Args[] = { 12206 emitUpdateLocation(CGF, C->getBeginLoc()), 12207 getThreadID(CGF, C->getBeginLoc()), 12208 CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()}; 12209 llvm::FunctionCallee RTLFn; 12210 if (C->getDependencyKind() == OMPC_DEPEND_source) { 12211 RTLFn = OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 12212 OMPRTL___kmpc_doacross_post); 12213 } else { 12214 assert(C->getDependencyKind() == OMPC_DEPEND_sink); 12215 RTLFn = OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), 12216 OMPRTL___kmpc_doacross_wait); 12217 } 12218 CGF.EmitRuntimeCall(RTLFn, Args); 12219 } 12220 12221 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc, 12222 llvm::FunctionCallee Callee, 12223 ArrayRef<llvm::Value *> Args) const { 12224 assert(Loc.isValid() && "Outlined function call location must be valid."); 12225 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 12226 12227 if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) { 12228 if (Fn->doesNotThrow()) { 12229 CGF.EmitNounwindRuntimeCall(Fn, Args); 12230 return; 12231 } 12232 } 12233 CGF.EmitRuntimeCall(Callee, Args); 12234 } 12235 12236 void CGOpenMPRuntime::emitOutlinedFunctionCall( 12237 CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn, 12238 ArrayRef<llvm::Value *> Args) const { 12239 emitCall(CGF, Loc, OutlinedFn, Args); 12240 } 12241 12242 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) { 12243 if (const auto *FD = dyn_cast<FunctionDecl>(D)) 12244 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD)) 12245 HasEmittedDeclareTargetRegion = true; 12246 } 12247 12248 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF, 12249 const VarDecl *NativeParam, 12250 const VarDecl *TargetParam) const { 12251 return CGF.GetAddrOfLocalVar(NativeParam); 12252 } 12253 12254 /// Return allocator value from expression, or return a null allocator (default 12255 /// when no allocator specified). 12256 static llvm::Value *getAllocatorVal(CodeGenFunction &CGF, 12257 const Expr *Allocator) { 12258 llvm::Value *AllocVal; 12259 if (Allocator) { 12260 AllocVal = CGF.EmitScalarExpr(Allocator); 12261 // According to the standard, the original allocator type is a enum 12262 // (integer). Convert to pointer type, if required. 12263 AllocVal = CGF.EmitScalarConversion(AllocVal, Allocator->getType(), 12264 CGF.getContext().VoidPtrTy, 12265 Allocator->getExprLoc()); 12266 } else { 12267 // If no allocator specified, it defaults to the null allocator. 12268 AllocVal = llvm::Constant::getNullValue( 12269 CGF.CGM.getTypes().ConvertType(CGF.getContext().VoidPtrTy)); 12270 } 12271 return AllocVal; 12272 } 12273 12274 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF, 12275 const VarDecl *VD) { 12276 if (!VD) 12277 return Address::invalid(); 12278 Address UntiedAddr = Address::invalid(); 12279 Address UntiedRealAddr = Address::invalid(); 12280 auto It = FunctionToUntiedTaskStackMap.find(CGF.CurFn); 12281 if (It != FunctionToUntiedTaskStackMap.end()) { 12282 const UntiedLocalVarsAddressesMap &UntiedData = 12283 UntiedLocalVarsStack[It->second]; 12284 auto I = UntiedData.find(VD); 12285 if (I != UntiedData.end()) { 12286 UntiedAddr = I->second.first; 12287 UntiedRealAddr = I->second.second; 12288 } 12289 } 12290 const VarDecl *CVD = VD->getCanonicalDecl(); 12291 if (CVD->hasAttr<OMPAllocateDeclAttr>()) { 12292 // Use the default allocation. 12293 if (!isAllocatableDecl(VD)) 12294 return UntiedAddr; 12295 llvm::Value *Size; 12296 CharUnits Align = CGM.getContext().getDeclAlign(CVD); 12297 if (CVD->getType()->isVariablyModifiedType()) { 12298 Size = CGF.getTypeSize(CVD->getType()); 12299 // Align the size: ((size + align - 1) / align) * align 12300 Size = CGF.Builder.CreateNUWAdd( 12301 Size, CGM.getSize(Align - CharUnits::fromQuantity(1))); 12302 Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align)); 12303 Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align)); 12304 } else { 12305 CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType()); 12306 Size = CGM.getSize(Sz.alignTo(Align)); 12307 } 12308 llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc()); 12309 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>(); 12310 const Expr *Allocator = AA->getAllocator(); 12311 llvm::Value *AllocVal = getAllocatorVal(CGF, Allocator); 12312 llvm::Value *Alignment = 12313 AA->getAlignment() 12314 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(AA->getAlignment()), 12315 CGM.SizeTy, /*isSigned=*/false) 12316 : nullptr; 12317 SmallVector<llvm::Value *, 4> Args; 12318 Args.push_back(ThreadID); 12319 if (Alignment) 12320 Args.push_back(Alignment); 12321 Args.push_back(Size); 12322 Args.push_back(AllocVal); 12323 llvm::omp::RuntimeFunction FnID = 12324 Alignment ? OMPRTL___kmpc_aligned_alloc : OMPRTL___kmpc_alloc; 12325 llvm::Value *Addr = CGF.EmitRuntimeCall( 12326 OMPBuilder.getOrCreateRuntimeFunction(CGM.getModule(), FnID), Args, 12327 getName({CVD->getName(), ".void.addr"})); 12328 llvm::FunctionCallee FiniRTLFn = OMPBuilder.getOrCreateRuntimeFunction( 12329 CGM.getModule(), OMPRTL___kmpc_free); 12330 QualType Ty = CGM.getContext().getPointerType(CVD->getType()); 12331 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 12332 Addr, CGF.ConvertTypeForMem(Ty), getName({CVD->getName(), ".addr"})); 12333 if (UntiedAddr.isValid()) 12334 CGF.EmitStoreOfScalar(Addr, UntiedAddr, /*Volatile=*/false, Ty); 12335 12336 // Cleanup action for allocate support. 12337 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup { 12338 llvm::FunctionCallee RTLFn; 12339 SourceLocation::UIntTy LocEncoding; 12340 Address Addr; 12341 const Expr *AllocExpr; 12342 12343 public: 12344 OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn, 12345 SourceLocation::UIntTy LocEncoding, Address Addr, 12346 const Expr *AllocExpr) 12347 : RTLFn(RTLFn), LocEncoding(LocEncoding), Addr(Addr), 12348 AllocExpr(AllocExpr) {} 12349 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 12350 if (!CGF.HaveInsertPoint()) 12351 return; 12352 llvm::Value *Args[3]; 12353 Args[0] = CGF.CGM.getOpenMPRuntime().getThreadID( 12354 CGF, SourceLocation::getFromRawEncoding(LocEncoding)); 12355 Args[1] = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 12356 Addr.getPointer(), CGF.VoidPtrTy); 12357 llvm::Value *AllocVal = getAllocatorVal(CGF, AllocExpr); 12358 Args[2] = AllocVal; 12359 CGF.EmitRuntimeCall(RTLFn, Args); 12360 } 12361 }; 12362 Address VDAddr = 12363 UntiedRealAddr.isValid() 12364 ? UntiedRealAddr 12365 : Address(Addr, CGF.ConvertTypeForMem(CVD->getType()), Align); 12366 CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>( 12367 NormalAndEHCleanup, FiniRTLFn, CVD->getLocation().getRawEncoding(), 12368 VDAddr, Allocator); 12369 if (UntiedRealAddr.isValid()) 12370 if (auto *Region = 12371 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 12372 Region->emitUntiedSwitch(CGF); 12373 return VDAddr; 12374 } 12375 return UntiedAddr; 12376 } 12377 12378 bool CGOpenMPRuntime::isLocalVarInUntiedTask(CodeGenFunction &CGF, 12379 const VarDecl *VD) const { 12380 auto It = FunctionToUntiedTaskStackMap.find(CGF.CurFn); 12381 if (It == FunctionToUntiedTaskStackMap.end()) 12382 return false; 12383 return UntiedLocalVarsStack[It->second].count(VD) > 0; 12384 } 12385 12386 CGOpenMPRuntime::NontemporalDeclsRAII::NontemporalDeclsRAII( 12387 CodeGenModule &CGM, const OMPLoopDirective &S) 12388 : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) { 12389 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 12390 if (!NeedToPush) 12391 return; 12392 NontemporalDeclsSet &DS = 12393 CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back(); 12394 for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) { 12395 for (const Stmt *Ref : C->private_refs()) { 12396 const auto *SimpleRefExpr = cast<Expr>(Ref)->IgnoreParenImpCasts(); 12397 const ValueDecl *VD; 12398 if (const auto *DRE = dyn_cast<DeclRefExpr>(SimpleRefExpr)) { 12399 VD = DRE->getDecl(); 12400 } else { 12401 const auto *ME = cast<MemberExpr>(SimpleRefExpr); 12402 assert((ME->isImplicitCXXThis() || 12403 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) && 12404 "Expected member of current class."); 12405 VD = ME->getMemberDecl(); 12406 } 12407 DS.insert(VD); 12408 } 12409 } 12410 } 12411 12412 CGOpenMPRuntime::NontemporalDeclsRAII::~NontemporalDeclsRAII() { 12413 if (!NeedToPush) 12414 return; 12415 CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back(); 12416 } 12417 12418 CGOpenMPRuntime::UntiedTaskLocalDeclsRAII::UntiedTaskLocalDeclsRAII( 12419 CodeGenFunction &CGF, 12420 const llvm::MapVector<CanonicalDeclPtr<const VarDecl>, 12421 std::pair<Address, Address>> &LocalVars) 12422 : CGM(CGF.CGM), NeedToPush(!LocalVars.empty()) { 12423 if (!NeedToPush) 12424 return; 12425 CGM.getOpenMPRuntime().FunctionToUntiedTaskStackMap.try_emplace( 12426 CGF.CurFn, CGM.getOpenMPRuntime().UntiedLocalVarsStack.size()); 12427 CGM.getOpenMPRuntime().UntiedLocalVarsStack.push_back(LocalVars); 12428 } 12429 12430 CGOpenMPRuntime::UntiedTaskLocalDeclsRAII::~UntiedTaskLocalDeclsRAII() { 12431 if (!NeedToPush) 12432 return; 12433 CGM.getOpenMPRuntime().UntiedLocalVarsStack.pop_back(); 12434 } 12435 12436 bool CGOpenMPRuntime::isNontemporalDecl(const ValueDecl *VD) const { 12437 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 12438 12439 return llvm::any_of( 12440 CGM.getOpenMPRuntime().NontemporalDeclsStack, 12441 [VD](const NontemporalDeclsSet &Set) { return Set.contains(VD); }); 12442 } 12443 12444 void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis( 12445 const OMPExecutableDirective &S, 12446 llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled) 12447 const { 12448 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs; 12449 // Vars in target/task regions must be excluded completely. 12450 if (isOpenMPTargetExecutionDirective(S.getDirectiveKind()) || 12451 isOpenMPTaskingDirective(S.getDirectiveKind())) { 12452 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions; 12453 getOpenMPCaptureRegions(CaptureRegions, S.getDirectiveKind()); 12454 const CapturedStmt *CS = S.getCapturedStmt(CaptureRegions.front()); 12455 for (const CapturedStmt::Capture &Cap : CS->captures()) { 12456 if (Cap.capturesVariable() || Cap.capturesVariableByCopy()) 12457 NeedToCheckForLPCs.insert(Cap.getCapturedVar()); 12458 } 12459 } 12460 // Exclude vars in private clauses. 12461 for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) { 12462 for (const Expr *Ref : C->varlists()) { 12463 if (!Ref->getType()->isScalarType()) 12464 continue; 12465 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 12466 if (!DRE) 12467 continue; 12468 NeedToCheckForLPCs.insert(DRE->getDecl()); 12469 } 12470 } 12471 for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) { 12472 for (const Expr *Ref : C->varlists()) { 12473 if (!Ref->getType()->isScalarType()) 12474 continue; 12475 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 12476 if (!DRE) 12477 continue; 12478 NeedToCheckForLPCs.insert(DRE->getDecl()); 12479 } 12480 } 12481 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) { 12482 for (const Expr *Ref : C->varlists()) { 12483 if (!Ref->getType()->isScalarType()) 12484 continue; 12485 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 12486 if (!DRE) 12487 continue; 12488 NeedToCheckForLPCs.insert(DRE->getDecl()); 12489 } 12490 } 12491 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) { 12492 for (const Expr *Ref : C->varlists()) { 12493 if (!Ref->getType()->isScalarType()) 12494 continue; 12495 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 12496 if (!DRE) 12497 continue; 12498 NeedToCheckForLPCs.insert(DRE->getDecl()); 12499 } 12500 } 12501 for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) { 12502 for (const Expr *Ref : C->varlists()) { 12503 if (!Ref->getType()->isScalarType()) 12504 continue; 12505 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 12506 if (!DRE) 12507 continue; 12508 NeedToCheckForLPCs.insert(DRE->getDecl()); 12509 } 12510 } 12511 for (const Decl *VD : NeedToCheckForLPCs) { 12512 for (const LastprivateConditionalData &Data : 12513 llvm::reverse(CGM.getOpenMPRuntime().LastprivateConditionalStack)) { 12514 if (Data.DeclToUniqueName.count(VD) > 0) { 12515 if (!Data.Disabled) 12516 NeedToAddForLPCsAsDisabled.insert(VD); 12517 break; 12518 } 12519 } 12520 } 12521 } 12522 12523 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII( 12524 CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal) 12525 : CGM(CGF.CGM), 12526 Action((CGM.getLangOpts().OpenMP >= 50 && 12527 llvm::any_of(S.getClausesOfKind<OMPLastprivateClause>(), 12528 [](const OMPLastprivateClause *C) { 12529 return C->getKind() == 12530 OMPC_LASTPRIVATE_conditional; 12531 })) 12532 ? ActionToDo::PushAsLastprivateConditional 12533 : ActionToDo::DoNotPush) { 12534 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 12535 if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush) 12536 return; 12537 assert(Action == ActionToDo::PushAsLastprivateConditional && 12538 "Expected a push action."); 12539 LastprivateConditionalData &Data = 12540 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back(); 12541 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) { 12542 if (C->getKind() != OMPC_LASTPRIVATE_conditional) 12543 continue; 12544 12545 for (const Expr *Ref : C->varlists()) { 12546 Data.DeclToUniqueName.insert(std::make_pair( 12547 cast<DeclRefExpr>(Ref->IgnoreParenImpCasts())->getDecl(), 12548 SmallString<16>(generateUniqueName(CGM, "pl_cond", Ref)))); 12549 } 12550 } 12551 Data.IVLVal = IVLVal; 12552 Data.Fn = CGF.CurFn; 12553 } 12554 12555 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII( 12556 CodeGenFunction &CGF, const OMPExecutableDirective &S) 12557 : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) { 12558 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 12559 if (CGM.getLangOpts().OpenMP < 50) 12560 return; 12561 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled; 12562 tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled); 12563 if (!NeedToAddForLPCsAsDisabled.empty()) { 12564 Action = ActionToDo::DisableLastprivateConditional; 12565 LastprivateConditionalData &Data = 12566 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back(); 12567 for (const Decl *VD : NeedToAddForLPCsAsDisabled) 12568 Data.DeclToUniqueName.insert(std::make_pair(VD, SmallString<16>())); 12569 Data.Fn = CGF.CurFn; 12570 Data.Disabled = true; 12571 } 12572 } 12573 12574 CGOpenMPRuntime::LastprivateConditionalRAII 12575 CGOpenMPRuntime::LastprivateConditionalRAII::disable( 12576 CodeGenFunction &CGF, const OMPExecutableDirective &S) { 12577 return LastprivateConditionalRAII(CGF, S); 12578 } 12579 12580 CGOpenMPRuntime::LastprivateConditionalRAII::~LastprivateConditionalRAII() { 12581 if (CGM.getLangOpts().OpenMP < 50) 12582 return; 12583 if (Action == ActionToDo::DisableLastprivateConditional) { 12584 assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled && 12585 "Expected list of disabled private vars."); 12586 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back(); 12587 } 12588 if (Action == ActionToDo::PushAsLastprivateConditional) { 12589 assert( 12590 !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled && 12591 "Expected list of lastprivate conditional vars."); 12592 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back(); 12593 } 12594 } 12595 12596 Address CGOpenMPRuntime::emitLastprivateConditionalInit(CodeGenFunction &CGF, 12597 const VarDecl *VD) { 12598 ASTContext &C = CGM.getContext(); 12599 auto I = LastprivateConditionalToTypes.find(CGF.CurFn); 12600 if (I == LastprivateConditionalToTypes.end()) 12601 I = LastprivateConditionalToTypes.try_emplace(CGF.CurFn).first; 12602 QualType NewType; 12603 const FieldDecl *VDField; 12604 const FieldDecl *FiredField; 12605 LValue BaseLVal; 12606 auto VI = I->getSecond().find(VD); 12607 if (VI == I->getSecond().end()) { 12608 RecordDecl *RD = C.buildImplicitRecord("lasprivate.conditional"); 12609 RD->startDefinition(); 12610 VDField = addFieldToRecordDecl(C, RD, VD->getType().getNonReferenceType()); 12611 FiredField = addFieldToRecordDecl(C, RD, C.CharTy); 12612 RD->completeDefinition(); 12613 NewType = C.getRecordType(RD); 12614 Address Addr = CGF.CreateMemTemp(NewType, C.getDeclAlign(VD), VD->getName()); 12615 BaseLVal = CGF.MakeAddrLValue(Addr, NewType, AlignmentSource::Decl); 12616 I->getSecond().try_emplace(VD, NewType, VDField, FiredField, BaseLVal); 12617 } else { 12618 NewType = std::get<0>(VI->getSecond()); 12619 VDField = std::get<1>(VI->getSecond()); 12620 FiredField = std::get<2>(VI->getSecond()); 12621 BaseLVal = std::get<3>(VI->getSecond()); 12622 } 12623 LValue FiredLVal = 12624 CGF.EmitLValueForField(BaseLVal, FiredField); 12625 CGF.EmitStoreOfScalar( 12626 llvm::ConstantInt::getNullValue(CGF.ConvertTypeForMem(C.CharTy)), 12627 FiredLVal); 12628 return CGF.EmitLValueForField(BaseLVal, VDField).getAddress(CGF); 12629 } 12630 12631 namespace { 12632 /// Checks if the lastprivate conditional variable is referenced in LHS. 12633 class LastprivateConditionalRefChecker final 12634 : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> { 12635 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM; 12636 const Expr *FoundE = nullptr; 12637 const Decl *FoundD = nullptr; 12638 StringRef UniqueDeclName; 12639 LValue IVLVal; 12640 llvm::Function *FoundFn = nullptr; 12641 SourceLocation Loc; 12642 12643 public: 12644 bool VisitDeclRefExpr(const DeclRefExpr *E) { 12645 for (const CGOpenMPRuntime::LastprivateConditionalData &D : 12646 llvm::reverse(LPM)) { 12647 auto It = D.DeclToUniqueName.find(E->getDecl()); 12648 if (It == D.DeclToUniqueName.end()) 12649 continue; 12650 if (D.Disabled) 12651 return false; 12652 FoundE = E; 12653 FoundD = E->getDecl()->getCanonicalDecl(); 12654 UniqueDeclName = It->second; 12655 IVLVal = D.IVLVal; 12656 FoundFn = D.Fn; 12657 break; 12658 } 12659 return FoundE == E; 12660 } 12661 bool VisitMemberExpr(const MemberExpr *E) { 12662 if (!CodeGenFunction::IsWrappedCXXThis(E->getBase())) 12663 return false; 12664 for (const CGOpenMPRuntime::LastprivateConditionalData &D : 12665 llvm::reverse(LPM)) { 12666 auto It = D.DeclToUniqueName.find(E->getMemberDecl()); 12667 if (It == D.DeclToUniqueName.end()) 12668 continue; 12669 if (D.Disabled) 12670 return false; 12671 FoundE = E; 12672 FoundD = E->getMemberDecl()->getCanonicalDecl(); 12673 UniqueDeclName = It->second; 12674 IVLVal = D.IVLVal; 12675 FoundFn = D.Fn; 12676 break; 12677 } 12678 return FoundE == E; 12679 } 12680 bool VisitStmt(const Stmt *S) { 12681 for (const Stmt *Child : S->children()) { 12682 if (!Child) 12683 continue; 12684 if (const auto *E = dyn_cast<Expr>(Child)) 12685 if (!E->isGLValue()) 12686 continue; 12687 if (Visit(Child)) 12688 return true; 12689 } 12690 return false; 12691 } 12692 explicit LastprivateConditionalRefChecker( 12693 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM) 12694 : LPM(LPM) {} 12695 std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *> 12696 getFoundData() const { 12697 return std::make_tuple(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn); 12698 } 12699 }; 12700 } // namespace 12701 12702 void CGOpenMPRuntime::emitLastprivateConditionalUpdate(CodeGenFunction &CGF, 12703 LValue IVLVal, 12704 StringRef UniqueDeclName, 12705 LValue LVal, 12706 SourceLocation Loc) { 12707 // Last updated loop counter for the lastprivate conditional var. 12708 // int<xx> last_iv = 0; 12709 llvm::Type *LLIVTy = CGF.ConvertTypeForMem(IVLVal.getType()); 12710 llvm::Constant *LastIV = 12711 getOrCreateInternalVariable(LLIVTy, getName({UniqueDeclName, "iv"})); 12712 cast<llvm::GlobalVariable>(LastIV)->setAlignment( 12713 IVLVal.getAlignment().getAsAlign()); 12714 LValue LastIVLVal = CGF.MakeNaturalAlignAddrLValue(LastIV, IVLVal.getType()); 12715 12716 // Last value of the lastprivate conditional. 12717 // decltype(priv_a) last_a; 12718 llvm::GlobalVariable *Last = getOrCreateInternalVariable( 12719 CGF.ConvertTypeForMem(LVal.getType()), UniqueDeclName); 12720 Last->setAlignment(LVal.getAlignment().getAsAlign()); 12721 LValue LastLVal = CGF.MakeAddrLValue( 12722 Address(Last, Last->getValueType(), LVal.getAlignment()), LVal.getType()); 12723 12724 // Global loop counter. Required to handle inner parallel-for regions. 12725 // iv 12726 llvm::Value *IVVal = CGF.EmitLoadOfScalar(IVLVal, Loc); 12727 12728 // #pragma omp critical(a) 12729 // if (last_iv <= iv) { 12730 // last_iv = iv; 12731 // last_a = priv_a; 12732 // } 12733 auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal, 12734 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 12735 Action.Enter(CGF); 12736 llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(LastIVLVal, Loc); 12737 // (last_iv <= iv) ? Check if the variable is updated and store new 12738 // value in global var. 12739 llvm::Value *CmpRes; 12740 if (IVLVal.getType()->isSignedIntegerType()) { 12741 CmpRes = CGF.Builder.CreateICmpSLE(LastIVVal, IVVal); 12742 } else { 12743 assert(IVLVal.getType()->isUnsignedIntegerType() && 12744 "Loop iteration variable must be integer."); 12745 CmpRes = CGF.Builder.CreateICmpULE(LastIVVal, IVVal); 12746 } 12747 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lp_cond_then"); 12748 llvm::BasicBlock *ExitBB = CGF.createBasicBlock("lp_cond_exit"); 12749 CGF.Builder.CreateCondBr(CmpRes, ThenBB, ExitBB); 12750 // { 12751 CGF.EmitBlock(ThenBB); 12752 12753 // last_iv = iv; 12754 CGF.EmitStoreOfScalar(IVVal, LastIVLVal); 12755 12756 // last_a = priv_a; 12757 switch (CGF.getEvaluationKind(LVal.getType())) { 12758 case TEK_Scalar: { 12759 llvm::Value *PrivVal = CGF.EmitLoadOfScalar(LVal, Loc); 12760 CGF.EmitStoreOfScalar(PrivVal, LastLVal); 12761 break; 12762 } 12763 case TEK_Complex: { 12764 CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(LVal, Loc); 12765 CGF.EmitStoreOfComplex(PrivVal, LastLVal, /*isInit=*/false); 12766 break; 12767 } 12768 case TEK_Aggregate: 12769 llvm_unreachable( 12770 "Aggregates are not supported in lastprivate conditional."); 12771 } 12772 // } 12773 CGF.EmitBranch(ExitBB); 12774 // There is no need to emit line number for unconditional branch. 12775 (void)ApplyDebugLocation::CreateEmpty(CGF); 12776 CGF.EmitBlock(ExitBB, /*IsFinished=*/true); 12777 }; 12778 12779 if (CGM.getLangOpts().OpenMPSimd) { 12780 // Do not emit as a critical region as no parallel region could be emitted. 12781 RegionCodeGenTy ThenRCG(CodeGen); 12782 ThenRCG(CGF); 12783 } else { 12784 emitCriticalRegion(CGF, UniqueDeclName, CodeGen, Loc); 12785 } 12786 } 12787 12788 void CGOpenMPRuntime::checkAndEmitLastprivateConditional(CodeGenFunction &CGF, 12789 const Expr *LHS) { 12790 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty()) 12791 return; 12792 LastprivateConditionalRefChecker Checker(LastprivateConditionalStack); 12793 if (!Checker.Visit(LHS)) 12794 return; 12795 const Expr *FoundE; 12796 const Decl *FoundD; 12797 StringRef UniqueDeclName; 12798 LValue IVLVal; 12799 llvm::Function *FoundFn; 12800 std::tie(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn) = 12801 Checker.getFoundData(); 12802 if (FoundFn != CGF.CurFn) { 12803 // Special codegen for inner parallel regions. 12804 // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1; 12805 auto It = LastprivateConditionalToTypes[FoundFn].find(FoundD); 12806 assert(It != LastprivateConditionalToTypes[FoundFn].end() && 12807 "Lastprivate conditional is not found in outer region."); 12808 QualType StructTy = std::get<0>(It->getSecond()); 12809 const FieldDecl* FiredDecl = std::get<2>(It->getSecond()); 12810 LValue PrivLVal = CGF.EmitLValue(FoundE); 12811 Address StructAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 12812 PrivLVal.getAddress(CGF), 12813 CGF.ConvertTypeForMem(CGF.getContext().getPointerType(StructTy)), 12814 CGF.ConvertTypeForMem(StructTy)); 12815 LValue BaseLVal = 12816 CGF.MakeAddrLValue(StructAddr, StructTy, AlignmentSource::Decl); 12817 LValue FiredLVal = CGF.EmitLValueForField(BaseLVal, FiredDecl); 12818 CGF.EmitAtomicStore(RValue::get(llvm::ConstantInt::get( 12819 CGF.ConvertTypeForMem(FiredDecl->getType()), 1)), 12820 FiredLVal, llvm::AtomicOrdering::Unordered, 12821 /*IsVolatile=*/true, /*isInit=*/false); 12822 return; 12823 } 12824 12825 // Private address of the lastprivate conditional in the current context. 12826 // priv_a 12827 LValue LVal = CGF.EmitLValue(FoundE); 12828 emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal, 12829 FoundE->getExprLoc()); 12830 } 12831 12832 void CGOpenMPRuntime::checkAndEmitSharedLastprivateConditional( 12833 CodeGenFunction &CGF, const OMPExecutableDirective &D, 12834 const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) { 12835 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty()) 12836 return; 12837 auto Range = llvm::reverse(LastprivateConditionalStack); 12838 auto It = llvm::find_if( 12839 Range, [](const LastprivateConditionalData &D) { return !D.Disabled; }); 12840 if (It == Range.end() || It->Fn != CGF.CurFn) 12841 return; 12842 auto LPCI = LastprivateConditionalToTypes.find(It->Fn); 12843 assert(LPCI != LastprivateConditionalToTypes.end() && 12844 "Lastprivates must be registered already."); 12845 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions; 12846 getOpenMPCaptureRegions(CaptureRegions, D.getDirectiveKind()); 12847 const CapturedStmt *CS = D.getCapturedStmt(CaptureRegions.back()); 12848 for (const auto &Pair : It->DeclToUniqueName) { 12849 const auto *VD = cast<VarDecl>(Pair.first->getCanonicalDecl()); 12850 if (!CS->capturesVariable(VD) || IgnoredDecls.contains(VD)) 12851 continue; 12852 auto I = LPCI->getSecond().find(Pair.first); 12853 assert(I != LPCI->getSecond().end() && 12854 "Lastprivate must be rehistered already."); 12855 // bool Cmp = priv_a.Fired != 0; 12856 LValue BaseLVal = std::get<3>(I->getSecond()); 12857 LValue FiredLVal = 12858 CGF.EmitLValueForField(BaseLVal, std::get<2>(I->getSecond())); 12859 llvm::Value *Res = CGF.EmitLoadOfScalar(FiredLVal, D.getBeginLoc()); 12860 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Res); 12861 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lpc.then"); 12862 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("lpc.done"); 12863 // if (Cmp) { 12864 CGF.Builder.CreateCondBr(Cmp, ThenBB, DoneBB); 12865 CGF.EmitBlock(ThenBB); 12866 Address Addr = CGF.GetAddrOfLocalVar(VD); 12867 LValue LVal; 12868 if (VD->getType()->isReferenceType()) 12869 LVal = CGF.EmitLoadOfReferenceLValue(Addr, VD->getType(), 12870 AlignmentSource::Decl); 12871 else 12872 LVal = CGF.MakeAddrLValue(Addr, VD->getType().getNonReferenceType(), 12873 AlignmentSource::Decl); 12874 emitLastprivateConditionalUpdate(CGF, It->IVLVal, Pair.second, LVal, 12875 D.getBeginLoc()); 12876 auto AL = ApplyDebugLocation::CreateArtificial(CGF); 12877 CGF.EmitBlock(DoneBB, /*IsFinal=*/true); 12878 // } 12879 } 12880 } 12881 12882 void CGOpenMPRuntime::emitLastprivateConditionalFinalUpdate( 12883 CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD, 12884 SourceLocation Loc) { 12885 if (CGF.getLangOpts().OpenMP < 50) 12886 return; 12887 auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(VD); 12888 assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() && 12889 "Unknown lastprivate conditional variable."); 12890 StringRef UniqueName = It->second; 12891 llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(UniqueName); 12892 // The variable was not updated in the region - exit. 12893 if (!GV) 12894 return; 12895 LValue LPLVal = CGF.MakeAddrLValue( 12896 Address(GV, GV->getValueType(), PrivLVal.getAlignment()), 12897 PrivLVal.getType().getNonReferenceType()); 12898 llvm::Value *Res = CGF.EmitLoadOfScalar(LPLVal, Loc); 12899 CGF.EmitStoreOfScalar(Res, PrivLVal); 12900 } 12901 12902 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction( 12903 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 12904 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 12905 llvm_unreachable("Not supported in SIMD-only mode"); 12906 } 12907 12908 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction( 12909 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 12910 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 12911 llvm_unreachable("Not supported in SIMD-only mode"); 12912 } 12913 12914 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction( 12915 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 12916 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 12917 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 12918 bool Tied, unsigned &NumberOfParts) { 12919 llvm_unreachable("Not supported in SIMD-only mode"); 12920 } 12921 12922 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF, 12923 SourceLocation Loc, 12924 llvm::Function *OutlinedFn, 12925 ArrayRef<llvm::Value *> CapturedVars, 12926 const Expr *IfCond, 12927 llvm::Value *NumThreads) { 12928 llvm_unreachable("Not supported in SIMD-only mode"); 12929 } 12930 12931 void CGOpenMPSIMDRuntime::emitCriticalRegion( 12932 CodeGenFunction &CGF, StringRef CriticalName, 12933 const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc, 12934 const Expr *Hint) { 12935 llvm_unreachable("Not supported in SIMD-only mode"); 12936 } 12937 12938 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF, 12939 const RegionCodeGenTy &MasterOpGen, 12940 SourceLocation Loc) { 12941 llvm_unreachable("Not supported in SIMD-only mode"); 12942 } 12943 12944 void CGOpenMPSIMDRuntime::emitMaskedRegion(CodeGenFunction &CGF, 12945 const RegionCodeGenTy &MasterOpGen, 12946 SourceLocation Loc, 12947 const Expr *Filter) { 12948 llvm_unreachable("Not supported in SIMD-only mode"); 12949 } 12950 12951 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 12952 SourceLocation Loc) { 12953 llvm_unreachable("Not supported in SIMD-only mode"); 12954 } 12955 12956 void CGOpenMPSIMDRuntime::emitTaskgroupRegion( 12957 CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen, 12958 SourceLocation Loc) { 12959 llvm_unreachable("Not supported in SIMD-only mode"); 12960 } 12961 12962 void CGOpenMPSIMDRuntime::emitSingleRegion( 12963 CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen, 12964 SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars, 12965 ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs, 12966 ArrayRef<const Expr *> AssignmentOps) { 12967 llvm_unreachable("Not supported in SIMD-only mode"); 12968 } 12969 12970 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF, 12971 const RegionCodeGenTy &OrderedOpGen, 12972 SourceLocation Loc, 12973 bool IsThreads) { 12974 llvm_unreachable("Not supported in SIMD-only mode"); 12975 } 12976 12977 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF, 12978 SourceLocation Loc, 12979 OpenMPDirectiveKind Kind, 12980 bool EmitChecks, 12981 bool ForceSimpleCall) { 12982 llvm_unreachable("Not supported in SIMD-only mode"); 12983 } 12984 12985 void CGOpenMPSIMDRuntime::emitForDispatchInit( 12986 CodeGenFunction &CGF, SourceLocation Loc, 12987 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 12988 bool Ordered, const DispatchRTInput &DispatchValues) { 12989 llvm_unreachable("Not supported in SIMD-only mode"); 12990 } 12991 12992 void CGOpenMPSIMDRuntime::emitForStaticInit( 12993 CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind, 12994 const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) { 12995 llvm_unreachable("Not supported in SIMD-only mode"); 12996 } 12997 12998 void CGOpenMPSIMDRuntime::emitDistributeStaticInit( 12999 CodeGenFunction &CGF, SourceLocation Loc, 13000 OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) { 13001 llvm_unreachable("Not supported in SIMD-only mode"); 13002 } 13003 13004 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 13005 SourceLocation Loc, 13006 unsigned IVSize, 13007 bool IVSigned) { 13008 llvm_unreachable("Not supported in SIMD-only mode"); 13009 } 13010 13011 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF, 13012 SourceLocation Loc, 13013 OpenMPDirectiveKind DKind) { 13014 llvm_unreachable("Not supported in SIMD-only mode"); 13015 } 13016 13017 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF, 13018 SourceLocation Loc, 13019 unsigned IVSize, bool IVSigned, 13020 Address IL, Address LB, 13021 Address UB, Address ST) { 13022 llvm_unreachable("Not supported in SIMD-only mode"); 13023 } 13024 13025 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 13026 llvm::Value *NumThreads, 13027 SourceLocation Loc) { 13028 llvm_unreachable("Not supported in SIMD-only mode"); 13029 } 13030 13031 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF, 13032 ProcBindKind ProcBind, 13033 SourceLocation Loc) { 13034 llvm_unreachable("Not supported in SIMD-only mode"); 13035 } 13036 13037 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 13038 const VarDecl *VD, 13039 Address VDAddr, 13040 SourceLocation Loc) { 13041 llvm_unreachable("Not supported in SIMD-only mode"); 13042 } 13043 13044 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition( 13045 const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit, 13046 CodeGenFunction *CGF) { 13047 llvm_unreachable("Not supported in SIMD-only mode"); 13048 } 13049 13050 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate( 13051 CodeGenFunction &CGF, QualType VarType, StringRef Name) { 13052 llvm_unreachable("Not supported in SIMD-only mode"); 13053 } 13054 13055 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF, 13056 ArrayRef<const Expr *> Vars, 13057 SourceLocation Loc, 13058 llvm::AtomicOrdering AO) { 13059 llvm_unreachable("Not supported in SIMD-only mode"); 13060 } 13061 13062 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 13063 const OMPExecutableDirective &D, 13064 llvm::Function *TaskFunction, 13065 QualType SharedsTy, Address Shareds, 13066 const Expr *IfCond, 13067 const OMPTaskDataTy &Data) { 13068 llvm_unreachable("Not supported in SIMD-only mode"); 13069 } 13070 13071 void CGOpenMPSIMDRuntime::emitTaskLoopCall( 13072 CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D, 13073 llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, 13074 const Expr *IfCond, const OMPTaskDataTy &Data) { 13075 llvm_unreachable("Not supported in SIMD-only mode"); 13076 } 13077 13078 void CGOpenMPSIMDRuntime::emitReduction( 13079 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates, 13080 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 13081 ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) { 13082 assert(Options.SimpleReduction && "Only simple reduction is expected."); 13083 CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs, 13084 ReductionOps, Options); 13085 } 13086 13087 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit( 13088 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 13089 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 13090 llvm_unreachable("Not supported in SIMD-only mode"); 13091 } 13092 13093 void CGOpenMPSIMDRuntime::emitTaskReductionFini(CodeGenFunction &CGF, 13094 SourceLocation Loc, 13095 bool IsWorksharingReduction) { 13096 llvm_unreachable("Not supported in SIMD-only mode"); 13097 } 13098 13099 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 13100 SourceLocation Loc, 13101 ReductionCodeGen &RCG, 13102 unsigned N) { 13103 llvm_unreachable("Not supported in SIMD-only mode"); 13104 } 13105 13106 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF, 13107 SourceLocation Loc, 13108 llvm::Value *ReductionsPtr, 13109 LValue SharedLVal) { 13110 llvm_unreachable("Not supported in SIMD-only mode"); 13111 } 13112 13113 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 13114 SourceLocation Loc, 13115 const OMPTaskDataTy &Data) { 13116 llvm_unreachable("Not supported in SIMD-only mode"); 13117 } 13118 13119 void CGOpenMPSIMDRuntime::emitCancellationPointCall( 13120 CodeGenFunction &CGF, SourceLocation Loc, 13121 OpenMPDirectiveKind CancelRegion) { 13122 llvm_unreachable("Not supported in SIMD-only mode"); 13123 } 13124 13125 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF, 13126 SourceLocation Loc, const Expr *IfCond, 13127 OpenMPDirectiveKind CancelRegion) { 13128 llvm_unreachable("Not supported in SIMD-only mode"); 13129 } 13130 13131 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction( 13132 const OMPExecutableDirective &D, StringRef ParentName, 13133 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 13134 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 13135 llvm_unreachable("Not supported in SIMD-only mode"); 13136 } 13137 13138 void CGOpenMPSIMDRuntime::emitTargetCall( 13139 CodeGenFunction &CGF, const OMPExecutableDirective &D, 13140 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 13141 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device, 13142 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 13143 const OMPLoopDirective &D)> 13144 SizeEmitter) { 13145 llvm_unreachable("Not supported in SIMD-only mode"); 13146 } 13147 13148 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) { 13149 llvm_unreachable("Not supported in SIMD-only mode"); 13150 } 13151 13152 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 13153 llvm_unreachable("Not supported in SIMD-only mode"); 13154 } 13155 13156 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) { 13157 return false; 13158 } 13159 13160 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF, 13161 const OMPExecutableDirective &D, 13162 SourceLocation Loc, 13163 llvm::Function *OutlinedFn, 13164 ArrayRef<llvm::Value *> CapturedVars) { 13165 llvm_unreachable("Not supported in SIMD-only mode"); 13166 } 13167 13168 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 13169 const Expr *NumTeams, 13170 const Expr *ThreadLimit, 13171 SourceLocation Loc) { 13172 llvm_unreachable("Not supported in SIMD-only mode"); 13173 } 13174 13175 void CGOpenMPSIMDRuntime::emitTargetDataCalls( 13176 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 13177 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 13178 llvm_unreachable("Not supported in SIMD-only mode"); 13179 } 13180 13181 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall( 13182 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 13183 const Expr *Device) { 13184 llvm_unreachable("Not supported in SIMD-only mode"); 13185 } 13186 13187 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF, 13188 const OMPLoopDirective &D, 13189 ArrayRef<Expr *> NumIterations) { 13190 llvm_unreachable("Not supported in SIMD-only mode"); 13191 } 13192 13193 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 13194 const OMPDependClause *C) { 13195 llvm_unreachable("Not supported in SIMD-only mode"); 13196 } 13197 13198 const VarDecl * 13199 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD, 13200 const VarDecl *NativeParam) const { 13201 llvm_unreachable("Not supported in SIMD-only mode"); 13202 } 13203 13204 Address 13205 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF, 13206 const VarDecl *NativeParam, 13207 const VarDecl *TargetParam) const { 13208 llvm_unreachable("Not supported in SIMD-only mode"); 13209 } 13210