1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This provides a class for OpenMP runtime code generation. 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "CGOpenMPRuntime.h" 14 #include "CGCXXABI.h" 15 #include "CGCleanup.h" 16 #include "CGRecordLayout.h" 17 #include "CodeGenFunction.h" 18 #include "clang/AST/Attr.h" 19 #include "clang/AST/Decl.h" 20 #include "clang/AST/OpenMPClause.h" 21 #include "clang/AST/StmtOpenMP.h" 22 #include "clang/AST/StmtVisitor.h" 23 #include "clang/Basic/BitmaskEnum.h" 24 #include "clang/Basic/FileManager.h" 25 #include "clang/Basic/OpenMPKinds.h" 26 #include "clang/Basic/SourceManager.h" 27 #include "clang/CodeGen/ConstantInitBuilder.h" 28 #include "llvm/ADT/ArrayRef.h" 29 #include "llvm/ADT/SetOperations.h" 30 #include "llvm/ADT/StringExtras.h" 31 #include "llvm/Bitcode/BitcodeReader.h" 32 #include "llvm/Frontend/OpenMP/OMPIRBuilder.h" 33 #include "llvm/IR/Constants.h" 34 #include "llvm/IR/DerivedTypes.h" 35 #include "llvm/IR/GlobalValue.h" 36 #include "llvm/IR/Value.h" 37 #include "llvm/Support/AtomicOrdering.h" 38 #include "llvm/Support/Format.h" 39 #include "llvm/Support/raw_ostream.h" 40 #include <cassert> 41 42 using namespace clang; 43 using namespace CodeGen; 44 using namespace llvm::omp; 45 46 namespace { 47 /// Base class for handling code generation inside OpenMP regions. 48 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo { 49 public: 50 /// Kinds of OpenMP regions used in codegen. 51 enum CGOpenMPRegionKind { 52 /// Region with outlined function for standalone 'parallel' 53 /// directive. 54 ParallelOutlinedRegion, 55 /// Region with outlined function for standalone 'task' directive. 56 TaskOutlinedRegion, 57 /// Region for constructs that do not require function outlining, 58 /// like 'for', 'sections', 'atomic' etc. directives. 59 InlinedRegion, 60 /// Region with outlined function for standalone 'target' directive. 61 TargetRegion, 62 }; 63 64 CGOpenMPRegionInfo(const CapturedStmt &CS, 65 const CGOpenMPRegionKind RegionKind, 66 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 67 bool HasCancel) 68 : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind), 69 CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {} 70 71 CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind, 72 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 73 bool HasCancel) 74 : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen), 75 Kind(Kind), HasCancel(HasCancel) {} 76 77 /// Get a variable or parameter for storing global thread id 78 /// inside OpenMP construct. 79 virtual const VarDecl *getThreadIDVariable() const = 0; 80 81 /// Emit the captured statement body. 82 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override; 83 84 /// Get an LValue for the current ThreadID variable. 85 /// \return LValue for thread id variable. This LValue always has type int32*. 86 virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF); 87 88 virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {} 89 90 CGOpenMPRegionKind getRegionKind() const { return RegionKind; } 91 92 OpenMPDirectiveKind getDirectiveKind() const { return Kind; } 93 94 bool hasCancel() const { return HasCancel; } 95 96 static bool classof(const CGCapturedStmtInfo *Info) { 97 return Info->getKind() == CR_OpenMP; 98 } 99 100 ~CGOpenMPRegionInfo() override = default; 101 102 protected: 103 CGOpenMPRegionKind RegionKind; 104 RegionCodeGenTy CodeGen; 105 OpenMPDirectiveKind Kind; 106 bool HasCancel; 107 }; 108 109 /// API for captured statement code generation in OpenMP constructs. 110 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo { 111 public: 112 CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar, 113 const RegionCodeGenTy &CodeGen, 114 OpenMPDirectiveKind Kind, bool HasCancel, 115 StringRef HelperName) 116 : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind, 117 HasCancel), 118 ThreadIDVar(ThreadIDVar), HelperName(HelperName) { 119 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 120 } 121 122 /// Get a variable or parameter for storing global thread id 123 /// inside OpenMP construct. 124 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 125 126 /// Get the name of the capture helper. 127 StringRef getHelperName() const override { return HelperName; } 128 129 static bool classof(const CGCapturedStmtInfo *Info) { 130 return CGOpenMPRegionInfo::classof(Info) && 131 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 132 ParallelOutlinedRegion; 133 } 134 135 private: 136 /// A variable or parameter storing global thread id for OpenMP 137 /// constructs. 138 const VarDecl *ThreadIDVar; 139 StringRef HelperName; 140 }; 141 142 /// API for captured statement code generation in OpenMP constructs. 143 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo { 144 public: 145 class UntiedTaskActionTy final : public PrePostActionTy { 146 bool Untied; 147 const VarDecl *PartIDVar; 148 const RegionCodeGenTy UntiedCodeGen; 149 llvm::SwitchInst *UntiedSwitch = nullptr; 150 151 public: 152 UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar, 153 const RegionCodeGenTy &UntiedCodeGen) 154 : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {} 155 void Enter(CodeGenFunction &CGF) override { 156 if (Untied) { 157 // Emit task switching point. 158 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 159 CGF.GetAddrOfLocalVar(PartIDVar), 160 PartIDVar->getType()->castAs<PointerType>()); 161 llvm::Value *Res = 162 CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation()); 163 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done."); 164 UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB); 165 CGF.EmitBlock(DoneBB); 166 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 167 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 168 UntiedSwitch->addCase(CGF.Builder.getInt32(0), 169 CGF.Builder.GetInsertBlock()); 170 emitUntiedSwitch(CGF); 171 } 172 } 173 void emitUntiedSwitch(CodeGenFunction &CGF) const { 174 if (Untied) { 175 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 176 CGF.GetAddrOfLocalVar(PartIDVar), 177 PartIDVar->getType()->castAs<PointerType>()); 178 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 179 PartIdLVal); 180 UntiedCodeGen(CGF); 181 CodeGenFunction::JumpDest CurPoint = 182 CGF.getJumpDestInCurrentScope(".untied.next."); 183 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 184 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 185 UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 186 CGF.Builder.GetInsertBlock()); 187 CGF.EmitBranchThroughCleanup(CurPoint); 188 CGF.EmitBlock(CurPoint.getBlock()); 189 } 190 } 191 unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); } 192 }; 193 CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS, 194 const VarDecl *ThreadIDVar, 195 const RegionCodeGenTy &CodeGen, 196 OpenMPDirectiveKind Kind, bool HasCancel, 197 const UntiedTaskActionTy &Action) 198 : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel), 199 ThreadIDVar(ThreadIDVar), Action(Action) { 200 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 201 } 202 203 /// Get a variable or parameter for storing global thread id 204 /// inside OpenMP construct. 205 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 206 207 /// Get an LValue for the current ThreadID variable. 208 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override; 209 210 /// Get the name of the capture helper. 211 StringRef getHelperName() const override { return ".omp_outlined."; } 212 213 void emitUntiedSwitch(CodeGenFunction &CGF) override { 214 Action.emitUntiedSwitch(CGF); 215 } 216 217 static bool classof(const CGCapturedStmtInfo *Info) { 218 return CGOpenMPRegionInfo::classof(Info) && 219 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 220 TaskOutlinedRegion; 221 } 222 223 private: 224 /// A variable or parameter storing global thread id for OpenMP 225 /// constructs. 226 const VarDecl *ThreadIDVar; 227 /// Action for emitting code for untied tasks. 228 const UntiedTaskActionTy &Action; 229 }; 230 231 /// API for inlined captured statement code generation in OpenMP 232 /// constructs. 233 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo { 234 public: 235 CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI, 236 const RegionCodeGenTy &CodeGen, 237 OpenMPDirectiveKind Kind, bool HasCancel) 238 : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel), 239 OldCSI(OldCSI), 240 OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {} 241 242 // Retrieve the value of the context parameter. 243 llvm::Value *getContextValue() const override { 244 if (OuterRegionInfo) 245 return OuterRegionInfo->getContextValue(); 246 llvm_unreachable("No context value for inlined OpenMP region"); 247 } 248 249 void setContextValue(llvm::Value *V) override { 250 if (OuterRegionInfo) { 251 OuterRegionInfo->setContextValue(V); 252 return; 253 } 254 llvm_unreachable("No context value for inlined OpenMP region"); 255 } 256 257 /// Lookup the captured field decl for a variable. 258 const FieldDecl *lookup(const VarDecl *VD) const override { 259 if (OuterRegionInfo) 260 return OuterRegionInfo->lookup(VD); 261 // If there is no outer outlined region,no need to lookup in a list of 262 // captured variables, we can use the original one. 263 return nullptr; 264 } 265 266 FieldDecl *getThisFieldDecl() const override { 267 if (OuterRegionInfo) 268 return OuterRegionInfo->getThisFieldDecl(); 269 return nullptr; 270 } 271 272 /// Get a variable or parameter for storing global thread id 273 /// inside OpenMP construct. 274 const VarDecl *getThreadIDVariable() const override { 275 if (OuterRegionInfo) 276 return OuterRegionInfo->getThreadIDVariable(); 277 return nullptr; 278 } 279 280 /// Get an LValue for the current ThreadID variable. 281 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override { 282 if (OuterRegionInfo) 283 return OuterRegionInfo->getThreadIDVariableLValue(CGF); 284 llvm_unreachable("No LValue for inlined OpenMP construct"); 285 } 286 287 /// Get the name of the capture helper. 288 StringRef getHelperName() const override { 289 if (auto *OuterRegionInfo = getOldCSI()) 290 return OuterRegionInfo->getHelperName(); 291 llvm_unreachable("No helper name for inlined OpenMP construct"); 292 } 293 294 void emitUntiedSwitch(CodeGenFunction &CGF) override { 295 if (OuterRegionInfo) 296 OuterRegionInfo->emitUntiedSwitch(CGF); 297 } 298 299 CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; } 300 301 static bool classof(const CGCapturedStmtInfo *Info) { 302 return CGOpenMPRegionInfo::classof(Info) && 303 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion; 304 } 305 306 ~CGOpenMPInlinedRegionInfo() override = default; 307 308 private: 309 /// CodeGen info about outer OpenMP region. 310 CodeGenFunction::CGCapturedStmtInfo *OldCSI; 311 CGOpenMPRegionInfo *OuterRegionInfo; 312 }; 313 314 /// API for captured statement code generation in OpenMP target 315 /// constructs. For this captures, implicit parameters are used instead of the 316 /// captured fields. The name of the target region has to be unique in a given 317 /// application so it is provided by the client, because only the client has 318 /// the information to generate that. 319 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo { 320 public: 321 CGOpenMPTargetRegionInfo(const CapturedStmt &CS, 322 const RegionCodeGenTy &CodeGen, StringRef HelperName) 323 : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target, 324 /*HasCancel=*/false), 325 HelperName(HelperName) {} 326 327 /// This is unused for target regions because each starts executing 328 /// with a single thread. 329 const VarDecl *getThreadIDVariable() const override { return nullptr; } 330 331 /// Get the name of the capture helper. 332 StringRef getHelperName() const override { return HelperName; } 333 334 static bool classof(const CGCapturedStmtInfo *Info) { 335 return CGOpenMPRegionInfo::classof(Info) && 336 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion; 337 } 338 339 private: 340 StringRef HelperName; 341 }; 342 343 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) { 344 llvm_unreachable("No codegen for expressions"); 345 } 346 /// API for generation of expressions captured in a innermost OpenMP 347 /// region. 348 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo { 349 public: 350 CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS) 351 : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen, 352 OMPD_unknown, 353 /*HasCancel=*/false), 354 PrivScope(CGF) { 355 // Make sure the globals captured in the provided statement are local by 356 // using the privatization logic. We assume the same variable is not 357 // captured more than once. 358 for (const auto &C : CS.captures()) { 359 if (!C.capturesVariable() && !C.capturesVariableByCopy()) 360 continue; 361 362 const VarDecl *VD = C.getCapturedVar(); 363 if (VD->isLocalVarDeclOrParm()) 364 continue; 365 366 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD), 367 /*RefersToEnclosingVariableOrCapture=*/false, 368 VD->getType().getNonReferenceType(), VK_LValue, 369 C.getLocation()); 370 PrivScope.addPrivate( 371 VD, [&CGF, &DRE]() { return CGF.EmitLValue(&DRE).getAddress(CGF); }); 372 } 373 (void)PrivScope.Privatize(); 374 } 375 376 /// Lookup the captured field decl for a variable. 377 const FieldDecl *lookup(const VarDecl *VD) const override { 378 if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD)) 379 return FD; 380 return nullptr; 381 } 382 383 /// Emit the captured statement body. 384 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override { 385 llvm_unreachable("No body for expressions"); 386 } 387 388 /// Get a variable or parameter for storing global thread id 389 /// inside OpenMP construct. 390 const VarDecl *getThreadIDVariable() const override { 391 llvm_unreachable("No thread id for expressions"); 392 } 393 394 /// Get the name of the capture helper. 395 StringRef getHelperName() const override { 396 llvm_unreachable("No helper name for expressions"); 397 } 398 399 static bool classof(const CGCapturedStmtInfo *Info) { return false; } 400 401 private: 402 /// Private scope to capture global variables. 403 CodeGenFunction::OMPPrivateScope PrivScope; 404 }; 405 406 /// RAII for emitting code of OpenMP constructs. 407 class InlinedOpenMPRegionRAII { 408 CodeGenFunction &CGF; 409 llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields; 410 FieldDecl *LambdaThisCaptureField = nullptr; 411 const CodeGen::CGBlockInfo *BlockInfo = nullptr; 412 413 public: 414 /// Constructs region for combined constructs. 415 /// \param CodeGen Code generation sequence for combined directives. Includes 416 /// a list of functions used for code generation of implicitly inlined 417 /// regions. 418 InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen, 419 OpenMPDirectiveKind Kind, bool HasCancel) 420 : CGF(CGF) { 421 // Start emission for the construct. 422 CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo( 423 CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel); 424 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 425 LambdaThisCaptureField = CGF.LambdaThisCaptureField; 426 CGF.LambdaThisCaptureField = nullptr; 427 BlockInfo = CGF.BlockInfo; 428 CGF.BlockInfo = nullptr; 429 } 430 431 ~InlinedOpenMPRegionRAII() { 432 // Restore original CapturedStmtInfo only if we're done with code emission. 433 auto *OldCSI = 434 cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI(); 435 delete CGF.CapturedStmtInfo; 436 CGF.CapturedStmtInfo = OldCSI; 437 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 438 CGF.LambdaThisCaptureField = LambdaThisCaptureField; 439 CGF.BlockInfo = BlockInfo; 440 } 441 }; 442 443 /// Values for bit flags used in the ident_t to describe the fields. 444 /// All enumeric elements are named and described in accordance with the code 445 /// from https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 446 enum OpenMPLocationFlags : unsigned { 447 /// Use trampoline for internal microtask. 448 OMP_IDENT_IMD = 0x01, 449 /// Use c-style ident structure. 450 OMP_IDENT_KMPC = 0x02, 451 /// Atomic reduction option for kmpc_reduce. 452 OMP_ATOMIC_REDUCE = 0x10, 453 /// Explicit 'barrier' directive. 454 OMP_IDENT_BARRIER_EXPL = 0x20, 455 /// Implicit barrier in code. 456 OMP_IDENT_BARRIER_IMPL = 0x40, 457 /// Implicit barrier in 'for' directive. 458 OMP_IDENT_BARRIER_IMPL_FOR = 0x40, 459 /// Implicit barrier in 'sections' directive. 460 OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0, 461 /// Implicit barrier in 'single' directive. 462 OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140, 463 /// Call of __kmp_for_static_init for static loop. 464 OMP_IDENT_WORK_LOOP = 0x200, 465 /// Call of __kmp_for_static_init for sections. 466 OMP_IDENT_WORK_SECTIONS = 0x400, 467 /// Call of __kmp_for_static_init for distribute. 468 OMP_IDENT_WORK_DISTRIBUTE = 0x800, 469 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE) 470 }; 471 472 namespace { 473 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 474 /// Values for bit flags for marking which requires clauses have been used. 475 enum OpenMPOffloadingRequiresDirFlags : int64_t { 476 /// flag undefined. 477 OMP_REQ_UNDEFINED = 0x000, 478 /// no requires clause present. 479 OMP_REQ_NONE = 0x001, 480 /// reverse_offload clause. 481 OMP_REQ_REVERSE_OFFLOAD = 0x002, 482 /// unified_address clause. 483 OMP_REQ_UNIFIED_ADDRESS = 0x004, 484 /// unified_shared_memory clause. 485 OMP_REQ_UNIFIED_SHARED_MEMORY = 0x008, 486 /// dynamic_allocators clause. 487 OMP_REQ_DYNAMIC_ALLOCATORS = 0x010, 488 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS) 489 }; 490 491 enum OpenMPOffloadingReservedDeviceIDs { 492 /// Device ID if the device was not defined, runtime should get it 493 /// from environment variables in the spec. 494 OMP_DEVICEID_UNDEF = -1, 495 }; 496 } // anonymous namespace 497 498 /// Describes ident structure that describes a source location. 499 /// All descriptions are taken from 500 /// https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 501 /// Original structure: 502 /// typedef struct ident { 503 /// kmp_int32 reserved_1; /**< might be used in Fortran; 504 /// see above */ 505 /// kmp_int32 flags; /**< also f.flags; KMP_IDENT_xxx flags; 506 /// KMP_IDENT_KMPC identifies this union 507 /// member */ 508 /// kmp_int32 reserved_2; /**< not really used in Fortran any more; 509 /// see above */ 510 ///#if USE_ITT_BUILD 511 /// /* but currently used for storing 512 /// region-specific ITT */ 513 /// /* contextual information. */ 514 ///#endif /* USE_ITT_BUILD */ 515 /// kmp_int32 reserved_3; /**< source[4] in Fortran, do not use for 516 /// C++ */ 517 /// char const *psource; /**< String describing the source location. 518 /// The string is composed of semi-colon separated 519 // fields which describe the source file, 520 /// the function and a pair of line numbers that 521 /// delimit the construct. 522 /// */ 523 /// } ident_t; 524 enum IdentFieldIndex { 525 /// might be used in Fortran 526 IdentField_Reserved_1, 527 /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member. 528 IdentField_Flags, 529 /// Not really used in Fortran any more 530 IdentField_Reserved_2, 531 /// Source[4] in Fortran, do not use for C++ 532 IdentField_Reserved_3, 533 /// String describing the source location. The string is composed of 534 /// semi-colon separated fields which describe the source file, the function 535 /// and a pair of line numbers that delimit the construct. 536 IdentField_PSource 537 }; 538 539 /// Schedule types for 'omp for' loops (these enumerators are taken from 540 /// the enum sched_type in kmp.h). 541 enum OpenMPSchedType { 542 /// Lower bound for default (unordered) versions. 543 OMP_sch_lower = 32, 544 OMP_sch_static_chunked = 33, 545 OMP_sch_static = 34, 546 OMP_sch_dynamic_chunked = 35, 547 OMP_sch_guided_chunked = 36, 548 OMP_sch_runtime = 37, 549 OMP_sch_auto = 38, 550 /// static with chunk adjustment (e.g., simd) 551 OMP_sch_static_balanced_chunked = 45, 552 /// Lower bound for 'ordered' versions. 553 OMP_ord_lower = 64, 554 OMP_ord_static_chunked = 65, 555 OMP_ord_static = 66, 556 OMP_ord_dynamic_chunked = 67, 557 OMP_ord_guided_chunked = 68, 558 OMP_ord_runtime = 69, 559 OMP_ord_auto = 70, 560 OMP_sch_default = OMP_sch_static, 561 /// dist_schedule types 562 OMP_dist_sch_static_chunked = 91, 563 OMP_dist_sch_static = 92, 564 /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers. 565 /// Set if the monotonic schedule modifier was present. 566 OMP_sch_modifier_monotonic = (1 << 29), 567 /// Set if the nonmonotonic schedule modifier was present. 568 OMP_sch_modifier_nonmonotonic = (1 << 30), 569 }; 570 571 enum OpenMPRTLFunction { 572 /// Call to void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, 573 /// kmpc_micro microtask, ...); 574 OMPRTL__kmpc_fork_call, 575 /// Call to void *__kmpc_threadprivate_cached(ident_t *loc, 576 /// kmp_int32 global_tid, void *data, size_t size, void ***cache); 577 OMPRTL__kmpc_threadprivate_cached, 578 /// Call to void __kmpc_threadprivate_register( ident_t *, 579 /// void *data, kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor); 580 OMPRTL__kmpc_threadprivate_register, 581 // Call to __kmpc_int32 kmpc_global_thread_num(ident_t *loc); 582 OMPRTL__kmpc_global_thread_num, 583 // Call to void __kmpc_critical(ident_t *loc, kmp_int32 global_tid, 584 // kmp_critical_name *crit); 585 OMPRTL__kmpc_critical, 586 // Call to void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 587 // global_tid, kmp_critical_name *crit, uintptr_t hint); 588 OMPRTL__kmpc_critical_with_hint, 589 // Call to void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid, 590 // kmp_critical_name *crit); 591 OMPRTL__kmpc_end_critical, 592 // Call to kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32 593 // global_tid); 594 OMPRTL__kmpc_cancel_barrier, 595 // Call to void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid); 596 OMPRTL__kmpc_barrier, 597 // Call to void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid); 598 OMPRTL__kmpc_for_static_fini, 599 // Call to void __kmpc_serialized_parallel(ident_t *loc, kmp_int32 600 // global_tid); 601 OMPRTL__kmpc_serialized_parallel, 602 // Call to void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32 603 // global_tid); 604 OMPRTL__kmpc_end_serialized_parallel, 605 // Call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid, 606 // kmp_int32 num_threads); 607 OMPRTL__kmpc_push_num_threads, 608 // Call to void __kmpc_flush(ident_t *loc); 609 OMPRTL__kmpc_flush, 610 // Call to kmp_int32 __kmpc_master(ident_t *, kmp_int32 global_tid); 611 OMPRTL__kmpc_master, 612 // Call to void __kmpc_end_master(ident_t *, kmp_int32 global_tid); 613 OMPRTL__kmpc_end_master, 614 // Call to kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid, 615 // int end_part); 616 OMPRTL__kmpc_omp_taskyield, 617 // Call to kmp_int32 __kmpc_single(ident_t *, kmp_int32 global_tid); 618 OMPRTL__kmpc_single, 619 // Call to void __kmpc_end_single(ident_t *, kmp_int32 global_tid); 620 OMPRTL__kmpc_end_single, 621 // Call to kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 622 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 623 // kmp_routine_entry_t *task_entry); 624 OMPRTL__kmpc_omp_task_alloc, 625 // Call to kmp_task_t * __kmpc_omp_target_task_alloc(ident_t *, 626 // kmp_int32 gtid, kmp_int32 flags, size_t sizeof_kmp_task_t, 627 // size_t sizeof_shareds, kmp_routine_entry_t *task_entry, 628 // kmp_int64 device_id); 629 OMPRTL__kmpc_omp_target_task_alloc, 630 // Call to kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t * 631 // new_task); 632 OMPRTL__kmpc_omp_task, 633 // Call to void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid, 634 // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *), 635 // kmp_int32 didit); 636 OMPRTL__kmpc_copyprivate, 637 // Call to kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid, 638 // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void 639 // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck); 640 OMPRTL__kmpc_reduce, 641 // Call to kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32 642 // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, 643 // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name 644 // *lck); 645 OMPRTL__kmpc_reduce_nowait, 646 // Call to void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid, 647 // kmp_critical_name *lck); 648 OMPRTL__kmpc_end_reduce, 649 // Call to void __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid, 650 // kmp_critical_name *lck); 651 OMPRTL__kmpc_end_reduce_nowait, 652 // Call to void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 653 // kmp_task_t * new_task); 654 OMPRTL__kmpc_omp_task_begin_if0, 655 // Call to void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 656 // kmp_task_t * new_task); 657 OMPRTL__kmpc_omp_task_complete_if0, 658 // Call to void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid); 659 OMPRTL__kmpc_ordered, 660 // Call to void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid); 661 OMPRTL__kmpc_end_ordered, 662 // Call to kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 663 // global_tid); 664 OMPRTL__kmpc_omp_taskwait, 665 // Call to void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid); 666 OMPRTL__kmpc_taskgroup, 667 // Call to void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid); 668 OMPRTL__kmpc_end_taskgroup, 669 // Call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid, 670 // int proc_bind); 671 OMPRTL__kmpc_push_proc_bind, 672 // Call to kmp_int32 __kmpc_omp_task_with_deps(ident_t *loc_ref, kmp_int32 673 // gtid, kmp_task_t * new_task, kmp_int32 ndeps, kmp_depend_info_t 674 // *dep_list, kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 675 OMPRTL__kmpc_omp_task_with_deps, 676 // Call to void __kmpc_omp_wait_deps(ident_t *loc_ref, kmp_int32 677 // gtid, kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 678 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 679 OMPRTL__kmpc_omp_wait_deps, 680 // Call to kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 681 // global_tid, kmp_int32 cncl_kind); 682 OMPRTL__kmpc_cancellationpoint, 683 // Call to kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 684 // kmp_int32 cncl_kind); 685 OMPRTL__kmpc_cancel, 686 // Call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32 global_tid, 687 // kmp_int32 num_teams, kmp_int32 thread_limit); 688 OMPRTL__kmpc_push_num_teams, 689 // Call to void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro 690 // microtask, ...); 691 OMPRTL__kmpc_fork_teams, 692 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 693 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 694 // sched, kmp_uint64 grainsize, void *task_dup); 695 OMPRTL__kmpc_taskloop, 696 // Call to void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32 697 // num_dims, struct kmp_dim *dims); 698 OMPRTL__kmpc_doacross_init, 699 // Call to void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid); 700 OMPRTL__kmpc_doacross_fini, 701 // Call to void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64 702 // *vec); 703 OMPRTL__kmpc_doacross_post, 704 // Call to void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64 705 // *vec); 706 OMPRTL__kmpc_doacross_wait, 707 // Call to void *__kmpc_task_reduction_init(int gtid, int num_data, void 708 // *data); 709 OMPRTL__kmpc_task_reduction_init, 710 // Call to void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 711 // *d); 712 OMPRTL__kmpc_task_reduction_get_th_data, 713 // Call to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t al); 714 OMPRTL__kmpc_alloc, 715 // Call to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t al); 716 OMPRTL__kmpc_free, 717 718 // 719 // Offloading related calls 720 // 721 // Call to void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64 722 // size); 723 OMPRTL__kmpc_push_target_tripcount, 724 // Call to int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t 725 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 726 // *arg_types); 727 OMPRTL__tgt_target, 728 // Call to int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr, 729 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 730 // *arg_types); 731 OMPRTL__tgt_target_nowait, 732 // Call to int32_t __tgt_target_teams(int64_t device_id, void *host_ptr, 733 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 734 // *arg_types, int32_t num_teams, int32_t thread_limit); 735 OMPRTL__tgt_target_teams, 736 // Call to int32_t __tgt_target_teams_nowait(int64_t device_id, void 737 // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t 738 // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 739 OMPRTL__tgt_target_teams_nowait, 740 // Call to void __tgt_register_requires(int64_t flags); 741 OMPRTL__tgt_register_requires, 742 // Call to void __tgt_target_data_begin(int64_t device_id, int32_t arg_num, 743 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 744 OMPRTL__tgt_target_data_begin, 745 // Call to void __tgt_target_data_begin_nowait(int64_t device_id, int32_t 746 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 747 // *arg_types); 748 OMPRTL__tgt_target_data_begin_nowait, 749 // Call to void __tgt_target_data_end(int64_t device_id, int32_t arg_num, 750 // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types); 751 OMPRTL__tgt_target_data_end, 752 // Call to void __tgt_target_data_end_nowait(int64_t device_id, int32_t 753 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 754 // *arg_types); 755 OMPRTL__tgt_target_data_end_nowait, 756 // Call to void __tgt_target_data_update(int64_t device_id, int32_t arg_num, 757 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 758 OMPRTL__tgt_target_data_update, 759 // Call to void __tgt_target_data_update_nowait(int64_t device_id, int32_t 760 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 761 // *arg_types); 762 OMPRTL__tgt_target_data_update_nowait, 763 // Call to int64_t __tgt_mapper_num_components(void *rt_mapper_handle); 764 OMPRTL__tgt_mapper_num_components, 765 // Call to void __tgt_push_mapper_component(void *rt_mapper_handle, void 766 // *base, void *begin, int64_t size, int64_t type); 767 OMPRTL__tgt_push_mapper_component, 768 // Call to kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref, 769 // int gtid, kmp_task_t *task); 770 OMPRTL__kmpc_task_allow_completion_event, 771 }; 772 773 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP 774 /// region. 775 class CleanupTy final : public EHScopeStack::Cleanup { 776 PrePostActionTy *Action; 777 778 public: 779 explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {} 780 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 781 if (!CGF.HaveInsertPoint()) 782 return; 783 Action->Exit(CGF); 784 } 785 }; 786 787 } // anonymous namespace 788 789 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const { 790 CodeGenFunction::RunCleanupsScope Scope(CGF); 791 if (PrePostAction) { 792 CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction); 793 Callback(CodeGen, CGF, *PrePostAction); 794 } else { 795 PrePostActionTy Action; 796 Callback(CodeGen, CGF, Action); 797 } 798 } 799 800 /// Check if the combiner is a call to UDR combiner and if it is so return the 801 /// UDR decl used for reduction. 802 static const OMPDeclareReductionDecl * 803 getReductionInit(const Expr *ReductionOp) { 804 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 805 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 806 if (const auto *DRE = 807 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 808 if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) 809 return DRD; 810 return nullptr; 811 } 812 813 static void emitInitWithReductionInitializer(CodeGenFunction &CGF, 814 const OMPDeclareReductionDecl *DRD, 815 const Expr *InitOp, 816 Address Private, Address Original, 817 QualType Ty) { 818 if (DRD->getInitializer()) { 819 std::pair<llvm::Function *, llvm::Function *> Reduction = 820 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 821 const auto *CE = cast<CallExpr>(InitOp); 822 const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee()); 823 const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts(); 824 const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts(); 825 const auto *LHSDRE = 826 cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr()); 827 const auto *RHSDRE = 828 cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr()); 829 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 830 PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()), 831 [=]() { return Private; }); 832 PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()), 833 [=]() { return Original; }); 834 (void)PrivateScope.Privatize(); 835 RValue Func = RValue::get(Reduction.second); 836 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 837 CGF.EmitIgnoredExpr(InitOp); 838 } else { 839 llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty); 840 std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"}); 841 auto *GV = new llvm::GlobalVariable( 842 CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true, 843 llvm::GlobalValue::PrivateLinkage, Init, Name); 844 LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty); 845 RValue InitRVal; 846 switch (CGF.getEvaluationKind(Ty)) { 847 case TEK_Scalar: 848 InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation()); 849 break; 850 case TEK_Complex: 851 InitRVal = 852 RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation())); 853 break; 854 case TEK_Aggregate: 855 InitRVal = RValue::getAggregate(LV.getAddress(CGF)); 856 break; 857 } 858 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_RValue); 859 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal); 860 CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(), 861 /*IsInitializer=*/false); 862 } 863 } 864 865 /// Emit initialization of arrays of complex types. 866 /// \param DestAddr Address of the array. 867 /// \param Type Type of array. 868 /// \param Init Initial expression of array. 869 /// \param SrcAddr Address of the original array. 870 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr, 871 QualType Type, bool EmitDeclareReductionInit, 872 const Expr *Init, 873 const OMPDeclareReductionDecl *DRD, 874 Address SrcAddr = Address::invalid()) { 875 // Perform element-by-element initialization. 876 QualType ElementTy; 877 878 // Drill down to the base element type on both arrays. 879 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 880 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr); 881 DestAddr = 882 CGF.Builder.CreateElementBitCast(DestAddr, DestAddr.getElementType()); 883 if (DRD) 884 SrcAddr = 885 CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType()); 886 887 llvm::Value *SrcBegin = nullptr; 888 if (DRD) 889 SrcBegin = SrcAddr.getPointer(); 890 llvm::Value *DestBegin = DestAddr.getPointer(); 891 // Cast from pointer to array type to pointer to single element. 892 llvm::Value *DestEnd = CGF.Builder.CreateGEP(DestBegin, NumElements); 893 // The basic structure here is a while-do loop. 894 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body"); 895 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done"); 896 llvm::Value *IsEmpty = 897 CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty"); 898 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 899 900 // Enter the loop body, making that address the current address. 901 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 902 CGF.EmitBlock(BodyBB); 903 904 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 905 906 llvm::PHINode *SrcElementPHI = nullptr; 907 Address SrcElementCurrent = Address::invalid(); 908 if (DRD) { 909 SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2, 910 "omp.arraycpy.srcElementPast"); 911 SrcElementPHI->addIncoming(SrcBegin, EntryBB); 912 SrcElementCurrent = 913 Address(SrcElementPHI, 914 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 915 } 916 llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI( 917 DestBegin->getType(), 2, "omp.arraycpy.destElementPast"); 918 DestElementPHI->addIncoming(DestBegin, EntryBB); 919 Address DestElementCurrent = 920 Address(DestElementPHI, 921 DestAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 922 923 // Emit copy. 924 { 925 CodeGenFunction::RunCleanupsScope InitScope(CGF); 926 if (EmitDeclareReductionInit) { 927 emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent, 928 SrcElementCurrent, ElementTy); 929 } else 930 CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(), 931 /*IsInitializer=*/false); 932 } 933 934 if (DRD) { 935 // Shift the address forward by one element. 936 llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32( 937 SrcElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 938 SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock()); 939 } 940 941 // Shift the address forward by one element. 942 llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32( 943 DestElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 944 // Check whether we've reached the end. 945 llvm::Value *Done = 946 CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done"); 947 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 948 DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock()); 949 950 // Done. 951 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 952 } 953 954 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) { 955 return CGF.EmitOMPSharedLValue(E); 956 } 957 958 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF, 959 const Expr *E) { 960 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E)) 961 return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false); 962 return LValue(); 963 } 964 965 void ReductionCodeGen::emitAggregateInitialization( 966 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 967 const OMPDeclareReductionDecl *DRD) { 968 // Emit VarDecl with copy init for arrays. 969 // Get the address of the original variable captured in current 970 // captured region. 971 const auto *PrivateVD = 972 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 973 bool EmitDeclareReductionInit = 974 DRD && (DRD->getInitializer() || !PrivateVD->hasInit()); 975 EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(), 976 EmitDeclareReductionInit, 977 EmitDeclareReductionInit ? ClausesData[N].ReductionOp 978 : PrivateVD->getInit(), 979 DRD, SharedLVal.getAddress(CGF)); 980 } 981 982 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds, 983 ArrayRef<const Expr *> Privates, 984 ArrayRef<const Expr *> ReductionOps) { 985 ClausesData.reserve(Shareds.size()); 986 SharedAddresses.reserve(Shareds.size()); 987 Sizes.reserve(Shareds.size()); 988 BaseDecls.reserve(Shareds.size()); 989 auto IPriv = Privates.begin(); 990 auto IRed = ReductionOps.begin(); 991 for (const Expr *Ref : Shareds) { 992 ClausesData.emplace_back(Ref, *IPriv, *IRed); 993 std::advance(IPriv, 1); 994 std::advance(IRed, 1); 995 } 996 } 997 998 void ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, unsigned N) { 999 assert(SharedAddresses.size() == N && 1000 "Number of generated lvalues must be exactly N."); 1001 LValue First = emitSharedLValue(CGF, ClausesData[N].Ref); 1002 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref); 1003 SharedAddresses.emplace_back(First, Second); 1004 } 1005 1006 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) { 1007 const auto *PrivateVD = 1008 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1009 QualType PrivateType = PrivateVD->getType(); 1010 bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref); 1011 if (!PrivateType->isVariablyModifiedType()) { 1012 Sizes.emplace_back( 1013 CGF.getTypeSize( 1014 SharedAddresses[N].first.getType().getNonReferenceType()), 1015 nullptr); 1016 return; 1017 } 1018 llvm::Value *Size; 1019 llvm::Value *SizeInChars; 1020 auto *ElemType = cast<llvm::PointerType>( 1021 SharedAddresses[N].first.getPointer(CGF)->getType()) 1022 ->getElementType(); 1023 auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType); 1024 if (AsArraySection) { 1025 Size = CGF.Builder.CreatePtrDiff(SharedAddresses[N].second.getPointer(CGF), 1026 SharedAddresses[N].first.getPointer(CGF)); 1027 Size = CGF.Builder.CreateNUWAdd( 1028 Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1)); 1029 SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf); 1030 } else { 1031 SizeInChars = CGF.getTypeSize( 1032 SharedAddresses[N].first.getType().getNonReferenceType()); 1033 Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf); 1034 } 1035 Sizes.emplace_back(SizeInChars, Size); 1036 CodeGenFunction::OpaqueValueMapping OpaqueMap( 1037 CGF, 1038 cast<OpaqueValueExpr>( 1039 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 1040 RValue::get(Size)); 1041 CGF.EmitVariablyModifiedType(PrivateType); 1042 } 1043 1044 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N, 1045 llvm::Value *Size) { 1046 const auto *PrivateVD = 1047 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1048 QualType PrivateType = PrivateVD->getType(); 1049 if (!PrivateType->isVariablyModifiedType()) { 1050 assert(!Size && !Sizes[N].second && 1051 "Size should be nullptr for non-variably modified reduction " 1052 "items."); 1053 return; 1054 } 1055 CodeGenFunction::OpaqueValueMapping OpaqueMap( 1056 CGF, 1057 cast<OpaqueValueExpr>( 1058 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 1059 RValue::get(Size)); 1060 CGF.EmitVariablyModifiedType(PrivateType); 1061 } 1062 1063 void ReductionCodeGen::emitInitialization( 1064 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 1065 llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) { 1066 assert(SharedAddresses.size() > N && "No variable was generated"); 1067 const auto *PrivateVD = 1068 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1069 const OMPDeclareReductionDecl *DRD = 1070 getReductionInit(ClausesData[N].ReductionOp); 1071 QualType PrivateType = PrivateVD->getType(); 1072 PrivateAddr = CGF.Builder.CreateElementBitCast( 1073 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 1074 QualType SharedType = SharedAddresses[N].first.getType(); 1075 SharedLVal = CGF.MakeAddrLValue( 1076 CGF.Builder.CreateElementBitCast(SharedLVal.getAddress(CGF), 1077 CGF.ConvertTypeForMem(SharedType)), 1078 SharedType, SharedAddresses[N].first.getBaseInfo(), 1079 CGF.CGM.getTBAAInfoForSubobject(SharedAddresses[N].first, SharedType)); 1080 if (CGF.getContext().getAsArrayType(PrivateVD->getType())) { 1081 emitAggregateInitialization(CGF, N, PrivateAddr, SharedLVal, DRD); 1082 } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) { 1083 emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp, 1084 PrivateAddr, SharedLVal.getAddress(CGF), 1085 SharedLVal.getType()); 1086 } else if (!DefaultInit(CGF) && PrivateVD->hasInit() && 1087 !CGF.isTrivialInitializer(PrivateVD->getInit())) { 1088 CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr, 1089 PrivateVD->getType().getQualifiers(), 1090 /*IsInitializer=*/false); 1091 } 1092 } 1093 1094 bool ReductionCodeGen::needCleanups(unsigned N) { 1095 const auto *PrivateVD = 1096 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1097 QualType PrivateType = PrivateVD->getType(); 1098 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 1099 return DTorKind != QualType::DK_none; 1100 } 1101 1102 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N, 1103 Address PrivateAddr) { 1104 const auto *PrivateVD = 1105 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1106 QualType PrivateType = PrivateVD->getType(); 1107 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 1108 if (needCleanups(N)) { 1109 PrivateAddr = CGF.Builder.CreateElementBitCast( 1110 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 1111 CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType); 1112 } 1113 } 1114 1115 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 1116 LValue BaseLV) { 1117 BaseTy = BaseTy.getNonReferenceType(); 1118 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 1119 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 1120 if (const auto *PtrTy = BaseTy->getAs<PointerType>()) { 1121 BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(CGF), PtrTy); 1122 } else { 1123 LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(CGF), BaseTy); 1124 BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal); 1125 } 1126 BaseTy = BaseTy->getPointeeType(); 1127 } 1128 return CGF.MakeAddrLValue( 1129 CGF.Builder.CreateElementBitCast(BaseLV.getAddress(CGF), 1130 CGF.ConvertTypeForMem(ElTy)), 1131 BaseLV.getType(), BaseLV.getBaseInfo(), 1132 CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType())); 1133 } 1134 1135 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 1136 llvm::Type *BaseLVType, CharUnits BaseLVAlignment, 1137 llvm::Value *Addr) { 1138 Address Tmp = Address::invalid(); 1139 Address TopTmp = Address::invalid(); 1140 Address MostTopTmp = Address::invalid(); 1141 BaseTy = BaseTy.getNonReferenceType(); 1142 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 1143 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 1144 Tmp = CGF.CreateMemTemp(BaseTy); 1145 if (TopTmp.isValid()) 1146 CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp); 1147 else 1148 MostTopTmp = Tmp; 1149 TopTmp = Tmp; 1150 BaseTy = BaseTy->getPointeeType(); 1151 } 1152 llvm::Type *Ty = BaseLVType; 1153 if (Tmp.isValid()) 1154 Ty = Tmp.getElementType(); 1155 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty); 1156 if (Tmp.isValid()) { 1157 CGF.Builder.CreateStore(Addr, Tmp); 1158 return MostTopTmp; 1159 } 1160 return Address(Addr, BaseLVAlignment); 1161 } 1162 1163 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) { 1164 const VarDecl *OrigVD = nullptr; 1165 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) { 1166 const Expr *Base = OASE->getBase()->IgnoreParenImpCasts(); 1167 while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base)) 1168 Base = TempOASE->getBase()->IgnoreParenImpCasts(); 1169 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 1170 Base = TempASE->getBase()->IgnoreParenImpCasts(); 1171 DE = cast<DeclRefExpr>(Base); 1172 OrigVD = cast<VarDecl>(DE->getDecl()); 1173 } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) { 1174 const Expr *Base = ASE->getBase()->IgnoreParenImpCasts(); 1175 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 1176 Base = TempASE->getBase()->IgnoreParenImpCasts(); 1177 DE = cast<DeclRefExpr>(Base); 1178 OrigVD = cast<VarDecl>(DE->getDecl()); 1179 } 1180 return OrigVD; 1181 } 1182 1183 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N, 1184 Address PrivateAddr) { 1185 const DeclRefExpr *DE; 1186 if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) { 1187 BaseDecls.emplace_back(OrigVD); 1188 LValue OriginalBaseLValue = CGF.EmitLValue(DE); 1189 LValue BaseLValue = 1190 loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(), 1191 OriginalBaseLValue); 1192 llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff( 1193 BaseLValue.getPointer(CGF), SharedAddresses[N].first.getPointer(CGF)); 1194 llvm::Value *PrivatePointer = 1195 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 1196 PrivateAddr.getPointer(), 1197 SharedAddresses[N].first.getAddress(CGF).getType()); 1198 llvm::Value *Ptr = CGF.Builder.CreateGEP(PrivatePointer, Adjustment); 1199 return castToBase(CGF, OrigVD->getType(), 1200 SharedAddresses[N].first.getType(), 1201 OriginalBaseLValue.getAddress(CGF).getType(), 1202 OriginalBaseLValue.getAlignment(), Ptr); 1203 } 1204 BaseDecls.emplace_back( 1205 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl())); 1206 return PrivateAddr; 1207 } 1208 1209 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const { 1210 const OMPDeclareReductionDecl *DRD = 1211 getReductionInit(ClausesData[N].ReductionOp); 1212 return DRD && DRD->getInitializer(); 1213 } 1214 1215 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) { 1216 return CGF.EmitLoadOfPointerLValue( 1217 CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1218 getThreadIDVariable()->getType()->castAs<PointerType>()); 1219 } 1220 1221 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) { 1222 if (!CGF.HaveInsertPoint()) 1223 return; 1224 // 1.2.2 OpenMP Language Terminology 1225 // Structured block - An executable statement with a single entry at the 1226 // top and a single exit at the bottom. 1227 // The point of exit cannot be a branch out of the structured block. 1228 // longjmp() and throw() must not violate the entry/exit criteria. 1229 CGF.EHStack.pushTerminate(); 1230 CodeGen(CGF); 1231 CGF.EHStack.popTerminate(); 1232 } 1233 1234 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue( 1235 CodeGenFunction &CGF) { 1236 return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1237 getThreadIDVariable()->getType(), 1238 AlignmentSource::Decl); 1239 } 1240 1241 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC, 1242 QualType FieldTy) { 1243 auto *Field = FieldDecl::Create( 1244 C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy, 1245 C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()), 1246 /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit); 1247 Field->setAccess(AS_public); 1248 DC->addDecl(Field); 1249 return Field; 1250 } 1251 1252 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator, 1253 StringRef Separator) 1254 : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator), 1255 OffloadEntriesInfoManager(CGM) { 1256 ASTContext &C = CGM.getContext(); 1257 RecordDecl *RD = C.buildImplicitRecord("ident_t"); 1258 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 1259 RD->startDefinition(); 1260 // reserved_1 1261 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1262 // flags 1263 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1264 // reserved_2 1265 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1266 // reserved_3 1267 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1268 // psource 1269 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 1270 RD->completeDefinition(); 1271 IdentQTy = C.getRecordType(RD); 1272 IdentTy = CGM.getTypes().ConvertRecordDeclType(RD); 1273 KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8); 1274 1275 loadOffloadInfoMetadata(); 1276 } 1277 1278 void CGOpenMPRuntime::clear() { 1279 InternalVars.clear(); 1280 // Clean non-target variable declarations possibly used only in debug info. 1281 for (const auto &Data : EmittedNonTargetVariables) { 1282 if (!Data.getValue().pointsToAliveValue()) 1283 continue; 1284 auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue()); 1285 if (!GV) 1286 continue; 1287 if (!GV->isDeclaration() || GV->getNumUses() > 0) 1288 continue; 1289 GV->eraseFromParent(); 1290 } 1291 } 1292 1293 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const { 1294 SmallString<128> Buffer; 1295 llvm::raw_svector_ostream OS(Buffer); 1296 StringRef Sep = FirstSeparator; 1297 for (StringRef Part : Parts) { 1298 OS << Sep << Part; 1299 Sep = Separator; 1300 } 1301 return std::string(OS.str()); 1302 } 1303 1304 static llvm::Function * 1305 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty, 1306 const Expr *CombinerInitializer, const VarDecl *In, 1307 const VarDecl *Out, bool IsCombiner) { 1308 // void .omp_combiner.(Ty *in, Ty *out); 1309 ASTContext &C = CGM.getContext(); 1310 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 1311 FunctionArgList Args; 1312 ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(), 1313 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1314 ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(), 1315 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1316 Args.push_back(&OmpOutParm); 1317 Args.push_back(&OmpInParm); 1318 const CGFunctionInfo &FnInfo = 1319 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 1320 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 1321 std::string Name = CGM.getOpenMPRuntime().getName( 1322 {IsCombiner ? "omp_combiner" : "omp_initializer", ""}); 1323 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 1324 Name, &CGM.getModule()); 1325 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 1326 if (CGM.getLangOpts().Optimize) { 1327 Fn->removeFnAttr(llvm::Attribute::NoInline); 1328 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 1329 Fn->addFnAttr(llvm::Attribute::AlwaysInline); 1330 } 1331 CodeGenFunction CGF(CGM); 1332 // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions. 1333 // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions. 1334 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(), 1335 Out->getLocation()); 1336 CodeGenFunction::OMPPrivateScope Scope(CGF); 1337 Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm); 1338 Scope.addPrivate(In, [&CGF, AddrIn, PtrTy]() { 1339 return CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>()) 1340 .getAddress(CGF); 1341 }); 1342 Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm); 1343 Scope.addPrivate(Out, [&CGF, AddrOut, PtrTy]() { 1344 return CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>()) 1345 .getAddress(CGF); 1346 }); 1347 (void)Scope.Privatize(); 1348 if (!IsCombiner && Out->hasInit() && 1349 !CGF.isTrivialInitializer(Out->getInit())) { 1350 CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out), 1351 Out->getType().getQualifiers(), 1352 /*IsInitializer=*/true); 1353 } 1354 if (CombinerInitializer) 1355 CGF.EmitIgnoredExpr(CombinerInitializer); 1356 Scope.ForceCleanup(); 1357 CGF.FinishFunction(); 1358 return Fn; 1359 } 1360 1361 void CGOpenMPRuntime::emitUserDefinedReduction( 1362 CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) { 1363 if (UDRMap.count(D) > 0) 1364 return; 1365 llvm::Function *Combiner = emitCombinerOrInitializer( 1366 CGM, D->getType(), D->getCombiner(), 1367 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()), 1368 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()), 1369 /*IsCombiner=*/true); 1370 llvm::Function *Initializer = nullptr; 1371 if (const Expr *Init = D->getInitializer()) { 1372 Initializer = emitCombinerOrInitializer( 1373 CGM, D->getType(), 1374 D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init 1375 : nullptr, 1376 cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()), 1377 cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()), 1378 /*IsCombiner=*/false); 1379 } 1380 UDRMap.try_emplace(D, Combiner, Initializer); 1381 if (CGF) { 1382 auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn); 1383 Decls.second.push_back(D); 1384 } 1385 } 1386 1387 std::pair<llvm::Function *, llvm::Function *> 1388 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) { 1389 auto I = UDRMap.find(D); 1390 if (I != UDRMap.end()) 1391 return I->second; 1392 emitUserDefinedReduction(/*CGF=*/nullptr, D); 1393 return UDRMap.lookup(D); 1394 } 1395 1396 namespace { 1397 // Temporary RAII solution to perform a push/pop stack event on the OpenMP IR 1398 // Builder if one is present. 1399 struct PushAndPopStackRAII { 1400 PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF, 1401 bool HasCancel) 1402 : OMPBuilder(OMPBuilder) { 1403 if (!OMPBuilder) 1404 return; 1405 1406 // The following callback is the crucial part of clangs cleanup process. 1407 // 1408 // NOTE: 1409 // Once the OpenMPIRBuilder is used to create parallel regions (and 1410 // similar), the cancellation destination (Dest below) is determined via 1411 // IP. That means if we have variables to finalize we split the block at IP, 1412 // use the new block (=BB) as destination to build a JumpDest (via 1413 // getJumpDestInCurrentScope(BB)) which then is fed to 1414 // EmitBranchThroughCleanup. Furthermore, there will not be the need 1415 // to push & pop an FinalizationInfo object. 1416 // The FiniCB will still be needed but at the point where the 1417 // OpenMPIRBuilder is asked to construct a parallel (or similar) construct. 1418 auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) { 1419 assert(IP.getBlock()->end() == IP.getPoint() && 1420 "Clang CG should cause non-terminated block!"); 1421 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1422 CGF.Builder.restoreIP(IP); 1423 CodeGenFunction::JumpDest Dest = 1424 CGF.getOMPCancelDestination(OMPD_parallel); 1425 CGF.EmitBranchThroughCleanup(Dest); 1426 }; 1427 1428 // TODO: Remove this once we emit parallel regions through the 1429 // OpenMPIRBuilder as it can do this setup internally. 1430 llvm::OpenMPIRBuilder::FinalizationInfo FI( 1431 {FiniCB, OMPD_parallel, HasCancel}); 1432 OMPBuilder->pushFinalizationCB(std::move(FI)); 1433 } 1434 ~PushAndPopStackRAII() { 1435 if (OMPBuilder) 1436 OMPBuilder->popFinalizationCB(); 1437 } 1438 llvm::OpenMPIRBuilder *OMPBuilder; 1439 }; 1440 } // namespace 1441 1442 static llvm::Function *emitParallelOrTeamsOutlinedFunction( 1443 CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS, 1444 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, 1445 const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) { 1446 assert(ThreadIDVar->getType()->isPointerType() && 1447 "thread id variable must be of type kmp_int32 *"); 1448 CodeGenFunction CGF(CGM, true); 1449 bool HasCancel = false; 1450 if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D)) 1451 HasCancel = OPD->hasCancel(); 1452 else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D)) 1453 HasCancel = OPSD->hasCancel(); 1454 else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D)) 1455 HasCancel = OPFD->hasCancel(); 1456 else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D)) 1457 HasCancel = OPFD->hasCancel(); 1458 else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D)) 1459 HasCancel = OPFD->hasCancel(); 1460 else if (const auto *OPFD = 1461 dyn_cast<OMPTeamsDistributeParallelForDirective>(&D)) 1462 HasCancel = OPFD->hasCancel(); 1463 else if (const auto *OPFD = 1464 dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D)) 1465 HasCancel = OPFD->hasCancel(); 1466 1467 // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new 1468 // parallel region to make cancellation barriers work properly. 1469 llvm::OpenMPIRBuilder *OMPBuilder = CGM.getOpenMPIRBuilder(); 1470 PushAndPopStackRAII PSR(OMPBuilder, CGF, HasCancel); 1471 CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind, 1472 HasCancel, OutlinedHelperName); 1473 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1474 return CGF.GenerateOpenMPCapturedStmtFunction(*CS, D.getBeginLoc()); 1475 } 1476 1477 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction( 1478 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1479 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1480 const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel); 1481 return emitParallelOrTeamsOutlinedFunction( 1482 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1483 } 1484 1485 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction( 1486 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1487 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1488 const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams); 1489 return emitParallelOrTeamsOutlinedFunction( 1490 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1491 } 1492 1493 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction( 1494 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1495 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 1496 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 1497 bool Tied, unsigned &NumberOfParts) { 1498 auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF, 1499 PrePostActionTy &) { 1500 llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc()); 1501 llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 1502 llvm::Value *TaskArgs[] = { 1503 UpLoc, ThreadID, 1504 CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar), 1505 TaskTVar->getType()->castAs<PointerType>()) 1506 .getPointer(CGF)}; 1507 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs); 1508 }; 1509 CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar, 1510 UntiedCodeGen); 1511 CodeGen.setAction(Action); 1512 assert(!ThreadIDVar->getType()->isPointerType() && 1513 "thread id variable must be of type kmp_int32 for tasks"); 1514 const OpenMPDirectiveKind Region = 1515 isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop 1516 : OMPD_task; 1517 const CapturedStmt *CS = D.getCapturedStmt(Region); 1518 bool HasCancel = false; 1519 if (const auto *TD = dyn_cast<OMPTaskDirective>(&D)) 1520 HasCancel = TD->hasCancel(); 1521 else if (const auto *TD = dyn_cast<OMPTaskLoopDirective>(&D)) 1522 HasCancel = TD->hasCancel(); 1523 else if (const auto *TD = dyn_cast<OMPMasterTaskLoopDirective>(&D)) 1524 HasCancel = TD->hasCancel(); 1525 else if (const auto *TD = dyn_cast<OMPParallelMasterTaskLoopDirective>(&D)) 1526 HasCancel = TD->hasCancel(); 1527 1528 CodeGenFunction CGF(CGM, true); 1529 CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, 1530 InnermostKind, HasCancel, Action); 1531 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1532 llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS); 1533 if (!Tied) 1534 NumberOfParts = Action.getNumberOfParts(); 1535 return Res; 1536 } 1537 1538 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM, 1539 const RecordDecl *RD, const CGRecordLayout &RL, 1540 ArrayRef<llvm::Constant *> Data) { 1541 llvm::StructType *StructTy = RL.getLLVMType(); 1542 unsigned PrevIdx = 0; 1543 ConstantInitBuilder CIBuilder(CGM); 1544 auto DI = Data.begin(); 1545 for (const FieldDecl *FD : RD->fields()) { 1546 unsigned Idx = RL.getLLVMFieldNo(FD); 1547 // Fill the alignment. 1548 for (unsigned I = PrevIdx; I < Idx; ++I) 1549 Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I))); 1550 PrevIdx = Idx + 1; 1551 Fields.add(*DI); 1552 ++DI; 1553 } 1554 } 1555 1556 template <class... As> 1557 static llvm::GlobalVariable * 1558 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant, 1559 ArrayRef<llvm::Constant *> Data, const Twine &Name, 1560 As &&... Args) { 1561 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1562 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1563 ConstantInitBuilder CIBuilder(CGM); 1564 ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType()); 1565 buildStructValue(Fields, CGM, RD, RL, Data); 1566 return Fields.finishAndCreateGlobal( 1567 Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant, 1568 std::forward<As>(Args)...); 1569 } 1570 1571 template <typename T> 1572 static void 1573 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty, 1574 ArrayRef<llvm::Constant *> Data, 1575 T &Parent) { 1576 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1577 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1578 ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType()); 1579 buildStructValue(Fields, CGM, RD, RL, Data); 1580 Fields.finishAndAddTo(Parent); 1581 } 1582 1583 Address CGOpenMPRuntime::getOrCreateDefaultLocation(unsigned Flags) { 1584 CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy); 1585 unsigned Reserved2Flags = getDefaultLocationReserved2Flags(); 1586 FlagsTy FlagsKey(Flags, Reserved2Flags); 1587 llvm::Value *Entry = OpenMPDefaultLocMap.lookup(FlagsKey); 1588 if (!Entry) { 1589 if (!DefaultOpenMPPSource) { 1590 // Initialize default location for psource field of ident_t structure of 1591 // all ident_t objects. Format is ";file;function;line;column;;". 1592 // Taken from 1593 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp_str.cpp 1594 DefaultOpenMPPSource = 1595 CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;").getPointer(); 1596 DefaultOpenMPPSource = 1597 llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy); 1598 } 1599 1600 llvm::Constant *Data[] = { 1601 llvm::ConstantInt::getNullValue(CGM.Int32Ty), 1602 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 1603 llvm::ConstantInt::get(CGM.Int32Ty, Reserved2Flags), 1604 llvm::ConstantInt::getNullValue(CGM.Int32Ty), DefaultOpenMPPSource}; 1605 llvm::GlobalValue *DefaultOpenMPLocation = 1606 createGlobalStruct(CGM, IdentQTy, isDefaultLocationConstant(), Data, "", 1607 llvm::GlobalValue::PrivateLinkage); 1608 DefaultOpenMPLocation->setUnnamedAddr( 1609 llvm::GlobalValue::UnnamedAddr::Global); 1610 1611 OpenMPDefaultLocMap[FlagsKey] = Entry = DefaultOpenMPLocation; 1612 } 1613 return Address(Entry, Align); 1614 } 1615 1616 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF, 1617 bool AtCurrentPoint) { 1618 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1619 assert(!Elem.second.ServiceInsertPt && "Insert point is set already."); 1620 1621 llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty); 1622 if (AtCurrentPoint) { 1623 Elem.second.ServiceInsertPt = new llvm::BitCastInst( 1624 Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock()); 1625 } else { 1626 Elem.second.ServiceInsertPt = 1627 new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt"); 1628 Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt); 1629 } 1630 } 1631 1632 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) { 1633 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1634 if (Elem.second.ServiceInsertPt) { 1635 llvm::Instruction *Ptr = Elem.second.ServiceInsertPt; 1636 Elem.second.ServiceInsertPt = nullptr; 1637 Ptr->eraseFromParent(); 1638 } 1639 } 1640 1641 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF, 1642 SourceLocation Loc, 1643 unsigned Flags) { 1644 Flags |= OMP_IDENT_KMPC; 1645 // If no debug info is generated - return global default location. 1646 if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo || 1647 Loc.isInvalid()) 1648 return getOrCreateDefaultLocation(Flags).getPointer(); 1649 1650 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1651 1652 CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy); 1653 Address LocValue = Address::invalid(); 1654 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1655 if (I != OpenMPLocThreadIDMap.end()) 1656 LocValue = Address(I->second.DebugLoc, Align); 1657 1658 // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if 1659 // GetOpenMPThreadID was called before this routine. 1660 if (!LocValue.isValid()) { 1661 // Generate "ident_t .kmpc_loc.addr;" 1662 Address AI = CGF.CreateMemTemp(IdentQTy, ".kmpc_loc.addr"); 1663 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1664 Elem.second.DebugLoc = AI.getPointer(); 1665 LocValue = AI; 1666 1667 if (!Elem.second.ServiceInsertPt) 1668 setLocThreadIdInsertPt(CGF); 1669 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1670 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1671 CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags), 1672 CGF.getTypeSize(IdentQTy)); 1673 } 1674 1675 // char **psource = &.kmpc_loc_<flags>.addr.psource; 1676 LValue Base = CGF.MakeAddrLValue(LocValue, IdentQTy); 1677 auto Fields = cast<RecordDecl>(IdentQTy->getAsTagDecl())->field_begin(); 1678 LValue PSource = 1679 CGF.EmitLValueForField(Base, *std::next(Fields, IdentField_PSource)); 1680 1681 llvm::Value *OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding()); 1682 if (OMPDebugLoc == nullptr) { 1683 SmallString<128> Buffer2; 1684 llvm::raw_svector_ostream OS2(Buffer2); 1685 // Build debug location 1686 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc); 1687 OS2 << ";" << PLoc.getFilename() << ";"; 1688 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) 1689 OS2 << FD->getQualifiedNameAsString(); 1690 OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;"; 1691 OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str()); 1692 OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc; 1693 } 1694 // *psource = ";<File>;<Function>;<Line>;<Column>;;"; 1695 CGF.EmitStoreOfScalar(OMPDebugLoc, PSource); 1696 1697 // Our callers always pass this to a runtime function, so for 1698 // convenience, go ahead and return a naked pointer. 1699 return LocValue.getPointer(); 1700 } 1701 1702 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF, 1703 SourceLocation Loc) { 1704 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1705 1706 llvm::Value *ThreadID = nullptr; 1707 // Check whether we've already cached a load of the thread id in this 1708 // function. 1709 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1710 if (I != OpenMPLocThreadIDMap.end()) { 1711 ThreadID = I->second.ThreadID; 1712 if (ThreadID != nullptr) 1713 return ThreadID; 1714 } 1715 // If exceptions are enabled, do not use parameter to avoid possible crash. 1716 if (auto *OMPRegionInfo = 1717 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 1718 if (OMPRegionInfo->getThreadIDVariable()) { 1719 // Check if this an outlined function with thread id passed as argument. 1720 LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF); 1721 llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent(); 1722 if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions || 1723 !CGF.getLangOpts().CXXExceptions || 1724 CGF.Builder.GetInsertBlock() == TopBlock || 1725 !isa<llvm::Instruction>(LVal.getPointer(CGF)) || 1726 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() == 1727 TopBlock || 1728 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() == 1729 CGF.Builder.GetInsertBlock()) { 1730 ThreadID = CGF.EmitLoadOfScalar(LVal, Loc); 1731 // If value loaded in entry block, cache it and use it everywhere in 1732 // function. 1733 if (CGF.Builder.GetInsertBlock() == TopBlock) { 1734 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1735 Elem.second.ThreadID = ThreadID; 1736 } 1737 return ThreadID; 1738 } 1739 } 1740 } 1741 1742 // This is not an outlined function region - need to call __kmpc_int32 1743 // kmpc_global_thread_num(ident_t *loc). 1744 // Generate thread id value and cache this value for use across the 1745 // function. 1746 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1747 if (!Elem.second.ServiceInsertPt) 1748 setLocThreadIdInsertPt(CGF); 1749 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1750 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1751 llvm::CallInst *Call = CGF.Builder.CreateCall( 1752 createRuntimeFunction(OMPRTL__kmpc_global_thread_num), 1753 emitUpdateLocation(CGF, Loc)); 1754 Call->setCallingConv(CGF.getRuntimeCC()); 1755 Elem.second.ThreadID = Call; 1756 return Call; 1757 } 1758 1759 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) { 1760 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1761 if (OpenMPLocThreadIDMap.count(CGF.CurFn)) { 1762 clearLocThreadIdInsertPt(CGF); 1763 OpenMPLocThreadIDMap.erase(CGF.CurFn); 1764 } 1765 if (FunctionUDRMap.count(CGF.CurFn) > 0) { 1766 for(const auto *D : FunctionUDRMap[CGF.CurFn]) 1767 UDRMap.erase(D); 1768 FunctionUDRMap.erase(CGF.CurFn); 1769 } 1770 auto I = FunctionUDMMap.find(CGF.CurFn); 1771 if (I != FunctionUDMMap.end()) { 1772 for(const auto *D : I->second) 1773 UDMMap.erase(D); 1774 FunctionUDMMap.erase(I); 1775 } 1776 LastprivateConditionalToTypes.erase(CGF.CurFn); 1777 } 1778 1779 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() { 1780 return IdentTy->getPointerTo(); 1781 } 1782 1783 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() { 1784 if (!Kmpc_MicroTy) { 1785 // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...) 1786 llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty), 1787 llvm::PointerType::getUnqual(CGM.Int32Ty)}; 1788 Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true); 1789 } 1790 return llvm::PointerType::getUnqual(Kmpc_MicroTy); 1791 } 1792 1793 llvm::FunctionCallee CGOpenMPRuntime::createRuntimeFunction(unsigned Function) { 1794 llvm::FunctionCallee RTLFn = nullptr; 1795 switch (static_cast<OpenMPRTLFunction>(Function)) { 1796 case OMPRTL__kmpc_fork_call: { 1797 // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro 1798 // microtask, ...); 1799 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1800 getKmpc_MicroPointerTy()}; 1801 auto *FnTy = 1802 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true); 1803 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call"); 1804 if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) { 1805 if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) { 1806 llvm::LLVMContext &Ctx = F->getContext(); 1807 llvm::MDBuilder MDB(Ctx); 1808 // Annotate the callback behavior of the __kmpc_fork_call: 1809 // - The callback callee is argument number 2 (microtask). 1810 // - The first two arguments of the callback callee are unknown (-1). 1811 // - All variadic arguments to the __kmpc_fork_call are passed to the 1812 // callback callee. 1813 F->addMetadata( 1814 llvm::LLVMContext::MD_callback, 1815 *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding( 1816 2, {-1, -1}, 1817 /* VarArgsArePassed */ true)})); 1818 } 1819 } 1820 break; 1821 } 1822 case OMPRTL__kmpc_global_thread_num: { 1823 // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc); 1824 llvm::Type *TypeParams[] = {getIdentTyPointerTy()}; 1825 auto *FnTy = 1826 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1827 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num"); 1828 break; 1829 } 1830 case OMPRTL__kmpc_threadprivate_cached: { 1831 // Build void *__kmpc_threadprivate_cached(ident_t *loc, 1832 // kmp_int32 global_tid, void *data, size_t size, void ***cache); 1833 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1834 CGM.VoidPtrTy, CGM.SizeTy, 1835 CGM.VoidPtrTy->getPointerTo()->getPointerTo()}; 1836 auto *FnTy = 1837 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false); 1838 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached"); 1839 break; 1840 } 1841 case OMPRTL__kmpc_critical: { 1842 // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid, 1843 // kmp_critical_name *crit); 1844 llvm::Type *TypeParams[] = { 1845 getIdentTyPointerTy(), CGM.Int32Ty, 1846 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1847 auto *FnTy = 1848 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1849 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical"); 1850 break; 1851 } 1852 case OMPRTL__kmpc_critical_with_hint: { 1853 // Build void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 global_tid, 1854 // kmp_critical_name *crit, uintptr_t hint); 1855 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1856 llvm::PointerType::getUnqual(KmpCriticalNameTy), 1857 CGM.IntPtrTy}; 1858 auto *FnTy = 1859 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1860 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical_with_hint"); 1861 break; 1862 } 1863 case OMPRTL__kmpc_threadprivate_register: { 1864 // Build void __kmpc_threadprivate_register(ident_t *, void *data, 1865 // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor); 1866 // typedef void *(*kmpc_ctor)(void *); 1867 auto *KmpcCtorTy = 1868 llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 1869 /*isVarArg*/ false)->getPointerTo(); 1870 // typedef void *(*kmpc_cctor)(void *, void *); 1871 llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 1872 auto *KmpcCopyCtorTy = 1873 llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs, 1874 /*isVarArg*/ false) 1875 ->getPointerTo(); 1876 // typedef void (*kmpc_dtor)(void *); 1877 auto *KmpcDtorTy = 1878 llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false) 1879 ->getPointerTo(); 1880 llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy, 1881 KmpcCopyCtorTy, KmpcDtorTy}; 1882 auto *FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs, 1883 /*isVarArg*/ false); 1884 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register"); 1885 break; 1886 } 1887 case OMPRTL__kmpc_end_critical: { 1888 // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid, 1889 // kmp_critical_name *crit); 1890 llvm::Type *TypeParams[] = { 1891 getIdentTyPointerTy(), CGM.Int32Ty, 1892 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1893 auto *FnTy = 1894 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1895 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical"); 1896 break; 1897 } 1898 case OMPRTL__kmpc_cancel_barrier: { 1899 // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32 1900 // global_tid); 1901 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1902 auto *FnTy = 1903 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1904 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier"); 1905 break; 1906 } 1907 case OMPRTL__kmpc_barrier: { 1908 // Build void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid); 1909 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1910 auto *FnTy = 1911 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1912 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_barrier"); 1913 break; 1914 } 1915 case OMPRTL__kmpc_for_static_fini: { 1916 // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid); 1917 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1918 auto *FnTy = 1919 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1920 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini"); 1921 break; 1922 } 1923 case OMPRTL__kmpc_push_num_threads: { 1924 // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid, 1925 // kmp_int32 num_threads) 1926 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1927 CGM.Int32Ty}; 1928 auto *FnTy = 1929 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1930 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads"); 1931 break; 1932 } 1933 case OMPRTL__kmpc_serialized_parallel: { 1934 // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32 1935 // global_tid); 1936 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1937 auto *FnTy = 1938 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1939 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel"); 1940 break; 1941 } 1942 case OMPRTL__kmpc_end_serialized_parallel: { 1943 // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32 1944 // global_tid); 1945 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1946 auto *FnTy = 1947 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1948 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel"); 1949 break; 1950 } 1951 case OMPRTL__kmpc_flush: { 1952 // Build void __kmpc_flush(ident_t *loc); 1953 llvm::Type *TypeParams[] = {getIdentTyPointerTy()}; 1954 auto *FnTy = 1955 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1956 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush"); 1957 break; 1958 } 1959 case OMPRTL__kmpc_master: { 1960 // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid); 1961 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1962 auto *FnTy = 1963 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1964 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master"); 1965 break; 1966 } 1967 case OMPRTL__kmpc_end_master: { 1968 // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid); 1969 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1970 auto *FnTy = 1971 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1972 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master"); 1973 break; 1974 } 1975 case OMPRTL__kmpc_omp_taskyield: { 1976 // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid, 1977 // int end_part); 1978 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 1979 auto *FnTy = 1980 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1981 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield"); 1982 break; 1983 } 1984 case OMPRTL__kmpc_single: { 1985 // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid); 1986 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1987 auto *FnTy = 1988 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1989 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single"); 1990 break; 1991 } 1992 case OMPRTL__kmpc_end_single: { 1993 // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid); 1994 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1995 auto *FnTy = 1996 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1997 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single"); 1998 break; 1999 } 2000 case OMPRTL__kmpc_omp_task_alloc: { 2001 // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 2002 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 2003 // kmp_routine_entry_t *task_entry); 2004 assert(KmpRoutineEntryPtrTy != nullptr && 2005 "Type kmp_routine_entry_t must be created."); 2006 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2007 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy}; 2008 // Return void * and then cast to particular kmp_task_t type. 2009 auto *FnTy = 2010 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2011 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc"); 2012 break; 2013 } 2014 case OMPRTL__kmpc_omp_target_task_alloc: { 2015 // Build kmp_task_t *__kmpc_omp_target_task_alloc(ident_t *, kmp_int32 gtid, 2016 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 2017 // kmp_routine_entry_t *task_entry, kmp_int64 device_id); 2018 assert(KmpRoutineEntryPtrTy != nullptr && 2019 "Type kmp_routine_entry_t must be created."); 2020 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2021 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy, 2022 CGM.Int64Ty}; 2023 // Return void * and then cast to particular kmp_task_t type. 2024 auto *FnTy = 2025 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2026 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_target_task_alloc"); 2027 break; 2028 } 2029 case OMPRTL__kmpc_omp_task: { 2030 // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2031 // *new_task); 2032 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2033 CGM.VoidPtrTy}; 2034 auto *FnTy = 2035 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2036 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task"); 2037 break; 2038 } 2039 case OMPRTL__kmpc_copyprivate: { 2040 // Build void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid, 2041 // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *), 2042 // kmp_int32 didit); 2043 llvm::Type *CpyTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2044 auto *CpyFnTy = 2045 llvm::FunctionType::get(CGM.VoidTy, CpyTypeParams, /*isVarArg=*/false); 2046 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.SizeTy, 2047 CGM.VoidPtrTy, CpyFnTy->getPointerTo(), 2048 CGM.Int32Ty}; 2049 auto *FnTy = 2050 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2051 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_copyprivate"); 2052 break; 2053 } 2054 case OMPRTL__kmpc_reduce: { 2055 // Build kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid, 2056 // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void 2057 // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck); 2058 llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2059 auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams, 2060 /*isVarArg=*/false); 2061 llvm::Type *TypeParams[] = { 2062 getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy, 2063 CGM.VoidPtrTy, ReduceFnTy->getPointerTo(), 2064 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2065 auto *FnTy = 2066 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2067 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce"); 2068 break; 2069 } 2070 case OMPRTL__kmpc_reduce_nowait: { 2071 // Build kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32 2072 // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, 2073 // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name 2074 // *lck); 2075 llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2076 auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams, 2077 /*isVarArg=*/false); 2078 llvm::Type *TypeParams[] = { 2079 getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy, 2080 CGM.VoidPtrTy, ReduceFnTy->getPointerTo(), 2081 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2082 auto *FnTy = 2083 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2084 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce_nowait"); 2085 break; 2086 } 2087 case OMPRTL__kmpc_end_reduce: { 2088 // Build void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid, 2089 // kmp_critical_name *lck); 2090 llvm::Type *TypeParams[] = { 2091 getIdentTyPointerTy(), CGM.Int32Ty, 2092 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2093 auto *FnTy = 2094 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2095 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce"); 2096 break; 2097 } 2098 case OMPRTL__kmpc_end_reduce_nowait: { 2099 // Build __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid, 2100 // kmp_critical_name *lck); 2101 llvm::Type *TypeParams[] = { 2102 getIdentTyPointerTy(), CGM.Int32Ty, 2103 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2104 auto *FnTy = 2105 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2106 RTLFn = 2107 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce_nowait"); 2108 break; 2109 } 2110 case OMPRTL__kmpc_omp_task_begin_if0: { 2111 // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2112 // *new_task); 2113 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2114 CGM.VoidPtrTy}; 2115 auto *FnTy = 2116 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2117 RTLFn = 2118 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_begin_if0"); 2119 break; 2120 } 2121 case OMPRTL__kmpc_omp_task_complete_if0: { 2122 // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2123 // *new_task); 2124 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2125 CGM.VoidPtrTy}; 2126 auto *FnTy = 2127 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2128 RTLFn = CGM.CreateRuntimeFunction(FnTy, 2129 /*Name=*/"__kmpc_omp_task_complete_if0"); 2130 break; 2131 } 2132 case OMPRTL__kmpc_ordered: { 2133 // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid); 2134 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2135 auto *FnTy = 2136 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2137 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_ordered"); 2138 break; 2139 } 2140 case OMPRTL__kmpc_end_ordered: { 2141 // Build void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid); 2142 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2143 auto *FnTy = 2144 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2145 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_ordered"); 2146 break; 2147 } 2148 case OMPRTL__kmpc_omp_taskwait: { 2149 // Build kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 global_tid); 2150 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2151 auto *FnTy = 2152 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2153 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_omp_taskwait"); 2154 break; 2155 } 2156 case OMPRTL__kmpc_taskgroup: { 2157 // Build void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid); 2158 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2159 auto *FnTy = 2160 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2161 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_taskgroup"); 2162 break; 2163 } 2164 case OMPRTL__kmpc_end_taskgroup: { 2165 // Build void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid); 2166 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2167 auto *FnTy = 2168 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2169 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_taskgroup"); 2170 break; 2171 } 2172 case OMPRTL__kmpc_push_proc_bind: { 2173 // Build void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid, 2174 // int proc_bind) 2175 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2176 auto *FnTy = 2177 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2178 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_proc_bind"); 2179 break; 2180 } 2181 case OMPRTL__kmpc_omp_task_with_deps: { 2182 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 2183 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 2184 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 2185 llvm::Type *TypeParams[] = { 2186 getIdentTyPointerTy(), CGM.Int32Ty, CGM.VoidPtrTy, CGM.Int32Ty, 2187 CGM.VoidPtrTy, CGM.Int32Ty, CGM.VoidPtrTy}; 2188 auto *FnTy = 2189 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2190 RTLFn = 2191 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_with_deps"); 2192 break; 2193 } 2194 case OMPRTL__kmpc_omp_wait_deps: { 2195 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 2196 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 ndeps_noalias, 2197 // kmp_depend_info_t *noalias_dep_list); 2198 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2199 CGM.Int32Ty, CGM.VoidPtrTy, 2200 CGM.Int32Ty, CGM.VoidPtrTy}; 2201 auto *FnTy = 2202 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2203 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_wait_deps"); 2204 break; 2205 } 2206 case OMPRTL__kmpc_cancellationpoint: { 2207 // Build kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 2208 // global_tid, kmp_int32 cncl_kind) 2209 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2210 auto *FnTy = 2211 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2212 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancellationpoint"); 2213 break; 2214 } 2215 case OMPRTL__kmpc_cancel: { 2216 // Build kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 2217 // kmp_int32 cncl_kind) 2218 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2219 auto *FnTy = 2220 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2221 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancel"); 2222 break; 2223 } 2224 case OMPRTL__kmpc_push_num_teams: { 2225 // Build void kmpc_push_num_teams (ident_t loc, kmp_int32 global_tid, 2226 // kmp_int32 num_teams, kmp_int32 num_threads) 2227 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2228 CGM.Int32Ty}; 2229 auto *FnTy = 2230 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2231 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_teams"); 2232 break; 2233 } 2234 case OMPRTL__kmpc_fork_teams: { 2235 // Build void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro 2236 // microtask, ...); 2237 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2238 getKmpc_MicroPointerTy()}; 2239 auto *FnTy = 2240 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true); 2241 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_teams"); 2242 if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) { 2243 if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) { 2244 llvm::LLVMContext &Ctx = F->getContext(); 2245 llvm::MDBuilder MDB(Ctx); 2246 // Annotate the callback behavior of the __kmpc_fork_teams: 2247 // - The callback callee is argument number 2 (microtask). 2248 // - The first two arguments of the callback callee are unknown (-1). 2249 // - All variadic arguments to the __kmpc_fork_teams are passed to the 2250 // callback callee. 2251 F->addMetadata( 2252 llvm::LLVMContext::MD_callback, 2253 *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding( 2254 2, {-1, -1}, 2255 /* VarArgsArePassed */ true)})); 2256 } 2257 } 2258 break; 2259 } 2260 case OMPRTL__kmpc_taskloop: { 2261 // Build void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 2262 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 2263 // sched, kmp_uint64 grainsize, void *task_dup); 2264 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), 2265 CGM.IntTy, 2266 CGM.VoidPtrTy, 2267 CGM.IntTy, 2268 CGM.Int64Ty->getPointerTo(), 2269 CGM.Int64Ty->getPointerTo(), 2270 CGM.Int64Ty, 2271 CGM.IntTy, 2272 CGM.IntTy, 2273 CGM.Int64Ty, 2274 CGM.VoidPtrTy}; 2275 auto *FnTy = 2276 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2277 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_taskloop"); 2278 break; 2279 } 2280 case OMPRTL__kmpc_doacross_init: { 2281 // Build void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32 2282 // num_dims, struct kmp_dim *dims); 2283 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), 2284 CGM.Int32Ty, 2285 CGM.Int32Ty, 2286 CGM.VoidPtrTy}; 2287 auto *FnTy = 2288 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2289 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_init"); 2290 break; 2291 } 2292 case OMPRTL__kmpc_doacross_fini: { 2293 // Build void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid); 2294 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2295 auto *FnTy = 2296 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2297 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_fini"); 2298 break; 2299 } 2300 case OMPRTL__kmpc_doacross_post: { 2301 // Build void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64 2302 // *vec); 2303 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2304 CGM.Int64Ty->getPointerTo()}; 2305 auto *FnTy = 2306 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2307 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_post"); 2308 break; 2309 } 2310 case OMPRTL__kmpc_doacross_wait: { 2311 // Build void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64 2312 // *vec); 2313 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2314 CGM.Int64Ty->getPointerTo()}; 2315 auto *FnTy = 2316 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2317 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_wait"); 2318 break; 2319 } 2320 case OMPRTL__kmpc_task_reduction_init: { 2321 // Build void *__kmpc_task_reduction_init(int gtid, int num_data, void 2322 // *data); 2323 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.IntTy, CGM.VoidPtrTy}; 2324 auto *FnTy = 2325 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2326 RTLFn = 2327 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_task_reduction_init"); 2328 break; 2329 } 2330 case OMPRTL__kmpc_task_reduction_get_th_data: { 2331 // Build void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 2332 // *d); 2333 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy}; 2334 auto *FnTy = 2335 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2336 RTLFn = CGM.CreateRuntimeFunction( 2337 FnTy, /*Name=*/"__kmpc_task_reduction_get_th_data"); 2338 break; 2339 } 2340 case OMPRTL__kmpc_alloc: { 2341 // Build to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t 2342 // al); omp_allocator_handle_t type is void *. 2343 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.SizeTy, CGM.VoidPtrTy}; 2344 auto *FnTy = 2345 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2346 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_alloc"); 2347 break; 2348 } 2349 case OMPRTL__kmpc_free: { 2350 // Build to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t 2351 // al); omp_allocator_handle_t type is void *. 2352 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy}; 2353 auto *FnTy = 2354 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2355 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_free"); 2356 break; 2357 } 2358 case OMPRTL__kmpc_push_target_tripcount: { 2359 // Build void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64 2360 // size); 2361 llvm::Type *TypeParams[] = {CGM.Int64Ty, CGM.Int64Ty}; 2362 llvm::FunctionType *FnTy = 2363 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2364 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_target_tripcount"); 2365 break; 2366 } 2367 case OMPRTL__tgt_target: { 2368 // Build int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t 2369 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2370 // *arg_types); 2371 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2372 CGM.VoidPtrTy, 2373 CGM.Int32Ty, 2374 CGM.VoidPtrPtrTy, 2375 CGM.VoidPtrPtrTy, 2376 CGM.Int64Ty->getPointerTo(), 2377 CGM.Int64Ty->getPointerTo()}; 2378 auto *FnTy = 2379 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2380 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target"); 2381 break; 2382 } 2383 case OMPRTL__tgt_target_nowait: { 2384 // Build int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr, 2385 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, 2386 // int64_t *arg_types); 2387 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2388 CGM.VoidPtrTy, 2389 CGM.Int32Ty, 2390 CGM.VoidPtrPtrTy, 2391 CGM.VoidPtrPtrTy, 2392 CGM.Int64Ty->getPointerTo(), 2393 CGM.Int64Ty->getPointerTo()}; 2394 auto *FnTy = 2395 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2396 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_nowait"); 2397 break; 2398 } 2399 case OMPRTL__tgt_target_teams: { 2400 // Build int32_t __tgt_target_teams(int64_t device_id, void *host_ptr, 2401 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, 2402 // int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 2403 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2404 CGM.VoidPtrTy, 2405 CGM.Int32Ty, 2406 CGM.VoidPtrPtrTy, 2407 CGM.VoidPtrPtrTy, 2408 CGM.Int64Ty->getPointerTo(), 2409 CGM.Int64Ty->getPointerTo(), 2410 CGM.Int32Ty, 2411 CGM.Int32Ty}; 2412 auto *FnTy = 2413 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2414 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams"); 2415 break; 2416 } 2417 case OMPRTL__tgt_target_teams_nowait: { 2418 // Build int32_t __tgt_target_teams_nowait(int64_t device_id, void 2419 // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t 2420 // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 2421 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2422 CGM.VoidPtrTy, 2423 CGM.Int32Ty, 2424 CGM.VoidPtrPtrTy, 2425 CGM.VoidPtrPtrTy, 2426 CGM.Int64Ty->getPointerTo(), 2427 CGM.Int64Ty->getPointerTo(), 2428 CGM.Int32Ty, 2429 CGM.Int32Ty}; 2430 auto *FnTy = 2431 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2432 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams_nowait"); 2433 break; 2434 } 2435 case OMPRTL__tgt_register_requires: { 2436 // Build void __tgt_register_requires(int64_t flags); 2437 llvm::Type *TypeParams[] = {CGM.Int64Ty}; 2438 auto *FnTy = 2439 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2440 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_requires"); 2441 break; 2442 } 2443 case OMPRTL__tgt_target_data_begin: { 2444 // Build void __tgt_target_data_begin(int64_t device_id, int32_t arg_num, 2445 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2446 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2447 CGM.Int32Ty, 2448 CGM.VoidPtrPtrTy, 2449 CGM.VoidPtrPtrTy, 2450 CGM.Int64Ty->getPointerTo(), 2451 CGM.Int64Ty->getPointerTo()}; 2452 auto *FnTy = 2453 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2454 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin"); 2455 break; 2456 } 2457 case OMPRTL__tgt_target_data_begin_nowait: { 2458 // Build void __tgt_target_data_begin_nowait(int64_t device_id, int32_t 2459 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2460 // *arg_types); 2461 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2462 CGM.Int32Ty, 2463 CGM.VoidPtrPtrTy, 2464 CGM.VoidPtrPtrTy, 2465 CGM.Int64Ty->getPointerTo(), 2466 CGM.Int64Ty->getPointerTo()}; 2467 auto *FnTy = 2468 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2469 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin_nowait"); 2470 break; 2471 } 2472 case OMPRTL__tgt_target_data_end: { 2473 // Build void __tgt_target_data_end(int64_t device_id, int32_t arg_num, 2474 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2475 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2476 CGM.Int32Ty, 2477 CGM.VoidPtrPtrTy, 2478 CGM.VoidPtrPtrTy, 2479 CGM.Int64Ty->getPointerTo(), 2480 CGM.Int64Ty->getPointerTo()}; 2481 auto *FnTy = 2482 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2483 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end"); 2484 break; 2485 } 2486 case OMPRTL__tgt_target_data_end_nowait: { 2487 // Build void __tgt_target_data_end_nowait(int64_t device_id, int32_t 2488 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2489 // *arg_types); 2490 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2491 CGM.Int32Ty, 2492 CGM.VoidPtrPtrTy, 2493 CGM.VoidPtrPtrTy, 2494 CGM.Int64Ty->getPointerTo(), 2495 CGM.Int64Ty->getPointerTo()}; 2496 auto *FnTy = 2497 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2498 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end_nowait"); 2499 break; 2500 } 2501 case OMPRTL__tgt_target_data_update: { 2502 // Build void __tgt_target_data_update(int64_t device_id, int32_t arg_num, 2503 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2504 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2505 CGM.Int32Ty, 2506 CGM.VoidPtrPtrTy, 2507 CGM.VoidPtrPtrTy, 2508 CGM.Int64Ty->getPointerTo(), 2509 CGM.Int64Ty->getPointerTo()}; 2510 auto *FnTy = 2511 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2512 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update"); 2513 break; 2514 } 2515 case OMPRTL__tgt_target_data_update_nowait: { 2516 // Build void __tgt_target_data_update_nowait(int64_t device_id, int32_t 2517 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2518 // *arg_types); 2519 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2520 CGM.Int32Ty, 2521 CGM.VoidPtrPtrTy, 2522 CGM.VoidPtrPtrTy, 2523 CGM.Int64Ty->getPointerTo(), 2524 CGM.Int64Ty->getPointerTo()}; 2525 auto *FnTy = 2526 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2527 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update_nowait"); 2528 break; 2529 } 2530 case OMPRTL__tgt_mapper_num_components: { 2531 // Build int64_t __tgt_mapper_num_components(void *rt_mapper_handle); 2532 llvm::Type *TypeParams[] = {CGM.VoidPtrTy}; 2533 auto *FnTy = 2534 llvm::FunctionType::get(CGM.Int64Ty, TypeParams, /*isVarArg*/ false); 2535 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_mapper_num_components"); 2536 break; 2537 } 2538 case OMPRTL__tgt_push_mapper_component: { 2539 // Build void __tgt_push_mapper_component(void *rt_mapper_handle, void 2540 // *base, void *begin, int64_t size, int64_t type); 2541 llvm::Type *TypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy, CGM.VoidPtrTy, 2542 CGM.Int64Ty, CGM.Int64Ty}; 2543 auto *FnTy = 2544 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2545 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_push_mapper_component"); 2546 break; 2547 } 2548 case OMPRTL__kmpc_task_allow_completion_event: { 2549 // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref, 2550 // int gtid, kmp_task_t *task); 2551 auto *FnTy = llvm::FunctionType::get( 2552 CGM.VoidPtrTy, {getIdentTyPointerTy(), CGM.IntTy, CGM.VoidPtrTy}, 2553 /*isVarArg=*/false); 2554 RTLFn = 2555 CGM.CreateRuntimeFunction(FnTy, "__kmpc_task_allow_completion_event"); 2556 break; 2557 } 2558 } 2559 assert(RTLFn && "Unable to find OpenMP runtime function"); 2560 return RTLFn; 2561 } 2562 2563 llvm::FunctionCallee 2564 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned) { 2565 assert((IVSize == 32 || IVSize == 64) && 2566 "IV size is not compatible with the omp runtime"); 2567 StringRef Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4" 2568 : "__kmpc_for_static_init_4u") 2569 : (IVSigned ? "__kmpc_for_static_init_8" 2570 : "__kmpc_for_static_init_8u"); 2571 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2572 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 2573 llvm::Type *TypeParams[] = { 2574 getIdentTyPointerTy(), // loc 2575 CGM.Int32Ty, // tid 2576 CGM.Int32Ty, // schedtype 2577 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 2578 PtrTy, // p_lower 2579 PtrTy, // p_upper 2580 PtrTy, // p_stride 2581 ITy, // incr 2582 ITy // chunk 2583 }; 2584 auto *FnTy = 2585 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2586 return CGM.CreateRuntimeFunction(FnTy, Name); 2587 } 2588 2589 llvm::FunctionCallee 2590 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) { 2591 assert((IVSize == 32 || IVSize == 64) && 2592 "IV size is not compatible with the omp runtime"); 2593 StringRef Name = 2594 IVSize == 32 2595 ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u") 2596 : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u"); 2597 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2598 llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc 2599 CGM.Int32Ty, // tid 2600 CGM.Int32Ty, // schedtype 2601 ITy, // lower 2602 ITy, // upper 2603 ITy, // stride 2604 ITy // chunk 2605 }; 2606 auto *FnTy = 2607 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2608 return CGM.CreateRuntimeFunction(FnTy, Name); 2609 } 2610 2611 llvm::FunctionCallee 2612 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) { 2613 assert((IVSize == 32 || IVSize == 64) && 2614 "IV size is not compatible with the omp runtime"); 2615 StringRef Name = 2616 IVSize == 32 2617 ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u") 2618 : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u"); 2619 llvm::Type *TypeParams[] = { 2620 getIdentTyPointerTy(), // loc 2621 CGM.Int32Ty, // tid 2622 }; 2623 auto *FnTy = 2624 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2625 return CGM.CreateRuntimeFunction(FnTy, Name); 2626 } 2627 2628 llvm::FunctionCallee 2629 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) { 2630 assert((IVSize == 32 || IVSize == 64) && 2631 "IV size is not compatible with the omp runtime"); 2632 StringRef Name = 2633 IVSize == 32 2634 ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u") 2635 : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u"); 2636 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2637 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 2638 llvm::Type *TypeParams[] = { 2639 getIdentTyPointerTy(), // loc 2640 CGM.Int32Ty, // tid 2641 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 2642 PtrTy, // p_lower 2643 PtrTy, // p_upper 2644 PtrTy // p_stride 2645 }; 2646 auto *FnTy = 2647 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2648 return CGM.CreateRuntimeFunction(FnTy, Name); 2649 } 2650 2651 /// Obtain information that uniquely identifies a target entry. This 2652 /// consists of the file and device IDs as well as line number associated with 2653 /// the relevant entry source location. 2654 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc, 2655 unsigned &DeviceID, unsigned &FileID, 2656 unsigned &LineNum) { 2657 SourceManager &SM = C.getSourceManager(); 2658 2659 // The loc should be always valid and have a file ID (the user cannot use 2660 // #pragma directives in macros) 2661 2662 assert(Loc.isValid() && "Source location is expected to be always valid."); 2663 2664 PresumedLoc PLoc = SM.getPresumedLoc(Loc); 2665 assert(PLoc.isValid() && "Source location is expected to be always valid."); 2666 2667 llvm::sys::fs::UniqueID ID; 2668 if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID)) 2669 SM.getDiagnostics().Report(diag::err_cannot_open_file) 2670 << PLoc.getFilename() << EC.message(); 2671 2672 DeviceID = ID.getDevice(); 2673 FileID = ID.getFile(); 2674 LineNum = PLoc.getLine(); 2675 } 2676 2677 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) { 2678 if (CGM.getLangOpts().OpenMPSimd) 2679 return Address::invalid(); 2680 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 2681 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 2682 if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link || 2683 (*Res == OMPDeclareTargetDeclAttr::MT_To && 2684 HasRequiresUnifiedSharedMemory))) { 2685 SmallString<64> PtrName; 2686 { 2687 llvm::raw_svector_ostream OS(PtrName); 2688 OS << CGM.getMangledName(GlobalDecl(VD)); 2689 if (!VD->isExternallyVisible()) { 2690 unsigned DeviceID, FileID, Line; 2691 getTargetEntryUniqueInfo(CGM.getContext(), 2692 VD->getCanonicalDecl()->getBeginLoc(), 2693 DeviceID, FileID, Line); 2694 OS << llvm::format("_%x", FileID); 2695 } 2696 OS << "_decl_tgt_ref_ptr"; 2697 } 2698 llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName); 2699 if (!Ptr) { 2700 QualType PtrTy = CGM.getContext().getPointerType(VD->getType()); 2701 Ptr = getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(PtrTy), 2702 PtrName); 2703 2704 auto *GV = cast<llvm::GlobalVariable>(Ptr); 2705 GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 2706 2707 if (!CGM.getLangOpts().OpenMPIsDevice) 2708 GV->setInitializer(CGM.GetAddrOfGlobal(VD)); 2709 registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr)); 2710 } 2711 return Address(Ptr, CGM.getContext().getDeclAlign(VD)); 2712 } 2713 return Address::invalid(); 2714 } 2715 2716 llvm::Constant * 2717 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) { 2718 assert(!CGM.getLangOpts().OpenMPUseTLS || 2719 !CGM.getContext().getTargetInfo().isTLSSupported()); 2720 // Lookup the entry, lazily creating it if necessary. 2721 std::string Suffix = getName({"cache", ""}); 2722 return getOrCreateInternalVariable( 2723 CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix)); 2724 } 2725 2726 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 2727 const VarDecl *VD, 2728 Address VDAddr, 2729 SourceLocation Loc) { 2730 if (CGM.getLangOpts().OpenMPUseTLS && 2731 CGM.getContext().getTargetInfo().isTLSSupported()) 2732 return VDAddr; 2733 2734 llvm::Type *VarTy = VDAddr.getElementType(); 2735 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2736 CGF.Builder.CreatePointerCast(VDAddr.getPointer(), 2737 CGM.Int8PtrTy), 2738 CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)), 2739 getOrCreateThreadPrivateCache(VD)}; 2740 return Address(CGF.EmitRuntimeCall( 2741 createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args), 2742 VDAddr.getAlignment()); 2743 } 2744 2745 void CGOpenMPRuntime::emitThreadPrivateVarInit( 2746 CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor, 2747 llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) { 2748 // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime 2749 // library. 2750 llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc); 2751 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num), 2752 OMPLoc); 2753 // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor) 2754 // to register constructor/destructor for variable. 2755 llvm::Value *Args[] = { 2756 OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy), 2757 Ctor, CopyCtor, Dtor}; 2758 CGF.EmitRuntimeCall( 2759 createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args); 2760 } 2761 2762 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition( 2763 const VarDecl *VD, Address VDAddr, SourceLocation Loc, 2764 bool PerformInit, CodeGenFunction *CGF) { 2765 if (CGM.getLangOpts().OpenMPUseTLS && 2766 CGM.getContext().getTargetInfo().isTLSSupported()) 2767 return nullptr; 2768 2769 VD = VD->getDefinition(CGM.getContext()); 2770 if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) { 2771 QualType ASTTy = VD->getType(); 2772 2773 llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr; 2774 const Expr *Init = VD->getAnyInitializer(); 2775 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 2776 // Generate function that re-emits the declaration's initializer into the 2777 // threadprivate copy of the variable VD 2778 CodeGenFunction CtorCGF(CGM); 2779 FunctionArgList Args; 2780 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 2781 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 2782 ImplicitParamDecl::Other); 2783 Args.push_back(&Dst); 2784 2785 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 2786 CGM.getContext().VoidPtrTy, Args); 2787 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2788 std::string Name = getName({"__kmpc_global_ctor_", ""}); 2789 llvm::Function *Fn = 2790 CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc); 2791 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI, 2792 Args, Loc, Loc); 2793 llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar( 2794 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 2795 CGM.getContext().VoidPtrTy, Dst.getLocation()); 2796 Address Arg = Address(ArgVal, VDAddr.getAlignment()); 2797 Arg = CtorCGF.Builder.CreateElementBitCast( 2798 Arg, CtorCGF.ConvertTypeForMem(ASTTy)); 2799 CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(), 2800 /*IsInitializer=*/true); 2801 ArgVal = CtorCGF.EmitLoadOfScalar( 2802 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 2803 CGM.getContext().VoidPtrTy, Dst.getLocation()); 2804 CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue); 2805 CtorCGF.FinishFunction(); 2806 Ctor = Fn; 2807 } 2808 if (VD->getType().isDestructedType() != QualType::DK_none) { 2809 // Generate function that emits destructor call for the threadprivate copy 2810 // of the variable VD 2811 CodeGenFunction DtorCGF(CGM); 2812 FunctionArgList Args; 2813 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 2814 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 2815 ImplicitParamDecl::Other); 2816 Args.push_back(&Dst); 2817 2818 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 2819 CGM.getContext().VoidTy, Args); 2820 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2821 std::string Name = getName({"__kmpc_global_dtor_", ""}); 2822 llvm::Function *Fn = 2823 CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc); 2824 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 2825 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args, 2826 Loc, Loc); 2827 // Create a scope with an artificial location for the body of this function. 2828 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 2829 llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar( 2830 DtorCGF.GetAddrOfLocalVar(&Dst), 2831 /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation()); 2832 DtorCGF.emitDestroy(Address(ArgVal, VDAddr.getAlignment()), ASTTy, 2833 DtorCGF.getDestroyer(ASTTy.isDestructedType()), 2834 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 2835 DtorCGF.FinishFunction(); 2836 Dtor = Fn; 2837 } 2838 // Do not emit init function if it is not required. 2839 if (!Ctor && !Dtor) 2840 return nullptr; 2841 2842 llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2843 auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs, 2844 /*isVarArg=*/false) 2845 ->getPointerTo(); 2846 // Copying constructor for the threadprivate variable. 2847 // Must be NULL - reserved by runtime, but currently it requires that this 2848 // parameter is always NULL. Otherwise it fires assertion. 2849 CopyCtor = llvm::Constant::getNullValue(CopyCtorTy); 2850 if (Ctor == nullptr) { 2851 auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 2852 /*isVarArg=*/false) 2853 ->getPointerTo(); 2854 Ctor = llvm::Constant::getNullValue(CtorTy); 2855 } 2856 if (Dtor == nullptr) { 2857 auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, 2858 /*isVarArg=*/false) 2859 ->getPointerTo(); 2860 Dtor = llvm::Constant::getNullValue(DtorTy); 2861 } 2862 if (!CGF) { 2863 auto *InitFunctionTy = 2864 llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false); 2865 std::string Name = getName({"__omp_threadprivate_init_", ""}); 2866 llvm::Function *InitFunction = CGM.CreateGlobalInitOrDestructFunction( 2867 InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction()); 2868 CodeGenFunction InitCGF(CGM); 2869 FunctionArgList ArgList; 2870 InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction, 2871 CGM.getTypes().arrangeNullaryFunction(), ArgList, 2872 Loc, Loc); 2873 emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 2874 InitCGF.FinishFunction(); 2875 return InitFunction; 2876 } 2877 emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 2878 } 2879 return nullptr; 2880 } 2881 2882 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD, 2883 llvm::GlobalVariable *Addr, 2884 bool PerformInit) { 2885 if (CGM.getLangOpts().OMPTargetTriples.empty() && 2886 !CGM.getLangOpts().OpenMPIsDevice) 2887 return false; 2888 Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 2889 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 2890 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 2891 (*Res == OMPDeclareTargetDeclAttr::MT_To && 2892 HasRequiresUnifiedSharedMemory)) 2893 return CGM.getLangOpts().OpenMPIsDevice; 2894 VD = VD->getDefinition(CGM.getContext()); 2895 assert(VD && "Unknown VarDecl"); 2896 2897 if (!DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second) 2898 return CGM.getLangOpts().OpenMPIsDevice; 2899 2900 QualType ASTTy = VD->getType(); 2901 SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc(); 2902 2903 // Produce the unique prefix to identify the new target regions. We use 2904 // the source location of the variable declaration which we know to not 2905 // conflict with any target region. 2906 unsigned DeviceID; 2907 unsigned FileID; 2908 unsigned Line; 2909 getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line); 2910 SmallString<128> Buffer, Out; 2911 { 2912 llvm::raw_svector_ostream OS(Buffer); 2913 OS << "__omp_offloading_" << llvm::format("_%x", DeviceID) 2914 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 2915 } 2916 2917 const Expr *Init = VD->getAnyInitializer(); 2918 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 2919 llvm::Constant *Ctor; 2920 llvm::Constant *ID; 2921 if (CGM.getLangOpts().OpenMPIsDevice) { 2922 // Generate function that re-emits the declaration's initializer into 2923 // the threadprivate copy of the variable VD 2924 CodeGenFunction CtorCGF(CGM); 2925 2926 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 2927 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2928 llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction( 2929 FTy, Twine(Buffer, "_ctor"), FI, Loc); 2930 auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF); 2931 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 2932 FunctionArgList(), Loc, Loc); 2933 auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF); 2934 CtorCGF.EmitAnyExprToMem(Init, 2935 Address(Addr, CGM.getContext().getDeclAlign(VD)), 2936 Init->getType().getQualifiers(), 2937 /*IsInitializer=*/true); 2938 CtorCGF.FinishFunction(); 2939 Ctor = Fn; 2940 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 2941 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor)); 2942 } else { 2943 Ctor = new llvm::GlobalVariable( 2944 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 2945 llvm::GlobalValue::PrivateLinkage, 2946 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor")); 2947 ID = Ctor; 2948 } 2949 2950 // Register the information for the entry associated with the constructor. 2951 Out.clear(); 2952 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 2953 DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor, 2954 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor); 2955 } 2956 if (VD->getType().isDestructedType() != QualType::DK_none) { 2957 llvm::Constant *Dtor; 2958 llvm::Constant *ID; 2959 if (CGM.getLangOpts().OpenMPIsDevice) { 2960 // Generate function that emits destructor call for the threadprivate 2961 // copy of the variable VD 2962 CodeGenFunction DtorCGF(CGM); 2963 2964 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 2965 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2966 llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction( 2967 FTy, Twine(Buffer, "_dtor"), FI, Loc); 2968 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 2969 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 2970 FunctionArgList(), Loc, Loc); 2971 // Create a scope with an artificial location for the body of this 2972 // function. 2973 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 2974 DtorCGF.emitDestroy(Address(Addr, CGM.getContext().getDeclAlign(VD)), 2975 ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()), 2976 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 2977 DtorCGF.FinishFunction(); 2978 Dtor = Fn; 2979 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 2980 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor)); 2981 } else { 2982 Dtor = new llvm::GlobalVariable( 2983 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 2984 llvm::GlobalValue::PrivateLinkage, 2985 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor")); 2986 ID = Dtor; 2987 } 2988 // Register the information for the entry associated with the destructor. 2989 Out.clear(); 2990 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 2991 DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor, 2992 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor); 2993 } 2994 return CGM.getLangOpts().OpenMPIsDevice; 2995 } 2996 2997 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF, 2998 QualType VarType, 2999 StringRef Name) { 3000 std::string Suffix = getName({"artificial", ""}); 3001 llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType); 3002 llvm::Value *GAddr = 3003 getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix)); 3004 if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS && 3005 CGM.getTarget().isTLSSupported()) { 3006 cast<llvm::GlobalVariable>(GAddr)->setThreadLocal(/*Val=*/true); 3007 return Address(GAddr, CGM.getContext().getTypeAlignInChars(VarType)); 3008 } 3009 std::string CacheSuffix = getName({"cache", ""}); 3010 llvm::Value *Args[] = { 3011 emitUpdateLocation(CGF, SourceLocation()), 3012 getThreadID(CGF, SourceLocation()), 3013 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy), 3014 CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy, 3015 /*isSigned=*/false), 3016 getOrCreateInternalVariable( 3017 CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))}; 3018 return Address( 3019 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3020 CGF.EmitRuntimeCall( 3021 createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args), 3022 VarLVType->getPointerTo(/*AddrSpace=*/0)), 3023 CGM.getContext().getTypeAlignInChars(VarType)); 3024 } 3025 3026 void CGOpenMPRuntime::emitIfClause(CodeGenFunction &CGF, const Expr *Cond, 3027 const RegionCodeGenTy &ThenGen, 3028 const RegionCodeGenTy &ElseGen) { 3029 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange()); 3030 3031 // If the condition constant folds and can be elided, try to avoid emitting 3032 // the condition and the dead arm of the if/else. 3033 bool CondConstant; 3034 if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) { 3035 if (CondConstant) 3036 ThenGen(CGF); 3037 else 3038 ElseGen(CGF); 3039 return; 3040 } 3041 3042 // Otherwise, the condition did not fold, or we couldn't elide it. Just 3043 // emit the conditional branch. 3044 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then"); 3045 llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else"); 3046 llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end"); 3047 CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0); 3048 3049 // Emit the 'then' code. 3050 CGF.EmitBlock(ThenBlock); 3051 ThenGen(CGF); 3052 CGF.EmitBranch(ContBlock); 3053 // Emit the 'else' code if present. 3054 // There is no need to emit line number for unconditional branch. 3055 (void)ApplyDebugLocation::CreateEmpty(CGF); 3056 CGF.EmitBlock(ElseBlock); 3057 ElseGen(CGF); 3058 // There is no need to emit line number for unconditional branch. 3059 (void)ApplyDebugLocation::CreateEmpty(CGF); 3060 CGF.EmitBranch(ContBlock); 3061 // Emit the continuation block for code after the if. 3062 CGF.EmitBlock(ContBlock, /*IsFinished=*/true); 3063 } 3064 3065 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc, 3066 llvm::Function *OutlinedFn, 3067 ArrayRef<llvm::Value *> CapturedVars, 3068 const Expr *IfCond) { 3069 if (!CGF.HaveInsertPoint()) 3070 return; 3071 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 3072 auto &&ThenGen = [OutlinedFn, CapturedVars, RTLoc](CodeGenFunction &CGF, 3073 PrePostActionTy &) { 3074 // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn); 3075 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 3076 llvm::Value *Args[] = { 3077 RTLoc, 3078 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 3079 CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())}; 3080 llvm::SmallVector<llvm::Value *, 16> RealArgs; 3081 RealArgs.append(std::begin(Args), std::end(Args)); 3082 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 3083 3084 llvm::FunctionCallee RTLFn = 3085 RT.createRuntimeFunction(OMPRTL__kmpc_fork_call); 3086 CGF.EmitRuntimeCall(RTLFn, RealArgs); 3087 }; 3088 auto &&ElseGen = [OutlinedFn, CapturedVars, RTLoc, Loc](CodeGenFunction &CGF, 3089 PrePostActionTy &) { 3090 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 3091 llvm::Value *ThreadID = RT.getThreadID(CGF, Loc); 3092 // Build calls: 3093 // __kmpc_serialized_parallel(&Loc, GTid); 3094 llvm::Value *Args[] = {RTLoc, ThreadID}; 3095 CGF.EmitRuntimeCall( 3096 RT.createRuntimeFunction(OMPRTL__kmpc_serialized_parallel), Args); 3097 3098 // OutlinedFn(>id, &zero_bound, CapturedStruct); 3099 Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc); 3100 Address ZeroAddrBound = 3101 CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty, 3102 /*Name=*/".bound.zero.addr"); 3103 CGF.InitTempAlloca(ZeroAddrBound, CGF.Builder.getInt32(/*C*/ 0)); 3104 llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs; 3105 // ThreadId for serialized parallels is 0. 3106 OutlinedFnArgs.push_back(ThreadIDAddr.getPointer()); 3107 OutlinedFnArgs.push_back(ZeroAddrBound.getPointer()); 3108 OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end()); 3109 RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs); 3110 3111 // __kmpc_end_serialized_parallel(&Loc, GTid); 3112 llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID}; 3113 CGF.EmitRuntimeCall( 3114 RT.createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel), 3115 EndArgs); 3116 }; 3117 if (IfCond) { 3118 emitIfClause(CGF, IfCond, ThenGen, ElseGen); 3119 } else { 3120 RegionCodeGenTy ThenRCG(ThenGen); 3121 ThenRCG(CGF); 3122 } 3123 } 3124 3125 // If we're inside an (outlined) parallel region, use the region info's 3126 // thread-ID variable (it is passed in a first argument of the outlined function 3127 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in 3128 // regular serial code region, get thread ID by calling kmp_int32 3129 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and 3130 // return the address of that temp. 3131 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF, 3132 SourceLocation Loc) { 3133 if (auto *OMPRegionInfo = 3134 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 3135 if (OMPRegionInfo->getThreadIDVariable()) 3136 return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(CGF); 3137 3138 llvm::Value *ThreadID = getThreadID(CGF, Loc); 3139 QualType Int32Ty = 3140 CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true); 3141 Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp."); 3142 CGF.EmitStoreOfScalar(ThreadID, 3143 CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty)); 3144 3145 return ThreadIDTemp; 3146 } 3147 3148 llvm::Constant *CGOpenMPRuntime::getOrCreateInternalVariable( 3149 llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) { 3150 SmallString<256> Buffer; 3151 llvm::raw_svector_ostream Out(Buffer); 3152 Out << Name; 3153 StringRef RuntimeName = Out.str(); 3154 auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first; 3155 if (Elem.second) { 3156 assert(Elem.second->getType()->getPointerElementType() == Ty && 3157 "OMP internal variable has different type than requested"); 3158 return &*Elem.second; 3159 } 3160 3161 return Elem.second = new llvm::GlobalVariable( 3162 CGM.getModule(), Ty, /*IsConstant*/ false, 3163 llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty), 3164 Elem.first(), /*InsertBefore=*/nullptr, 3165 llvm::GlobalValue::NotThreadLocal, AddressSpace); 3166 } 3167 3168 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) { 3169 std::string Prefix = Twine("gomp_critical_user_", CriticalName).str(); 3170 std::string Name = getName({Prefix, "var"}); 3171 return getOrCreateInternalVariable(KmpCriticalNameTy, Name); 3172 } 3173 3174 namespace { 3175 /// Common pre(post)-action for different OpenMP constructs. 3176 class CommonActionTy final : public PrePostActionTy { 3177 llvm::FunctionCallee EnterCallee; 3178 ArrayRef<llvm::Value *> EnterArgs; 3179 llvm::FunctionCallee ExitCallee; 3180 ArrayRef<llvm::Value *> ExitArgs; 3181 bool Conditional; 3182 llvm::BasicBlock *ContBlock = nullptr; 3183 3184 public: 3185 CommonActionTy(llvm::FunctionCallee EnterCallee, 3186 ArrayRef<llvm::Value *> EnterArgs, 3187 llvm::FunctionCallee ExitCallee, 3188 ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false) 3189 : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee), 3190 ExitArgs(ExitArgs), Conditional(Conditional) {} 3191 void Enter(CodeGenFunction &CGF) override { 3192 llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs); 3193 if (Conditional) { 3194 llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes); 3195 auto *ThenBlock = CGF.createBasicBlock("omp_if.then"); 3196 ContBlock = CGF.createBasicBlock("omp_if.end"); 3197 // Generate the branch (If-stmt) 3198 CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock); 3199 CGF.EmitBlock(ThenBlock); 3200 } 3201 } 3202 void Done(CodeGenFunction &CGF) { 3203 // Emit the rest of blocks/branches 3204 CGF.EmitBranch(ContBlock); 3205 CGF.EmitBlock(ContBlock, true); 3206 } 3207 void Exit(CodeGenFunction &CGF) override { 3208 CGF.EmitRuntimeCall(ExitCallee, ExitArgs); 3209 } 3210 }; 3211 } // anonymous namespace 3212 3213 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF, 3214 StringRef CriticalName, 3215 const RegionCodeGenTy &CriticalOpGen, 3216 SourceLocation Loc, const Expr *Hint) { 3217 // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]); 3218 // CriticalOpGen(); 3219 // __kmpc_end_critical(ident_t *, gtid, Lock); 3220 // Prepare arguments and build a call to __kmpc_critical 3221 if (!CGF.HaveInsertPoint()) 3222 return; 3223 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3224 getCriticalRegionLock(CriticalName)}; 3225 llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args), 3226 std::end(Args)); 3227 if (Hint) { 3228 EnterArgs.push_back(CGF.Builder.CreateIntCast( 3229 CGF.EmitScalarExpr(Hint), CGM.IntPtrTy, /*isSigned=*/false)); 3230 } 3231 CommonActionTy Action( 3232 createRuntimeFunction(Hint ? OMPRTL__kmpc_critical_with_hint 3233 : OMPRTL__kmpc_critical), 3234 EnterArgs, createRuntimeFunction(OMPRTL__kmpc_end_critical), Args); 3235 CriticalOpGen.setAction(Action); 3236 emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen); 3237 } 3238 3239 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF, 3240 const RegionCodeGenTy &MasterOpGen, 3241 SourceLocation Loc) { 3242 if (!CGF.HaveInsertPoint()) 3243 return; 3244 // if(__kmpc_master(ident_t *, gtid)) { 3245 // MasterOpGen(); 3246 // __kmpc_end_master(ident_t *, gtid); 3247 // } 3248 // Prepare arguments and build a call to __kmpc_master 3249 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3250 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_master), Args, 3251 createRuntimeFunction(OMPRTL__kmpc_end_master), Args, 3252 /*Conditional=*/true); 3253 MasterOpGen.setAction(Action); 3254 emitInlinedDirective(CGF, OMPD_master, MasterOpGen); 3255 Action.Done(CGF); 3256 } 3257 3258 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 3259 SourceLocation Loc) { 3260 if (!CGF.HaveInsertPoint()) 3261 return; 3262 llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder(); 3263 if (OMPBuilder) { 3264 OMPBuilder->CreateTaskyield(CGF.Builder); 3265 } else { 3266 // Build call __kmpc_omp_taskyield(loc, thread_id, 0); 3267 llvm::Value *Args[] = { 3268 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3269 llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)}; 3270 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield), 3271 Args); 3272 } 3273 3274 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 3275 Region->emitUntiedSwitch(CGF); 3276 } 3277 3278 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF, 3279 const RegionCodeGenTy &TaskgroupOpGen, 3280 SourceLocation Loc) { 3281 if (!CGF.HaveInsertPoint()) 3282 return; 3283 // __kmpc_taskgroup(ident_t *, gtid); 3284 // TaskgroupOpGen(); 3285 // __kmpc_end_taskgroup(ident_t *, gtid); 3286 // Prepare arguments and build a call to __kmpc_taskgroup 3287 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3288 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_taskgroup), Args, 3289 createRuntimeFunction(OMPRTL__kmpc_end_taskgroup), 3290 Args); 3291 TaskgroupOpGen.setAction(Action); 3292 emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen); 3293 } 3294 3295 /// Given an array of pointers to variables, project the address of a 3296 /// given variable. 3297 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array, 3298 unsigned Index, const VarDecl *Var) { 3299 // Pull out the pointer to the variable. 3300 Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index); 3301 llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr); 3302 3303 Address Addr = Address(Ptr, CGF.getContext().getDeclAlign(Var)); 3304 Addr = CGF.Builder.CreateElementBitCast( 3305 Addr, CGF.ConvertTypeForMem(Var->getType())); 3306 return Addr; 3307 } 3308 3309 static llvm::Value *emitCopyprivateCopyFunction( 3310 CodeGenModule &CGM, llvm::Type *ArgsType, 3311 ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs, 3312 ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps, 3313 SourceLocation Loc) { 3314 ASTContext &C = CGM.getContext(); 3315 // void copy_func(void *LHSArg, void *RHSArg); 3316 FunctionArgList Args; 3317 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 3318 ImplicitParamDecl::Other); 3319 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 3320 ImplicitParamDecl::Other); 3321 Args.push_back(&LHSArg); 3322 Args.push_back(&RHSArg); 3323 const auto &CGFI = 3324 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 3325 std::string Name = 3326 CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"}); 3327 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 3328 llvm::GlobalValue::InternalLinkage, Name, 3329 &CGM.getModule()); 3330 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 3331 Fn->setDoesNotRecurse(); 3332 CodeGenFunction CGF(CGM); 3333 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 3334 // Dest = (void*[n])(LHSArg); 3335 // Src = (void*[n])(RHSArg); 3336 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3337 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 3338 ArgsType), CGF.getPointerAlign()); 3339 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3340 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 3341 ArgsType), CGF.getPointerAlign()); 3342 // *(Type0*)Dst[0] = *(Type0*)Src[0]; 3343 // *(Type1*)Dst[1] = *(Type1*)Src[1]; 3344 // ... 3345 // *(Typen*)Dst[n] = *(Typen*)Src[n]; 3346 for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) { 3347 const auto *DestVar = 3348 cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl()); 3349 Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar); 3350 3351 const auto *SrcVar = 3352 cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl()); 3353 Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar); 3354 3355 const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl(); 3356 QualType Type = VD->getType(); 3357 CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]); 3358 } 3359 CGF.FinishFunction(); 3360 return Fn; 3361 } 3362 3363 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF, 3364 const RegionCodeGenTy &SingleOpGen, 3365 SourceLocation Loc, 3366 ArrayRef<const Expr *> CopyprivateVars, 3367 ArrayRef<const Expr *> SrcExprs, 3368 ArrayRef<const Expr *> DstExprs, 3369 ArrayRef<const Expr *> AssignmentOps) { 3370 if (!CGF.HaveInsertPoint()) 3371 return; 3372 assert(CopyprivateVars.size() == SrcExprs.size() && 3373 CopyprivateVars.size() == DstExprs.size() && 3374 CopyprivateVars.size() == AssignmentOps.size()); 3375 ASTContext &C = CGM.getContext(); 3376 // int32 did_it = 0; 3377 // if(__kmpc_single(ident_t *, gtid)) { 3378 // SingleOpGen(); 3379 // __kmpc_end_single(ident_t *, gtid); 3380 // did_it = 1; 3381 // } 3382 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 3383 // <copy_func>, did_it); 3384 3385 Address DidIt = Address::invalid(); 3386 if (!CopyprivateVars.empty()) { 3387 // int32 did_it = 0; 3388 QualType KmpInt32Ty = 3389 C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 3390 DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it"); 3391 CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt); 3392 } 3393 // Prepare arguments and build a call to __kmpc_single 3394 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3395 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_single), Args, 3396 createRuntimeFunction(OMPRTL__kmpc_end_single), Args, 3397 /*Conditional=*/true); 3398 SingleOpGen.setAction(Action); 3399 emitInlinedDirective(CGF, OMPD_single, SingleOpGen); 3400 if (DidIt.isValid()) { 3401 // did_it = 1; 3402 CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt); 3403 } 3404 Action.Done(CGF); 3405 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 3406 // <copy_func>, did_it); 3407 if (DidIt.isValid()) { 3408 llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size()); 3409 QualType CopyprivateArrayTy = C.getConstantArrayType( 3410 C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 3411 /*IndexTypeQuals=*/0); 3412 // Create a list of all private variables for copyprivate. 3413 Address CopyprivateList = 3414 CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list"); 3415 for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) { 3416 Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I); 3417 CGF.Builder.CreateStore( 3418 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3419 CGF.EmitLValue(CopyprivateVars[I]).getPointer(CGF), 3420 CGF.VoidPtrTy), 3421 Elem); 3422 } 3423 // Build function that copies private values from single region to all other 3424 // threads in the corresponding parallel region. 3425 llvm::Value *CpyFn = emitCopyprivateCopyFunction( 3426 CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(), 3427 CopyprivateVars, SrcExprs, DstExprs, AssignmentOps, Loc); 3428 llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy); 3429 Address CL = 3430 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList, 3431 CGF.VoidPtrTy); 3432 llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt); 3433 llvm::Value *Args[] = { 3434 emitUpdateLocation(CGF, Loc), // ident_t *<loc> 3435 getThreadID(CGF, Loc), // i32 <gtid> 3436 BufSize, // size_t <buf_size> 3437 CL.getPointer(), // void *<copyprivate list> 3438 CpyFn, // void (*) (void *, void *) <copy_func> 3439 DidItVal // i32 did_it 3440 }; 3441 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_copyprivate), Args); 3442 } 3443 } 3444 3445 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF, 3446 const RegionCodeGenTy &OrderedOpGen, 3447 SourceLocation Loc, bool IsThreads) { 3448 if (!CGF.HaveInsertPoint()) 3449 return; 3450 // __kmpc_ordered(ident_t *, gtid); 3451 // OrderedOpGen(); 3452 // __kmpc_end_ordered(ident_t *, gtid); 3453 // Prepare arguments and build a call to __kmpc_ordered 3454 if (IsThreads) { 3455 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3456 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_ordered), Args, 3457 createRuntimeFunction(OMPRTL__kmpc_end_ordered), 3458 Args); 3459 OrderedOpGen.setAction(Action); 3460 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 3461 return; 3462 } 3463 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 3464 } 3465 3466 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) { 3467 unsigned Flags; 3468 if (Kind == OMPD_for) 3469 Flags = OMP_IDENT_BARRIER_IMPL_FOR; 3470 else if (Kind == OMPD_sections) 3471 Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS; 3472 else if (Kind == OMPD_single) 3473 Flags = OMP_IDENT_BARRIER_IMPL_SINGLE; 3474 else if (Kind == OMPD_barrier) 3475 Flags = OMP_IDENT_BARRIER_EXPL; 3476 else 3477 Flags = OMP_IDENT_BARRIER_IMPL; 3478 return Flags; 3479 } 3480 3481 void CGOpenMPRuntime::getDefaultScheduleAndChunk( 3482 CodeGenFunction &CGF, const OMPLoopDirective &S, 3483 OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const { 3484 // Check if the loop directive is actually a doacross loop directive. In this 3485 // case choose static, 1 schedule. 3486 if (llvm::any_of( 3487 S.getClausesOfKind<OMPOrderedClause>(), 3488 [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) { 3489 ScheduleKind = OMPC_SCHEDULE_static; 3490 // Chunk size is 1 in this case. 3491 llvm::APInt ChunkSize(32, 1); 3492 ChunkExpr = IntegerLiteral::Create( 3493 CGF.getContext(), ChunkSize, 3494 CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0), 3495 SourceLocation()); 3496 } 3497 } 3498 3499 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc, 3500 OpenMPDirectiveKind Kind, bool EmitChecks, 3501 bool ForceSimpleCall) { 3502 // Check if we should use the OMPBuilder 3503 auto *OMPRegionInfo = 3504 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo); 3505 llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder(); 3506 if (OMPBuilder) { 3507 CGF.Builder.restoreIP(OMPBuilder->CreateBarrier( 3508 CGF.Builder, Kind, ForceSimpleCall, EmitChecks)); 3509 return; 3510 } 3511 3512 if (!CGF.HaveInsertPoint()) 3513 return; 3514 // Build call __kmpc_cancel_barrier(loc, thread_id); 3515 // Build call __kmpc_barrier(loc, thread_id); 3516 unsigned Flags = getDefaultFlagsForBarriers(Kind); 3517 // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc, 3518 // thread_id); 3519 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags), 3520 getThreadID(CGF, Loc)}; 3521 if (OMPRegionInfo) { 3522 if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) { 3523 llvm::Value *Result = CGF.EmitRuntimeCall( 3524 createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args); 3525 if (EmitChecks) { 3526 // if (__kmpc_cancel_barrier()) { 3527 // exit from construct; 3528 // } 3529 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 3530 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 3531 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 3532 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 3533 CGF.EmitBlock(ExitBB); 3534 // exit from construct; 3535 CodeGenFunction::JumpDest CancelDestination = 3536 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 3537 CGF.EmitBranchThroughCleanup(CancelDestination); 3538 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 3539 } 3540 return; 3541 } 3542 } 3543 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_barrier), Args); 3544 } 3545 3546 /// Map the OpenMP loop schedule to the runtime enumeration. 3547 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind, 3548 bool Chunked, bool Ordered) { 3549 switch (ScheduleKind) { 3550 case OMPC_SCHEDULE_static: 3551 return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked) 3552 : (Ordered ? OMP_ord_static : OMP_sch_static); 3553 case OMPC_SCHEDULE_dynamic: 3554 return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked; 3555 case OMPC_SCHEDULE_guided: 3556 return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked; 3557 case OMPC_SCHEDULE_runtime: 3558 return Ordered ? OMP_ord_runtime : OMP_sch_runtime; 3559 case OMPC_SCHEDULE_auto: 3560 return Ordered ? OMP_ord_auto : OMP_sch_auto; 3561 case OMPC_SCHEDULE_unknown: 3562 assert(!Chunked && "chunk was specified but schedule kind not known"); 3563 return Ordered ? OMP_ord_static : OMP_sch_static; 3564 } 3565 llvm_unreachable("Unexpected runtime schedule"); 3566 } 3567 3568 /// Map the OpenMP distribute schedule to the runtime enumeration. 3569 static OpenMPSchedType 3570 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) { 3571 // only static is allowed for dist_schedule 3572 return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static; 3573 } 3574 3575 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind, 3576 bool Chunked) const { 3577 OpenMPSchedType Schedule = 3578 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 3579 return Schedule == OMP_sch_static; 3580 } 3581 3582 bool CGOpenMPRuntime::isStaticNonchunked( 3583 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 3584 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 3585 return Schedule == OMP_dist_sch_static; 3586 } 3587 3588 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind, 3589 bool Chunked) const { 3590 OpenMPSchedType Schedule = 3591 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 3592 return Schedule == OMP_sch_static_chunked; 3593 } 3594 3595 bool CGOpenMPRuntime::isStaticChunked( 3596 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 3597 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 3598 return Schedule == OMP_dist_sch_static_chunked; 3599 } 3600 3601 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const { 3602 OpenMPSchedType Schedule = 3603 getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false); 3604 assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here"); 3605 return Schedule != OMP_sch_static; 3606 } 3607 3608 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule, 3609 OpenMPScheduleClauseModifier M1, 3610 OpenMPScheduleClauseModifier M2) { 3611 int Modifier = 0; 3612 switch (M1) { 3613 case OMPC_SCHEDULE_MODIFIER_monotonic: 3614 Modifier = OMP_sch_modifier_monotonic; 3615 break; 3616 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 3617 Modifier = OMP_sch_modifier_nonmonotonic; 3618 break; 3619 case OMPC_SCHEDULE_MODIFIER_simd: 3620 if (Schedule == OMP_sch_static_chunked) 3621 Schedule = OMP_sch_static_balanced_chunked; 3622 break; 3623 case OMPC_SCHEDULE_MODIFIER_last: 3624 case OMPC_SCHEDULE_MODIFIER_unknown: 3625 break; 3626 } 3627 switch (M2) { 3628 case OMPC_SCHEDULE_MODIFIER_monotonic: 3629 Modifier = OMP_sch_modifier_monotonic; 3630 break; 3631 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 3632 Modifier = OMP_sch_modifier_nonmonotonic; 3633 break; 3634 case OMPC_SCHEDULE_MODIFIER_simd: 3635 if (Schedule == OMP_sch_static_chunked) 3636 Schedule = OMP_sch_static_balanced_chunked; 3637 break; 3638 case OMPC_SCHEDULE_MODIFIER_last: 3639 case OMPC_SCHEDULE_MODIFIER_unknown: 3640 break; 3641 } 3642 // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription. 3643 // If the static schedule kind is specified or if the ordered clause is 3644 // specified, and if the nonmonotonic modifier is not specified, the effect is 3645 // as if the monotonic modifier is specified. Otherwise, unless the monotonic 3646 // modifier is specified, the effect is as if the nonmonotonic modifier is 3647 // specified. 3648 if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) { 3649 if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static || 3650 Schedule == OMP_sch_static_balanced_chunked || 3651 Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static || 3652 Schedule == OMP_dist_sch_static_chunked || 3653 Schedule == OMP_dist_sch_static)) 3654 Modifier = OMP_sch_modifier_nonmonotonic; 3655 } 3656 return Schedule | Modifier; 3657 } 3658 3659 void CGOpenMPRuntime::emitForDispatchInit( 3660 CodeGenFunction &CGF, SourceLocation Loc, 3661 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 3662 bool Ordered, const DispatchRTInput &DispatchValues) { 3663 if (!CGF.HaveInsertPoint()) 3664 return; 3665 OpenMPSchedType Schedule = getRuntimeSchedule( 3666 ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered); 3667 assert(Ordered || 3668 (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked && 3669 Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked && 3670 Schedule != OMP_sch_static_balanced_chunked)); 3671 // Call __kmpc_dispatch_init( 3672 // ident_t *loc, kmp_int32 tid, kmp_int32 schedule, 3673 // kmp_int[32|64] lower, kmp_int[32|64] upper, 3674 // kmp_int[32|64] stride, kmp_int[32|64] chunk); 3675 3676 // If the Chunk was not specified in the clause - use default value 1. 3677 llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk 3678 : CGF.Builder.getIntN(IVSize, 1); 3679 llvm::Value *Args[] = { 3680 emitUpdateLocation(CGF, Loc), 3681 getThreadID(CGF, Loc), 3682 CGF.Builder.getInt32(addMonoNonMonoModifier( 3683 CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type 3684 DispatchValues.LB, // Lower 3685 DispatchValues.UB, // Upper 3686 CGF.Builder.getIntN(IVSize, 1), // Stride 3687 Chunk // Chunk 3688 }; 3689 CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args); 3690 } 3691 3692 static void emitForStaticInitCall( 3693 CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId, 3694 llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule, 3695 OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2, 3696 const CGOpenMPRuntime::StaticRTInput &Values) { 3697 if (!CGF.HaveInsertPoint()) 3698 return; 3699 3700 assert(!Values.Ordered); 3701 assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked || 3702 Schedule == OMP_sch_static_balanced_chunked || 3703 Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked || 3704 Schedule == OMP_dist_sch_static || 3705 Schedule == OMP_dist_sch_static_chunked); 3706 3707 // Call __kmpc_for_static_init( 3708 // ident_t *loc, kmp_int32 tid, kmp_int32 schedtype, 3709 // kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower, 3710 // kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride, 3711 // kmp_int[32|64] incr, kmp_int[32|64] chunk); 3712 llvm::Value *Chunk = Values.Chunk; 3713 if (Chunk == nullptr) { 3714 assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static || 3715 Schedule == OMP_dist_sch_static) && 3716 "expected static non-chunked schedule"); 3717 // If the Chunk was not specified in the clause - use default value 1. 3718 Chunk = CGF.Builder.getIntN(Values.IVSize, 1); 3719 } else { 3720 assert((Schedule == OMP_sch_static_chunked || 3721 Schedule == OMP_sch_static_balanced_chunked || 3722 Schedule == OMP_ord_static_chunked || 3723 Schedule == OMP_dist_sch_static_chunked) && 3724 "expected static chunked schedule"); 3725 } 3726 llvm::Value *Args[] = { 3727 UpdateLocation, 3728 ThreadId, 3729 CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1, 3730 M2)), // Schedule type 3731 Values.IL.getPointer(), // &isLastIter 3732 Values.LB.getPointer(), // &LB 3733 Values.UB.getPointer(), // &UB 3734 Values.ST.getPointer(), // &Stride 3735 CGF.Builder.getIntN(Values.IVSize, 1), // Incr 3736 Chunk // Chunk 3737 }; 3738 CGF.EmitRuntimeCall(ForStaticInitFunction, Args); 3739 } 3740 3741 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF, 3742 SourceLocation Loc, 3743 OpenMPDirectiveKind DKind, 3744 const OpenMPScheduleTy &ScheduleKind, 3745 const StaticRTInput &Values) { 3746 OpenMPSchedType ScheduleNum = getRuntimeSchedule( 3747 ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered); 3748 assert(isOpenMPWorksharingDirective(DKind) && 3749 "Expected loop-based or sections-based directive."); 3750 llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc, 3751 isOpenMPLoopDirective(DKind) 3752 ? OMP_IDENT_WORK_LOOP 3753 : OMP_IDENT_WORK_SECTIONS); 3754 llvm::Value *ThreadId = getThreadID(CGF, Loc); 3755 llvm::FunctionCallee StaticInitFunction = 3756 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 3757 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 3758 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 3759 ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values); 3760 } 3761 3762 void CGOpenMPRuntime::emitDistributeStaticInit( 3763 CodeGenFunction &CGF, SourceLocation Loc, 3764 OpenMPDistScheduleClauseKind SchedKind, 3765 const CGOpenMPRuntime::StaticRTInput &Values) { 3766 OpenMPSchedType ScheduleNum = 3767 getRuntimeSchedule(SchedKind, Values.Chunk != nullptr); 3768 llvm::Value *UpdatedLocation = 3769 emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE); 3770 llvm::Value *ThreadId = getThreadID(CGF, Loc); 3771 llvm::FunctionCallee StaticInitFunction = 3772 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 3773 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 3774 ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown, 3775 OMPC_SCHEDULE_MODIFIER_unknown, Values); 3776 } 3777 3778 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF, 3779 SourceLocation Loc, 3780 OpenMPDirectiveKind DKind) { 3781 if (!CGF.HaveInsertPoint()) 3782 return; 3783 // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid); 3784 llvm::Value *Args[] = { 3785 emitUpdateLocation(CGF, Loc, 3786 isOpenMPDistributeDirective(DKind) 3787 ? OMP_IDENT_WORK_DISTRIBUTE 3788 : isOpenMPLoopDirective(DKind) 3789 ? OMP_IDENT_WORK_LOOP 3790 : OMP_IDENT_WORK_SECTIONS), 3791 getThreadID(CGF, Loc)}; 3792 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 3793 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini), 3794 Args); 3795 } 3796 3797 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 3798 SourceLocation Loc, 3799 unsigned IVSize, 3800 bool IVSigned) { 3801 if (!CGF.HaveInsertPoint()) 3802 return; 3803 // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid); 3804 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3805 CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args); 3806 } 3807 3808 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF, 3809 SourceLocation Loc, unsigned IVSize, 3810 bool IVSigned, Address IL, 3811 Address LB, Address UB, 3812 Address ST) { 3813 // Call __kmpc_dispatch_next( 3814 // ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter, 3815 // kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper, 3816 // kmp_int[32|64] *p_stride); 3817 llvm::Value *Args[] = { 3818 emitUpdateLocation(CGF, Loc), 3819 getThreadID(CGF, Loc), 3820 IL.getPointer(), // &isLastIter 3821 LB.getPointer(), // &Lower 3822 UB.getPointer(), // &Upper 3823 ST.getPointer() // &Stride 3824 }; 3825 llvm::Value *Call = 3826 CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args); 3827 return CGF.EmitScalarConversion( 3828 Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1), 3829 CGF.getContext().BoolTy, Loc); 3830 } 3831 3832 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 3833 llvm::Value *NumThreads, 3834 SourceLocation Loc) { 3835 if (!CGF.HaveInsertPoint()) 3836 return; 3837 // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads) 3838 llvm::Value *Args[] = { 3839 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3840 CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)}; 3841 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads), 3842 Args); 3843 } 3844 3845 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF, 3846 ProcBindKind ProcBind, 3847 SourceLocation Loc) { 3848 if (!CGF.HaveInsertPoint()) 3849 return; 3850 assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value."); 3851 // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind) 3852 llvm::Value *Args[] = { 3853 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3854 llvm::ConstantInt::get(CGM.IntTy, unsigned(ProcBind), /*isSigned=*/true)}; 3855 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_proc_bind), Args); 3856 } 3857 3858 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>, 3859 SourceLocation Loc, llvm::AtomicOrdering AO) { 3860 llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder(); 3861 if (OMPBuilder) { 3862 OMPBuilder->CreateFlush(CGF.Builder); 3863 } else { 3864 if (!CGF.HaveInsertPoint()) 3865 return; 3866 // Build call void __kmpc_flush(ident_t *loc) 3867 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush), 3868 emitUpdateLocation(CGF, Loc)); 3869 } 3870 } 3871 3872 namespace { 3873 /// Indexes of fields for type kmp_task_t. 3874 enum KmpTaskTFields { 3875 /// List of shared variables. 3876 KmpTaskTShareds, 3877 /// Task routine. 3878 KmpTaskTRoutine, 3879 /// Partition id for the untied tasks. 3880 KmpTaskTPartId, 3881 /// Function with call of destructors for private variables. 3882 Data1, 3883 /// Task priority. 3884 Data2, 3885 /// (Taskloops only) Lower bound. 3886 KmpTaskTLowerBound, 3887 /// (Taskloops only) Upper bound. 3888 KmpTaskTUpperBound, 3889 /// (Taskloops only) Stride. 3890 KmpTaskTStride, 3891 /// (Taskloops only) Is last iteration flag. 3892 KmpTaskTLastIter, 3893 /// (Taskloops only) Reduction data. 3894 KmpTaskTReductions, 3895 }; 3896 } // anonymous namespace 3897 3898 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const { 3899 return OffloadEntriesTargetRegion.empty() && 3900 OffloadEntriesDeviceGlobalVar.empty(); 3901 } 3902 3903 /// Initialize target region entry. 3904 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3905 initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3906 StringRef ParentName, unsigned LineNum, 3907 unsigned Order) { 3908 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3909 "only required for the device " 3910 "code generation."); 3911 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = 3912 OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr, 3913 OMPTargetRegionEntryTargetRegion); 3914 ++OffloadingEntriesNum; 3915 } 3916 3917 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3918 registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3919 StringRef ParentName, unsigned LineNum, 3920 llvm::Constant *Addr, llvm::Constant *ID, 3921 OMPTargetRegionEntryKind Flags) { 3922 // If we are emitting code for a target, the entry is already initialized, 3923 // only has to be registered. 3924 if (CGM.getLangOpts().OpenMPIsDevice) { 3925 if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) { 3926 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3927 DiagnosticsEngine::Error, 3928 "Unable to find target region on line '%0' in the device code."); 3929 CGM.getDiags().Report(DiagID) << LineNum; 3930 return; 3931 } 3932 auto &Entry = 3933 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum]; 3934 assert(Entry.isValid() && "Entry not initialized!"); 3935 Entry.setAddress(Addr); 3936 Entry.setID(ID); 3937 Entry.setFlags(Flags); 3938 } else { 3939 OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags); 3940 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry; 3941 ++OffloadingEntriesNum; 3942 } 3943 } 3944 3945 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo( 3946 unsigned DeviceID, unsigned FileID, StringRef ParentName, 3947 unsigned LineNum) const { 3948 auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID); 3949 if (PerDevice == OffloadEntriesTargetRegion.end()) 3950 return false; 3951 auto PerFile = PerDevice->second.find(FileID); 3952 if (PerFile == PerDevice->second.end()) 3953 return false; 3954 auto PerParentName = PerFile->second.find(ParentName); 3955 if (PerParentName == PerFile->second.end()) 3956 return false; 3957 auto PerLine = PerParentName->second.find(LineNum); 3958 if (PerLine == PerParentName->second.end()) 3959 return false; 3960 // Fail if this entry is already registered. 3961 if (PerLine->second.getAddress() || PerLine->second.getID()) 3962 return false; 3963 return true; 3964 } 3965 3966 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo( 3967 const OffloadTargetRegionEntryInfoActTy &Action) { 3968 // Scan all target region entries and perform the provided action. 3969 for (const auto &D : OffloadEntriesTargetRegion) 3970 for (const auto &F : D.second) 3971 for (const auto &P : F.second) 3972 for (const auto &L : P.second) 3973 Action(D.first, F.first, P.first(), L.first, L.second); 3974 } 3975 3976 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3977 initializeDeviceGlobalVarEntryInfo(StringRef Name, 3978 OMPTargetGlobalVarEntryKind Flags, 3979 unsigned Order) { 3980 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3981 "only required for the device " 3982 "code generation."); 3983 OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags); 3984 ++OffloadingEntriesNum; 3985 } 3986 3987 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3988 registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr, 3989 CharUnits VarSize, 3990 OMPTargetGlobalVarEntryKind Flags, 3991 llvm::GlobalValue::LinkageTypes Linkage) { 3992 if (CGM.getLangOpts().OpenMPIsDevice) { 3993 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 3994 assert(Entry.isValid() && Entry.getFlags() == Flags && 3995 "Entry not initialized!"); 3996 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 3997 "Resetting with the new address."); 3998 if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) { 3999 if (Entry.getVarSize().isZero()) { 4000 Entry.setVarSize(VarSize); 4001 Entry.setLinkage(Linkage); 4002 } 4003 return; 4004 } 4005 Entry.setVarSize(VarSize); 4006 Entry.setLinkage(Linkage); 4007 Entry.setAddress(Addr); 4008 } else { 4009 if (hasDeviceGlobalVarEntryInfo(VarName)) { 4010 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 4011 assert(Entry.isValid() && Entry.getFlags() == Flags && 4012 "Entry not initialized!"); 4013 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 4014 "Resetting with the new address."); 4015 if (Entry.getVarSize().isZero()) { 4016 Entry.setVarSize(VarSize); 4017 Entry.setLinkage(Linkage); 4018 } 4019 return; 4020 } 4021 OffloadEntriesDeviceGlobalVar.try_emplace( 4022 VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage); 4023 ++OffloadingEntriesNum; 4024 } 4025 } 4026 4027 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 4028 actOnDeviceGlobalVarEntriesInfo( 4029 const OffloadDeviceGlobalVarEntryInfoActTy &Action) { 4030 // Scan all target region entries and perform the provided action. 4031 for (const auto &E : OffloadEntriesDeviceGlobalVar) 4032 Action(E.getKey(), E.getValue()); 4033 } 4034 4035 void CGOpenMPRuntime::createOffloadEntry( 4036 llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags, 4037 llvm::GlobalValue::LinkageTypes Linkage) { 4038 StringRef Name = Addr->getName(); 4039 llvm::Module &M = CGM.getModule(); 4040 llvm::LLVMContext &C = M.getContext(); 4041 4042 // Create constant string with the name. 4043 llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name); 4044 4045 std::string StringName = getName({"omp_offloading", "entry_name"}); 4046 auto *Str = new llvm::GlobalVariable( 4047 M, StrPtrInit->getType(), /*isConstant=*/true, 4048 llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName); 4049 Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 4050 4051 llvm::Constant *Data[] = {llvm::ConstantExpr::getBitCast(ID, CGM.VoidPtrTy), 4052 llvm::ConstantExpr::getBitCast(Str, CGM.Int8PtrTy), 4053 llvm::ConstantInt::get(CGM.SizeTy, Size), 4054 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 4055 llvm::ConstantInt::get(CGM.Int32Ty, 0)}; 4056 std::string EntryName = getName({"omp_offloading", "entry", ""}); 4057 llvm::GlobalVariable *Entry = createGlobalStruct( 4058 CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data, 4059 Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage); 4060 4061 // The entry has to be created in the section the linker expects it to be. 4062 Entry->setSection("omp_offloading_entries"); 4063 } 4064 4065 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() { 4066 // Emit the offloading entries and metadata so that the device codegen side 4067 // can easily figure out what to emit. The produced metadata looks like 4068 // this: 4069 // 4070 // !omp_offload.info = !{!1, ...} 4071 // 4072 // Right now we only generate metadata for function that contain target 4073 // regions. 4074 4075 // If we are in simd mode or there are no entries, we don't need to do 4076 // anything. 4077 if (CGM.getLangOpts().OpenMPSimd || OffloadEntriesInfoManager.empty()) 4078 return; 4079 4080 llvm::Module &M = CGM.getModule(); 4081 llvm::LLVMContext &C = M.getContext(); 4082 SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *, 4083 SourceLocation, StringRef>, 4084 16> 4085 OrderedEntries(OffloadEntriesInfoManager.size()); 4086 llvm::SmallVector<StringRef, 16> ParentFunctions( 4087 OffloadEntriesInfoManager.size()); 4088 4089 // Auxiliary methods to create metadata values and strings. 4090 auto &&GetMDInt = [this](unsigned V) { 4091 return llvm::ConstantAsMetadata::get( 4092 llvm::ConstantInt::get(CGM.Int32Ty, V)); 4093 }; 4094 4095 auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); }; 4096 4097 // Create the offloading info metadata node. 4098 llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info"); 4099 4100 // Create function that emits metadata for each target region entry; 4101 auto &&TargetRegionMetadataEmitter = 4102 [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt, 4103 &GetMDString]( 4104 unsigned DeviceID, unsigned FileID, StringRef ParentName, 4105 unsigned Line, 4106 const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) { 4107 // Generate metadata for target regions. Each entry of this metadata 4108 // contains: 4109 // - Entry 0 -> Kind of this type of metadata (0). 4110 // - Entry 1 -> Device ID of the file where the entry was identified. 4111 // - Entry 2 -> File ID of the file where the entry was identified. 4112 // - Entry 3 -> Mangled name of the function where the entry was 4113 // identified. 4114 // - Entry 4 -> Line in the file where the entry was identified. 4115 // - Entry 5 -> Order the entry was created. 4116 // The first element of the metadata node is the kind. 4117 llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID), 4118 GetMDInt(FileID), GetMDString(ParentName), 4119 GetMDInt(Line), GetMDInt(E.getOrder())}; 4120 4121 SourceLocation Loc; 4122 for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(), 4123 E = CGM.getContext().getSourceManager().fileinfo_end(); 4124 I != E; ++I) { 4125 if (I->getFirst()->getUniqueID().getDevice() == DeviceID && 4126 I->getFirst()->getUniqueID().getFile() == FileID) { 4127 Loc = CGM.getContext().getSourceManager().translateFileLineCol( 4128 I->getFirst(), Line, 1); 4129 break; 4130 } 4131 } 4132 // Save this entry in the right position of the ordered entries array. 4133 OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName); 4134 ParentFunctions[E.getOrder()] = ParentName; 4135 4136 // Add metadata to the named metadata node. 4137 MD->addOperand(llvm::MDNode::get(C, Ops)); 4138 }; 4139 4140 OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo( 4141 TargetRegionMetadataEmitter); 4142 4143 // Create function that emits metadata for each device global variable entry; 4144 auto &&DeviceGlobalVarMetadataEmitter = 4145 [&C, &OrderedEntries, &GetMDInt, &GetMDString, 4146 MD](StringRef MangledName, 4147 const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar 4148 &E) { 4149 // Generate metadata for global variables. Each entry of this metadata 4150 // contains: 4151 // - Entry 0 -> Kind of this type of metadata (1). 4152 // - Entry 1 -> Mangled name of the variable. 4153 // - Entry 2 -> Declare target kind. 4154 // - Entry 3 -> Order the entry was created. 4155 // The first element of the metadata node is the kind. 4156 llvm::Metadata *Ops[] = { 4157 GetMDInt(E.getKind()), GetMDString(MangledName), 4158 GetMDInt(E.getFlags()), GetMDInt(E.getOrder())}; 4159 4160 // Save this entry in the right position of the ordered entries array. 4161 OrderedEntries[E.getOrder()] = 4162 std::make_tuple(&E, SourceLocation(), MangledName); 4163 4164 // Add metadata to the named metadata node. 4165 MD->addOperand(llvm::MDNode::get(C, Ops)); 4166 }; 4167 4168 OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo( 4169 DeviceGlobalVarMetadataEmitter); 4170 4171 for (const auto &E : OrderedEntries) { 4172 assert(std::get<0>(E) && "All ordered entries must exist!"); 4173 if (const auto *CE = 4174 dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>( 4175 std::get<0>(E))) { 4176 if (!CE->getID() || !CE->getAddress()) { 4177 // Do not blame the entry if the parent funtion is not emitted. 4178 StringRef FnName = ParentFunctions[CE->getOrder()]; 4179 if (!CGM.GetGlobalValue(FnName)) 4180 continue; 4181 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4182 DiagnosticsEngine::Error, 4183 "Offloading entry for target region in %0 is incorrect: either the " 4184 "address or the ID is invalid."); 4185 CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName; 4186 continue; 4187 } 4188 createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0, 4189 CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage); 4190 } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy:: 4191 OffloadEntryInfoDeviceGlobalVar>( 4192 std::get<0>(E))) { 4193 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags = 4194 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 4195 CE->getFlags()); 4196 switch (Flags) { 4197 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: { 4198 if (CGM.getLangOpts().OpenMPIsDevice && 4199 CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory()) 4200 continue; 4201 if (!CE->getAddress()) { 4202 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4203 DiagnosticsEngine::Error, "Offloading entry for declare target " 4204 "variable %0 is incorrect: the " 4205 "address is invalid."); 4206 CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E); 4207 continue; 4208 } 4209 // The vaiable has no definition - no need to add the entry. 4210 if (CE->getVarSize().isZero()) 4211 continue; 4212 break; 4213 } 4214 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink: 4215 assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) || 4216 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) && 4217 "Declaret target link address is set."); 4218 if (CGM.getLangOpts().OpenMPIsDevice) 4219 continue; 4220 if (!CE->getAddress()) { 4221 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4222 DiagnosticsEngine::Error, 4223 "Offloading entry for declare target variable is incorrect: the " 4224 "address is invalid."); 4225 CGM.getDiags().Report(DiagID); 4226 continue; 4227 } 4228 break; 4229 } 4230 createOffloadEntry(CE->getAddress(), CE->getAddress(), 4231 CE->getVarSize().getQuantity(), Flags, 4232 CE->getLinkage()); 4233 } else { 4234 llvm_unreachable("Unsupported entry kind."); 4235 } 4236 } 4237 } 4238 4239 /// Loads all the offload entries information from the host IR 4240 /// metadata. 4241 void CGOpenMPRuntime::loadOffloadInfoMetadata() { 4242 // If we are in target mode, load the metadata from the host IR. This code has 4243 // to match the metadaata creation in createOffloadEntriesAndInfoMetadata(). 4244 4245 if (!CGM.getLangOpts().OpenMPIsDevice) 4246 return; 4247 4248 if (CGM.getLangOpts().OMPHostIRFile.empty()) 4249 return; 4250 4251 auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile); 4252 if (auto EC = Buf.getError()) { 4253 CGM.getDiags().Report(diag::err_cannot_open_file) 4254 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 4255 return; 4256 } 4257 4258 llvm::LLVMContext C; 4259 auto ME = expectedToErrorOrAndEmitErrors( 4260 C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C)); 4261 4262 if (auto EC = ME.getError()) { 4263 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4264 DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'"); 4265 CGM.getDiags().Report(DiagID) 4266 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 4267 return; 4268 } 4269 4270 llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info"); 4271 if (!MD) 4272 return; 4273 4274 for (llvm::MDNode *MN : MD->operands()) { 4275 auto &&GetMDInt = [MN](unsigned Idx) { 4276 auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx)); 4277 return cast<llvm::ConstantInt>(V->getValue())->getZExtValue(); 4278 }; 4279 4280 auto &&GetMDString = [MN](unsigned Idx) { 4281 auto *V = cast<llvm::MDString>(MN->getOperand(Idx)); 4282 return V->getString(); 4283 }; 4284 4285 switch (GetMDInt(0)) { 4286 default: 4287 llvm_unreachable("Unexpected metadata!"); 4288 break; 4289 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 4290 OffloadingEntryInfoTargetRegion: 4291 OffloadEntriesInfoManager.initializeTargetRegionEntryInfo( 4292 /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2), 4293 /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4), 4294 /*Order=*/GetMDInt(5)); 4295 break; 4296 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 4297 OffloadingEntryInfoDeviceGlobalVar: 4298 OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo( 4299 /*MangledName=*/GetMDString(1), 4300 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 4301 /*Flags=*/GetMDInt(2)), 4302 /*Order=*/GetMDInt(3)); 4303 break; 4304 } 4305 } 4306 } 4307 4308 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) { 4309 if (!KmpRoutineEntryPtrTy) { 4310 // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type. 4311 ASTContext &C = CGM.getContext(); 4312 QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy}; 4313 FunctionProtoType::ExtProtoInfo EPI; 4314 KmpRoutineEntryPtrQTy = C.getPointerType( 4315 C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI)); 4316 KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy); 4317 } 4318 } 4319 4320 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() { 4321 // Make sure the type of the entry is already created. This is the type we 4322 // have to create: 4323 // struct __tgt_offload_entry{ 4324 // void *addr; // Pointer to the offload entry info. 4325 // // (function or global) 4326 // char *name; // Name of the function or global. 4327 // size_t size; // Size of the entry info (0 if it a function). 4328 // int32_t flags; // Flags associated with the entry, e.g. 'link'. 4329 // int32_t reserved; // Reserved, to use by the runtime library. 4330 // }; 4331 if (TgtOffloadEntryQTy.isNull()) { 4332 ASTContext &C = CGM.getContext(); 4333 RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry"); 4334 RD->startDefinition(); 4335 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4336 addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy)); 4337 addFieldToRecordDecl(C, RD, C.getSizeType()); 4338 addFieldToRecordDecl( 4339 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4340 addFieldToRecordDecl( 4341 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4342 RD->completeDefinition(); 4343 RD->addAttr(PackedAttr::CreateImplicit(C)); 4344 TgtOffloadEntryQTy = C.getRecordType(RD); 4345 } 4346 return TgtOffloadEntryQTy; 4347 } 4348 4349 namespace { 4350 struct PrivateHelpersTy { 4351 PrivateHelpersTy(const Expr *OriginalRef, const VarDecl *Original, 4352 const VarDecl *PrivateCopy, const VarDecl *PrivateElemInit) 4353 : OriginalRef(OriginalRef), Original(Original), PrivateCopy(PrivateCopy), 4354 PrivateElemInit(PrivateElemInit) {} 4355 const Expr *OriginalRef = nullptr; 4356 const VarDecl *Original = nullptr; 4357 const VarDecl *PrivateCopy = nullptr; 4358 const VarDecl *PrivateElemInit = nullptr; 4359 }; 4360 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy; 4361 } // anonymous namespace 4362 4363 static RecordDecl * 4364 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) { 4365 if (!Privates.empty()) { 4366 ASTContext &C = CGM.getContext(); 4367 // Build struct .kmp_privates_t. { 4368 // /* private vars */ 4369 // }; 4370 RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t"); 4371 RD->startDefinition(); 4372 for (const auto &Pair : Privates) { 4373 const VarDecl *VD = Pair.second.Original; 4374 QualType Type = VD->getType().getNonReferenceType(); 4375 FieldDecl *FD = addFieldToRecordDecl(C, RD, Type); 4376 if (VD->hasAttrs()) { 4377 for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()), 4378 E(VD->getAttrs().end()); 4379 I != E; ++I) 4380 FD->addAttr(*I); 4381 } 4382 } 4383 RD->completeDefinition(); 4384 return RD; 4385 } 4386 return nullptr; 4387 } 4388 4389 static RecordDecl * 4390 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind, 4391 QualType KmpInt32Ty, 4392 QualType KmpRoutineEntryPointerQTy) { 4393 ASTContext &C = CGM.getContext(); 4394 // Build struct kmp_task_t { 4395 // void * shareds; 4396 // kmp_routine_entry_t routine; 4397 // kmp_int32 part_id; 4398 // kmp_cmplrdata_t data1; 4399 // kmp_cmplrdata_t data2; 4400 // For taskloops additional fields: 4401 // kmp_uint64 lb; 4402 // kmp_uint64 ub; 4403 // kmp_int64 st; 4404 // kmp_int32 liter; 4405 // void * reductions; 4406 // }; 4407 RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union); 4408 UD->startDefinition(); 4409 addFieldToRecordDecl(C, UD, KmpInt32Ty); 4410 addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy); 4411 UD->completeDefinition(); 4412 QualType KmpCmplrdataTy = C.getRecordType(UD); 4413 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t"); 4414 RD->startDefinition(); 4415 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4416 addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy); 4417 addFieldToRecordDecl(C, RD, KmpInt32Ty); 4418 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 4419 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 4420 if (isOpenMPTaskLoopDirective(Kind)) { 4421 QualType KmpUInt64Ty = 4422 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0); 4423 QualType KmpInt64Ty = 4424 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 4425 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 4426 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 4427 addFieldToRecordDecl(C, RD, KmpInt64Ty); 4428 addFieldToRecordDecl(C, RD, KmpInt32Ty); 4429 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4430 } 4431 RD->completeDefinition(); 4432 return RD; 4433 } 4434 4435 static RecordDecl * 4436 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy, 4437 ArrayRef<PrivateDataTy> Privates) { 4438 ASTContext &C = CGM.getContext(); 4439 // Build struct kmp_task_t_with_privates { 4440 // kmp_task_t task_data; 4441 // .kmp_privates_t. privates; 4442 // }; 4443 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates"); 4444 RD->startDefinition(); 4445 addFieldToRecordDecl(C, RD, KmpTaskTQTy); 4446 if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates)) 4447 addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD)); 4448 RD->completeDefinition(); 4449 return RD; 4450 } 4451 4452 /// Emit a proxy function which accepts kmp_task_t as the second 4453 /// argument. 4454 /// \code 4455 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) { 4456 /// TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt, 4457 /// For taskloops: 4458 /// tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 4459 /// tt->reductions, tt->shareds); 4460 /// return 0; 4461 /// } 4462 /// \endcode 4463 static llvm::Function * 4464 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc, 4465 OpenMPDirectiveKind Kind, QualType KmpInt32Ty, 4466 QualType KmpTaskTWithPrivatesPtrQTy, 4467 QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy, 4468 QualType SharedsPtrTy, llvm::Function *TaskFunction, 4469 llvm::Value *TaskPrivatesMap) { 4470 ASTContext &C = CGM.getContext(); 4471 FunctionArgList Args; 4472 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 4473 ImplicitParamDecl::Other); 4474 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4475 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 4476 ImplicitParamDecl::Other); 4477 Args.push_back(&GtidArg); 4478 Args.push_back(&TaskTypeArg); 4479 const auto &TaskEntryFnInfo = 4480 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 4481 llvm::FunctionType *TaskEntryTy = 4482 CGM.getTypes().GetFunctionType(TaskEntryFnInfo); 4483 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""}); 4484 auto *TaskEntry = llvm::Function::Create( 4485 TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 4486 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo); 4487 TaskEntry->setDoesNotRecurse(); 4488 CodeGenFunction CGF(CGM); 4489 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args, 4490 Loc, Loc); 4491 4492 // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map, 4493 // tt, 4494 // For taskloops: 4495 // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 4496 // tt->task_data.shareds); 4497 llvm::Value *GtidParam = CGF.EmitLoadOfScalar( 4498 CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc); 4499 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4500 CGF.GetAddrOfLocalVar(&TaskTypeArg), 4501 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4502 const auto *KmpTaskTWithPrivatesQTyRD = 4503 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 4504 LValue Base = 4505 CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4506 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 4507 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 4508 LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI); 4509 llvm::Value *PartidParam = PartIdLVal.getPointer(CGF); 4510 4511 auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds); 4512 LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI); 4513 llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4514 CGF.EmitLoadOfScalar(SharedsLVal, Loc), 4515 CGF.ConvertTypeForMem(SharedsPtrTy)); 4516 4517 auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 4518 llvm::Value *PrivatesParam; 4519 if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) { 4520 LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI); 4521 PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4522 PrivatesLVal.getPointer(CGF), CGF.VoidPtrTy); 4523 } else { 4524 PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 4525 } 4526 4527 llvm::Value *CommonArgs[] = {GtidParam, PartidParam, PrivatesParam, 4528 TaskPrivatesMap, 4529 CGF.Builder 4530 .CreatePointerBitCastOrAddrSpaceCast( 4531 TDBase.getAddress(CGF), CGF.VoidPtrTy) 4532 .getPointer()}; 4533 SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs), 4534 std::end(CommonArgs)); 4535 if (isOpenMPTaskLoopDirective(Kind)) { 4536 auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound); 4537 LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI); 4538 llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc); 4539 auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound); 4540 LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI); 4541 llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc); 4542 auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride); 4543 LValue StLVal = CGF.EmitLValueForField(Base, *StFI); 4544 llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc); 4545 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 4546 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 4547 llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc); 4548 auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions); 4549 LValue RLVal = CGF.EmitLValueForField(Base, *RFI); 4550 llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc); 4551 CallArgs.push_back(LBParam); 4552 CallArgs.push_back(UBParam); 4553 CallArgs.push_back(StParam); 4554 CallArgs.push_back(LIParam); 4555 CallArgs.push_back(RParam); 4556 } 4557 CallArgs.push_back(SharedsParam); 4558 4559 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction, 4560 CallArgs); 4561 CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)), 4562 CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty)); 4563 CGF.FinishFunction(); 4564 return TaskEntry; 4565 } 4566 4567 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM, 4568 SourceLocation Loc, 4569 QualType KmpInt32Ty, 4570 QualType KmpTaskTWithPrivatesPtrQTy, 4571 QualType KmpTaskTWithPrivatesQTy) { 4572 ASTContext &C = CGM.getContext(); 4573 FunctionArgList Args; 4574 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 4575 ImplicitParamDecl::Other); 4576 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4577 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 4578 ImplicitParamDecl::Other); 4579 Args.push_back(&GtidArg); 4580 Args.push_back(&TaskTypeArg); 4581 const auto &DestructorFnInfo = 4582 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 4583 llvm::FunctionType *DestructorFnTy = 4584 CGM.getTypes().GetFunctionType(DestructorFnInfo); 4585 std::string Name = 4586 CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""}); 4587 auto *DestructorFn = 4588 llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage, 4589 Name, &CGM.getModule()); 4590 CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn, 4591 DestructorFnInfo); 4592 DestructorFn->setDoesNotRecurse(); 4593 CodeGenFunction CGF(CGM); 4594 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo, 4595 Args, Loc, Loc); 4596 4597 LValue Base = CGF.EmitLoadOfPointerLValue( 4598 CGF.GetAddrOfLocalVar(&TaskTypeArg), 4599 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4600 const auto *KmpTaskTWithPrivatesQTyRD = 4601 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 4602 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4603 Base = CGF.EmitLValueForField(Base, *FI); 4604 for (const auto *Field : 4605 cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) { 4606 if (QualType::DestructionKind DtorKind = 4607 Field->getType().isDestructedType()) { 4608 LValue FieldLValue = CGF.EmitLValueForField(Base, Field); 4609 CGF.pushDestroy(DtorKind, FieldLValue.getAddress(CGF), Field->getType()); 4610 } 4611 } 4612 CGF.FinishFunction(); 4613 return DestructorFn; 4614 } 4615 4616 /// Emit a privates mapping function for correct handling of private and 4617 /// firstprivate variables. 4618 /// \code 4619 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1> 4620 /// **noalias priv1,..., <tyn> **noalias privn) { 4621 /// *priv1 = &.privates.priv1; 4622 /// ...; 4623 /// *privn = &.privates.privn; 4624 /// } 4625 /// \endcode 4626 static llvm::Value * 4627 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc, 4628 ArrayRef<const Expr *> PrivateVars, 4629 ArrayRef<const Expr *> FirstprivateVars, 4630 ArrayRef<const Expr *> LastprivateVars, 4631 QualType PrivatesQTy, 4632 ArrayRef<PrivateDataTy> Privates) { 4633 ASTContext &C = CGM.getContext(); 4634 FunctionArgList Args; 4635 ImplicitParamDecl TaskPrivatesArg( 4636 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4637 C.getPointerType(PrivatesQTy).withConst().withRestrict(), 4638 ImplicitParamDecl::Other); 4639 Args.push_back(&TaskPrivatesArg); 4640 llvm::DenseMap<const VarDecl *, unsigned> PrivateVarsPos; 4641 unsigned Counter = 1; 4642 for (const Expr *E : PrivateVars) { 4643 Args.push_back(ImplicitParamDecl::Create( 4644 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4645 C.getPointerType(C.getPointerType(E->getType())) 4646 .withConst() 4647 .withRestrict(), 4648 ImplicitParamDecl::Other)); 4649 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4650 PrivateVarsPos[VD] = Counter; 4651 ++Counter; 4652 } 4653 for (const Expr *E : FirstprivateVars) { 4654 Args.push_back(ImplicitParamDecl::Create( 4655 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4656 C.getPointerType(C.getPointerType(E->getType())) 4657 .withConst() 4658 .withRestrict(), 4659 ImplicitParamDecl::Other)); 4660 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4661 PrivateVarsPos[VD] = Counter; 4662 ++Counter; 4663 } 4664 for (const Expr *E : LastprivateVars) { 4665 Args.push_back(ImplicitParamDecl::Create( 4666 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4667 C.getPointerType(C.getPointerType(E->getType())) 4668 .withConst() 4669 .withRestrict(), 4670 ImplicitParamDecl::Other)); 4671 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4672 PrivateVarsPos[VD] = Counter; 4673 ++Counter; 4674 } 4675 const auto &TaskPrivatesMapFnInfo = 4676 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4677 llvm::FunctionType *TaskPrivatesMapTy = 4678 CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo); 4679 std::string Name = 4680 CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""}); 4681 auto *TaskPrivatesMap = llvm::Function::Create( 4682 TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name, 4683 &CGM.getModule()); 4684 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap, 4685 TaskPrivatesMapFnInfo); 4686 if (CGM.getLangOpts().Optimize) { 4687 TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline); 4688 TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone); 4689 TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline); 4690 } 4691 CodeGenFunction CGF(CGM); 4692 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap, 4693 TaskPrivatesMapFnInfo, Args, Loc, Loc); 4694 4695 // *privi = &.privates.privi; 4696 LValue Base = CGF.EmitLoadOfPointerLValue( 4697 CGF.GetAddrOfLocalVar(&TaskPrivatesArg), 4698 TaskPrivatesArg.getType()->castAs<PointerType>()); 4699 const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl()); 4700 Counter = 0; 4701 for (const FieldDecl *Field : PrivatesQTyRD->fields()) { 4702 LValue FieldLVal = CGF.EmitLValueForField(Base, Field); 4703 const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]]; 4704 LValue RefLVal = 4705 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType()); 4706 LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue( 4707 RefLVal.getAddress(CGF), RefLVal.getType()->castAs<PointerType>()); 4708 CGF.EmitStoreOfScalar(FieldLVal.getPointer(CGF), RefLoadLVal); 4709 ++Counter; 4710 } 4711 CGF.FinishFunction(); 4712 return TaskPrivatesMap; 4713 } 4714 4715 /// Emit initialization for private variables in task-based directives. 4716 static void emitPrivatesInit(CodeGenFunction &CGF, 4717 const OMPExecutableDirective &D, 4718 Address KmpTaskSharedsPtr, LValue TDBase, 4719 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 4720 QualType SharedsTy, QualType SharedsPtrTy, 4721 const OMPTaskDataTy &Data, 4722 ArrayRef<PrivateDataTy> Privates, bool ForDup) { 4723 ASTContext &C = CGF.getContext(); 4724 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4725 LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI); 4726 OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind()) 4727 ? OMPD_taskloop 4728 : OMPD_task; 4729 const CapturedStmt &CS = *D.getCapturedStmt(Kind); 4730 CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS); 4731 LValue SrcBase; 4732 bool IsTargetTask = 4733 isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) || 4734 isOpenMPTargetExecutionDirective(D.getDirectiveKind()); 4735 // For target-based directives skip 3 firstprivate arrays BasePointersArray, 4736 // PointersArray and SizesArray. The original variables for these arrays are 4737 // not captured and we get their addresses explicitly. 4738 if ((!IsTargetTask && !Data.FirstprivateVars.empty() && ForDup) || 4739 (IsTargetTask && KmpTaskSharedsPtr.isValid())) { 4740 SrcBase = CGF.MakeAddrLValue( 4741 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4742 KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)), 4743 SharedsTy); 4744 } 4745 FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin(); 4746 for (const PrivateDataTy &Pair : Privates) { 4747 const VarDecl *VD = Pair.second.PrivateCopy; 4748 const Expr *Init = VD->getAnyInitializer(); 4749 if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) && 4750 !CGF.isTrivialInitializer(Init)))) { 4751 LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI); 4752 if (const VarDecl *Elem = Pair.second.PrivateElemInit) { 4753 const VarDecl *OriginalVD = Pair.second.Original; 4754 // Check if the variable is the target-based BasePointersArray, 4755 // PointersArray or SizesArray. 4756 LValue SharedRefLValue; 4757 QualType Type = PrivateLValue.getType(); 4758 const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD); 4759 if (IsTargetTask && !SharedField) { 4760 assert(isa<ImplicitParamDecl>(OriginalVD) && 4761 isa<CapturedDecl>(OriginalVD->getDeclContext()) && 4762 cast<CapturedDecl>(OriginalVD->getDeclContext()) 4763 ->getNumParams() == 0 && 4764 isa<TranslationUnitDecl>( 4765 cast<CapturedDecl>(OriginalVD->getDeclContext()) 4766 ->getDeclContext()) && 4767 "Expected artificial target data variable."); 4768 SharedRefLValue = 4769 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type); 4770 } else if (ForDup) { 4771 SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField); 4772 SharedRefLValue = CGF.MakeAddrLValue( 4773 Address(SharedRefLValue.getPointer(CGF), 4774 C.getDeclAlign(OriginalVD)), 4775 SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl), 4776 SharedRefLValue.getTBAAInfo()); 4777 } else { 4778 InlinedOpenMPRegionRAII Region( 4779 CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown, 4780 /*HasCancel=*/false); 4781 SharedRefLValue = CGF.EmitLValue(Pair.second.OriginalRef); 4782 } 4783 if (Type->isArrayType()) { 4784 // Initialize firstprivate array. 4785 if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) { 4786 // Perform simple memcpy. 4787 CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type); 4788 } else { 4789 // Initialize firstprivate array using element-by-element 4790 // initialization. 4791 CGF.EmitOMPAggregateAssign( 4792 PrivateLValue.getAddress(CGF), SharedRefLValue.getAddress(CGF), 4793 Type, 4794 [&CGF, Elem, Init, &CapturesInfo](Address DestElement, 4795 Address SrcElement) { 4796 // Clean up any temporaries needed by the initialization. 4797 CodeGenFunction::OMPPrivateScope InitScope(CGF); 4798 InitScope.addPrivate( 4799 Elem, [SrcElement]() -> Address { return SrcElement; }); 4800 (void)InitScope.Privatize(); 4801 // Emit initialization for single element. 4802 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII( 4803 CGF, &CapturesInfo); 4804 CGF.EmitAnyExprToMem(Init, DestElement, 4805 Init->getType().getQualifiers(), 4806 /*IsInitializer=*/false); 4807 }); 4808 } 4809 } else { 4810 CodeGenFunction::OMPPrivateScope InitScope(CGF); 4811 InitScope.addPrivate(Elem, [SharedRefLValue, &CGF]() -> Address { 4812 return SharedRefLValue.getAddress(CGF); 4813 }); 4814 (void)InitScope.Privatize(); 4815 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo); 4816 CGF.EmitExprAsInit(Init, VD, PrivateLValue, 4817 /*capturedByInit=*/false); 4818 } 4819 } else { 4820 CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false); 4821 } 4822 } 4823 ++FI; 4824 } 4825 } 4826 4827 /// Check if duplication function is required for taskloops. 4828 static bool checkInitIsRequired(CodeGenFunction &CGF, 4829 ArrayRef<PrivateDataTy> Privates) { 4830 bool InitRequired = false; 4831 for (const PrivateDataTy &Pair : Privates) { 4832 const VarDecl *VD = Pair.second.PrivateCopy; 4833 const Expr *Init = VD->getAnyInitializer(); 4834 InitRequired = InitRequired || (Init && isa<CXXConstructExpr>(Init) && 4835 !CGF.isTrivialInitializer(Init)); 4836 if (InitRequired) 4837 break; 4838 } 4839 return InitRequired; 4840 } 4841 4842 4843 /// Emit task_dup function (for initialization of 4844 /// private/firstprivate/lastprivate vars and last_iter flag) 4845 /// \code 4846 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int 4847 /// lastpriv) { 4848 /// // setup lastprivate flag 4849 /// task_dst->last = lastpriv; 4850 /// // could be constructor calls here... 4851 /// } 4852 /// \endcode 4853 static llvm::Value * 4854 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc, 4855 const OMPExecutableDirective &D, 4856 QualType KmpTaskTWithPrivatesPtrQTy, 4857 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 4858 const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy, 4859 QualType SharedsPtrTy, const OMPTaskDataTy &Data, 4860 ArrayRef<PrivateDataTy> Privates, bool WithLastIter) { 4861 ASTContext &C = CGM.getContext(); 4862 FunctionArgList Args; 4863 ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4864 KmpTaskTWithPrivatesPtrQTy, 4865 ImplicitParamDecl::Other); 4866 ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4867 KmpTaskTWithPrivatesPtrQTy, 4868 ImplicitParamDecl::Other); 4869 ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy, 4870 ImplicitParamDecl::Other); 4871 Args.push_back(&DstArg); 4872 Args.push_back(&SrcArg); 4873 Args.push_back(&LastprivArg); 4874 const auto &TaskDupFnInfo = 4875 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4876 llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo); 4877 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""}); 4878 auto *TaskDup = llvm::Function::Create( 4879 TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 4880 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo); 4881 TaskDup->setDoesNotRecurse(); 4882 CodeGenFunction CGF(CGM); 4883 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc, 4884 Loc); 4885 4886 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4887 CGF.GetAddrOfLocalVar(&DstArg), 4888 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4889 // task_dst->liter = lastpriv; 4890 if (WithLastIter) { 4891 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 4892 LValue Base = CGF.EmitLValueForField( 4893 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4894 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 4895 llvm::Value *Lastpriv = CGF.EmitLoadOfScalar( 4896 CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc); 4897 CGF.EmitStoreOfScalar(Lastpriv, LILVal); 4898 } 4899 4900 // Emit initial values for private copies (if any). 4901 assert(!Privates.empty()); 4902 Address KmpTaskSharedsPtr = Address::invalid(); 4903 if (!Data.FirstprivateVars.empty()) { 4904 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4905 CGF.GetAddrOfLocalVar(&SrcArg), 4906 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4907 LValue Base = CGF.EmitLValueForField( 4908 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4909 KmpTaskSharedsPtr = Address( 4910 CGF.EmitLoadOfScalar(CGF.EmitLValueForField( 4911 Base, *std::next(KmpTaskTQTyRD->field_begin(), 4912 KmpTaskTShareds)), 4913 Loc), 4914 CGF.getNaturalTypeAlignment(SharedsTy)); 4915 } 4916 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD, 4917 SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true); 4918 CGF.FinishFunction(); 4919 return TaskDup; 4920 } 4921 4922 /// Checks if destructor function is required to be generated. 4923 /// \return true if cleanups are required, false otherwise. 4924 static bool 4925 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD) { 4926 bool NeedsCleanup = false; 4927 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 4928 const auto *PrivateRD = cast<RecordDecl>(FI->getType()->getAsTagDecl()); 4929 for (const FieldDecl *FD : PrivateRD->fields()) { 4930 NeedsCleanup = NeedsCleanup || FD->getType().isDestructedType(); 4931 if (NeedsCleanup) 4932 break; 4933 } 4934 return NeedsCleanup; 4935 } 4936 4937 CGOpenMPRuntime::TaskResultTy 4938 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc, 4939 const OMPExecutableDirective &D, 4940 llvm::Function *TaskFunction, QualType SharedsTy, 4941 Address Shareds, const OMPTaskDataTy &Data) { 4942 ASTContext &C = CGM.getContext(); 4943 llvm::SmallVector<PrivateDataTy, 4> Privates; 4944 // Aggregate privates and sort them by the alignment. 4945 const auto *I = Data.PrivateCopies.begin(); 4946 for (const Expr *E : Data.PrivateVars) { 4947 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4948 Privates.emplace_back( 4949 C.getDeclAlign(VD), 4950 PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4951 /*PrivateElemInit=*/nullptr)); 4952 ++I; 4953 } 4954 I = Data.FirstprivateCopies.begin(); 4955 const auto *IElemInitRef = Data.FirstprivateInits.begin(); 4956 for (const Expr *E : Data.FirstprivateVars) { 4957 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4958 Privates.emplace_back( 4959 C.getDeclAlign(VD), 4960 PrivateHelpersTy( 4961 E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4962 cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl()))); 4963 ++I; 4964 ++IElemInitRef; 4965 } 4966 I = Data.LastprivateCopies.begin(); 4967 for (const Expr *E : Data.LastprivateVars) { 4968 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4969 Privates.emplace_back( 4970 C.getDeclAlign(VD), 4971 PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4972 /*PrivateElemInit=*/nullptr)); 4973 ++I; 4974 } 4975 llvm::stable_sort(Privates, [](PrivateDataTy L, PrivateDataTy R) { 4976 return L.first > R.first; 4977 }); 4978 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 4979 // Build type kmp_routine_entry_t (if not built yet). 4980 emitKmpRoutineEntryT(KmpInt32Ty); 4981 // Build type kmp_task_t (if not built yet). 4982 if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) { 4983 if (SavedKmpTaskloopTQTy.isNull()) { 4984 SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl( 4985 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 4986 } 4987 KmpTaskTQTy = SavedKmpTaskloopTQTy; 4988 } else { 4989 assert((D.getDirectiveKind() == OMPD_task || 4990 isOpenMPTargetExecutionDirective(D.getDirectiveKind()) || 4991 isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) && 4992 "Expected taskloop, task or target directive"); 4993 if (SavedKmpTaskTQTy.isNull()) { 4994 SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl( 4995 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 4996 } 4997 KmpTaskTQTy = SavedKmpTaskTQTy; 4998 } 4999 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 5000 // Build particular struct kmp_task_t for the given task. 5001 const RecordDecl *KmpTaskTWithPrivatesQTyRD = 5002 createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates); 5003 QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD); 5004 QualType KmpTaskTWithPrivatesPtrQTy = 5005 C.getPointerType(KmpTaskTWithPrivatesQTy); 5006 llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy); 5007 llvm::Type *KmpTaskTWithPrivatesPtrTy = 5008 KmpTaskTWithPrivatesTy->getPointerTo(); 5009 llvm::Value *KmpTaskTWithPrivatesTySize = 5010 CGF.getTypeSize(KmpTaskTWithPrivatesQTy); 5011 QualType SharedsPtrTy = C.getPointerType(SharedsTy); 5012 5013 // Emit initial values for private copies (if any). 5014 llvm::Value *TaskPrivatesMap = nullptr; 5015 llvm::Type *TaskPrivatesMapTy = 5016 std::next(TaskFunction->arg_begin(), 3)->getType(); 5017 if (!Privates.empty()) { 5018 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 5019 TaskPrivatesMap = emitTaskPrivateMappingFunction( 5020 CGM, Loc, Data.PrivateVars, Data.FirstprivateVars, Data.LastprivateVars, 5021 FI->getType(), Privates); 5022 TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5023 TaskPrivatesMap, TaskPrivatesMapTy); 5024 } else { 5025 TaskPrivatesMap = llvm::ConstantPointerNull::get( 5026 cast<llvm::PointerType>(TaskPrivatesMapTy)); 5027 } 5028 // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid, 5029 // kmp_task_t *tt); 5030 llvm::Function *TaskEntry = emitProxyTaskFunction( 5031 CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 5032 KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction, 5033 TaskPrivatesMap); 5034 5035 // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 5036 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 5037 // kmp_routine_entry_t *task_entry); 5038 // Task flags. Format is taken from 5039 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h, 5040 // description of kmp_tasking_flags struct. 5041 enum { 5042 TiedFlag = 0x1, 5043 FinalFlag = 0x2, 5044 DestructorsFlag = 0x8, 5045 PriorityFlag = 0x20, 5046 DetachableFlag = 0x40, 5047 }; 5048 unsigned Flags = Data.Tied ? TiedFlag : 0; 5049 bool NeedsCleanup = false; 5050 if (!Privates.empty()) { 5051 NeedsCleanup = checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD); 5052 if (NeedsCleanup) 5053 Flags = Flags | DestructorsFlag; 5054 } 5055 if (Data.Priority.getInt()) 5056 Flags = Flags | PriorityFlag; 5057 if (D.hasClausesOfKind<OMPDetachClause>()) 5058 Flags = Flags | DetachableFlag; 5059 llvm::Value *TaskFlags = 5060 Data.Final.getPointer() 5061 ? CGF.Builder.CreateSelect(Data.Final.getPointer(), 5062 CGF.Builder.getInt32(FinalFlag), 5063 CGF.Builder.getInt32(/*C=*/0)) 5064 : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0); 5065 TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags)); 5066 llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy)); 5067 SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc), 5068 getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize, 5069 SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5070 TaskEntry, KmpRoutineEntryPtrTy)}; 5071 llvm::Value *NewTask; 5072 if (D.hasClausesOfKind<OMPNowaitClause>()) { 5073 // Check if we have any device clause associated with the directive. 5074 const Expr *Device = nullptr; 5075 if (auto *C = D.getSingleClause<OMPDeviceClause>()) 5076 Device = C->getDevice(); 5077 // Emit device ID if any otherwise use default value. 5078 llvm::Value *DeviceID; 5079 if (Device) 5080 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 5081 CGF.Int64Ty, /*isSigned=*/true); 5082 else 5083 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 5084 AllocArgs.push_back(DeviceID); 5085 NewTask = CGF.EmitRuntimeCall( 5086 createRuntimeFunction(OMPRTL__kmpc_omp_target_task_alloc), AllocArgs); 5087 } else { 5088 NewTask = CGF.EmitRuntimeCall( 5089 createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs); 5090 } 5091 // Emit detach clause initialization. 5092 // evt = (typeof(evt))__kmpc_task_allow_completion_event(loc, tid, 5093 // task_descriptor); 5094 if (const auto *DC = D.getSingleClause<OMPDetachClause>()) { 5095 const Expr *Evt = DC->getEventHandler()->IgnoreParenImpCasts(); 5096 LValue EvtLVal = CGF.EmitLValue(Evt); 5097 5098 // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref, 5099 // int gtid, kmp_task_t *task); 5100 llvm::Value *Loc = emitUpdateLocation(CGF, DC->getBeginLoc()); 5101 llvm::Value *Tid = getThreadID(CGF, DC->getBeginLoc()); 5102 Tid = CGF.Builder.CreateIntCast(Tid, CGF.IntTy, /*isSigned=*/false); 5103 llvm::Value *EvtVal = CGF.EmitRuntimeCall( 5104 createRuntimeFunction(OMPRTL__kmpc_task_allow_completion_event), 5105 {Loc, Tid, NewTask}); 5106 EvtVal = CGF.EmitScalarConversion(EvtVal, C.VoidPtrTy, Evt->getType(), 5107 Evt->getExprLoc()); 5108 CGF.EmitStoreOfScalar(EvtVal, EvtLVal); 5109 } 5110 llvm::Value *NewTaskNewTaskTTy = 5111 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5112 NewTask, KmpTaskTWithPrivatesPtrTy); 5113 LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy, 5114 KmpTaskTWithPrivatesQTy); 5115 LValue TDBase = 5116 CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin()); 5117 // Fill the data in the resulting kmp_task_t record. 5118 // Copy shareds if there are any. 5119 Address KmpTaskSharedsPtr = Address::invalid(); 5120 if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) { 5121 KmpTaskSharedsPtr = 5122 Address(CGF.EmitLoadOfScalar( 5123 CGF.EmitLValueForField( 5124 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), 5125 KmpTaskTShareds)), 5126 Loc), 5127 CGF.getNaturalTypeAlignment(SharedsTy)); 5128 LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy); 5129 LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy); 5130 CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap); 5131 } 5132 // Emit initial values for private copies (if any). 5133 TaskResultTy Result; 5134 if (!Privates.empty()) { 5135 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD, 5136 SharedsTy, SharedsPtrTy, Data, Privates, 5137 /*ForDup=*/false); 5138 if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) && 5139 (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) { 5140 Result.TaskDupFn = emitTaskDupFunction( 5141 CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD, 5142 KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates, 5143 /*WithLastIter=*/!Data.LastprivateVars.empty()); 5144 } 5145 } 5146 // Fields of union "kmp_cmplrdata_t" for destructors and priority. 5147 enum { Priority = 0, Destructors = 1 }; 5148 // Provide pointer to function with destructors for privates. 5149 auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1); 5150 const RecordDecl *KmpCmplrdataUD = 5151 (*FI)->getType()->getAsUnionType()->getDecl(); 5152 if (NeedsCleanup) { 5153 llvm::Value *DestructorFn = emitDestructorsFunction( 5154 CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 5155 KmpTaskTWithPrivatesQTy); 5156 LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI); 5157 LValue DestructorsLV = CGF.EmitLValueForField( 5158 Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors)); 5159 CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5160 DestructorFn, KmpRoutineEntryPtrTy), 5161 DestructorsLV); 5162 } 5163 // Set priority. 5164 if (Data.Priority.getInt()) { 5165 LValue Data2LV = CGF.EmitLValueForField( 5166 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2)); 5167 LValue PriorityLV = CGF.EmitLValueForField( 5168 Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority)); 5169 CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV); 5170 } 5171 Result.NewTask = NewTask; 5172 Result.TaskEntry = TaskEntry; 5173 Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy; 5174 Result.TDBase = TDBase; 5175 Result.KmpTaskTQTyRD = KmpTaskTQTyRD; 5176 return Result; 5177 } 5178 5179 namespace { 5180 /// Dependence kind for RTL. 5181 enum RTLDependenceKindTy { 5182 DepIn = 0x01, 5183 DepInOut = 0x3, 5184 DepMutexInOutSet = 0x4 5185 }; 5186 /// Fields ids in kmp_depend_info record. 5187 enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags }; 5188 } // namespace 5189 5190 /// Translates internal dependency kind into the runtime kind. 5191 static RTLDependenceKindTy translateDependencyKind(OpenMPDependClauseKind K) { 5192 RTLDependenceKindTy DepKind; 5193 switch (K) { 5194 case OMPC_DEPEND_in: 5195 DepKind = DepIn; 5196 break; 5197 // Out and InOut dependencies must use the same code. 5198 case OMPC_DEPEND_out: 5199 case OMPC_DEPEND_inout: 5200 DepKind = DepInOut; 5201 break; 5202 case OMPC_DEPEND_mutexinoutset: 5203 DepKind = DepMutexInOutSet; 5204 break; 5205 case OMPC_DEPEND_source: 5206 case OMPC_DEPEND_sink: 5207 case OMPC_DEPEND_depobj: 5208 case OMPC_DEPEND_unknown: 5209 llvm_unreachable("Unknown task dependence type"); 5210 } 5211 return DepKind; 5212 } 5213 5214 /// Builds kmp_depend_info, if it is not built yet, and builds flags type. 5215 static void getDependTypes(ASTContext &C, QualType &KmpDependInfoTy, 5216 QualType &FlagsTy) { 5217 FlagsTy = C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false); 5218 if (KmpDependInfoTy.isNull()) { 5219 RecordDecl *KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info"); 5220 KmpDependInfoRD->startDefinition(); 5221 addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType()); 5222 addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType()); 5223 addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy); 5224 KmpDependInfoRD->completeDefinition(); 5225 KmpDependInfoTy = C.getRecordType(KmpDependInfoRD); 5226 } 5227 } 5228 5229 std::pair<llvm::Value *, LValue> 5230 CGOpenMPRuntime::getDepobjElements(CodeGenFunction &CGF, LValue DepobjLVal, 5231 SourceLocation Loc) { 5232 ASTContext &C = CGM.getContext(); 5233 QualType FlagsTy; 5234 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5235 RecordDecl *KmpDependInfoRD = 5236 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5237 LValue Base = CGF.EmitLoadOfPointerLValue( 5238 DepobjLVal.getAddress(CGF), 5239 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 5240 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 5241 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5242 Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy)); 5243 Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(), 5244 Base.getTBAAInfo()); 5245 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP( 5246 Addr.getPointer(), 5247 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 5248 LValue NumDepsBase = CGF.MakeAddrLValue( 5249 Address(DepObjAddr, Addr.getAlignment()), KmpDependInfoTy, 5250 Base.getBaseInfo(), Base.getTBAAInfo()); 5251 // NumDeps = deps[i].base_addr; 5252 LValue BaseAddrLVal = CGF.EmitLValueForField( 5253 NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 5254 llvm::Value *NumDeps = CGF.EmitLoadOfScalar(BaseAddrLVal, Loc); 5255 return std::make_pair(NumDeps, Base); 5256 } 5257 5258 std::pair<llvm::Value *, Address> CGOpenMPRuntime::emitDependClause( 5259 CodeGenFunction &CGF, 5260 ArrayRef<std::pair<OpenMPDependClauseKind, const Expr *>> Dependencies, 5261 bool ForDepobj, SourceLocation Loc) { 5262 // Process list of dependencies. 5263 ASTContext &C = CGM.getContext(); 5264 Address DependenciesArray = Address::invalid(); 5265 unsigned NumDependencies = Dependencies.size(); 5266 llvm::Value *NumOfElements = nullptr; 5267 if (NumDependencies) { 5268 QualType FlagsTy; 5269 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5270 RecordDecl *KmpDependInfoRD = 5271 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5272 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 5273 unsigned NumDepobjDependecies = 0; 5274 SmallVector<std::pair<llvm::Value *, LValue>, 4> Depobjs; 5275 llvm::Value *NumOfDepobjElements = llvm::ConstantInt::get(CGF.IntPtrTy, 0); 5276 // Calculate number of depobj dependecies. 5277 for (const std::pair<OpenMPDependClauseKind, const Expr *> &Pair : 5278 Dependencies) { 5279 if (Pair.first != OMPC_DEPEND_depobj) 5280 continue; 5281 LValue DepobjLVal = CGF.EmitLValue(Pair.second); 5282 llvm::Value *NumDeps; 5283 LValue Base; 5284 std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc); 5285 NumOfDepobjElements = 5286 CGF.Builder.CreateNUWAdd(NumOfDepobjElements, NumDeps); 5287 Depobjs.emplace_back(NumDeps, Base); 5288 ++NumDepobjDependecies; 5289 } 5290 5291 QualType KmpDependInfoArrayTy; 5292 // Define type kmp_depend_info[<Dependencies.size()>]; 5293 // For depobj reserve one extra element to store the number of elements. 5294 // It is required to handle depobj(x) update(in) construct. 5295 // kmp_depend_info[<Dependencies.size()>] deps; 5296 if (ForDepobj) { 5297 assert(NumDepobjDependecies == 0 && 5298 "depobj dependency kind is not expected in depobj directive."); 5299 KmpDependInfoArrayTy = C.getConstantArrayType( 5300 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies + 1), 5301 nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 5302 // Need to allocate on the dynamic memory. 5303 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5304 // Use default allocator. 5305 llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5306 CharUnits Align = C.getTypeAlignInChars(KmpDependInfoArrayTy); 5307 CharUnits Sz = C.getTypeSizeInChars(KmpDependInfoArrayTy); 5308 llvm::Value *Size = CGF.CGM.getSize(Sz.alignTo(Align)); 5309 llvm::Value *Args[] = {ThreadID, Size, Allocator}; 5310 5311 llvm::Value *Addr = CGF.EmitRuntimeCall( 5312 createRuntimeFunction(OMPRTL__kmpc_alloc), Args, ".dep.arr.addr"); 5313 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5314 Addr, CGF.ConvertTypeForMem(KmpDependInfoArrayTy)->getPointerTo()); 5315 DependenciesArray = Address(Addr, Align); 5316 NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies, 5317 /*isSigned=*/false); 5318 } else if (NumDepobjDependecies > 0) { 5319 NumOfElements = CGF.Builder.CreateNUWAdd( 5320 NumOfDepobjElements, 5321 llvm::ConstantInt::get(CGM.IntPtrTy, 5322 NumDependencies - NumDepobjDependecies, 5323 /*isSigned=*/false)); 5324 NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty, 5325 /*isSigned=*/false); 5326 OpaqueValueExpr OVE( 5327 Loc, C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0), 5328 VK_RValue); 5329 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, 5330 RValue::get(NumOfElements)); 5331 KmpDependInfoArrayTy = 5332 C.getVariableArrayType(KmpDependInfoTy, &OVE, ArrayType::Normal, 5333 /*IndexTypeQuals=*/0, SourceRange(Loc, Loc)); 5334 // CGF.EmitVariablyModifiedType(KmpDependInfoArrayTy); 5335 // Properly emit variable-sized array. 5336 auto *PD = ImplicitParamDecl::Create(C, KmpDependInfoArrayTy, 5337 ImplicitParamDecl::Other); 5338 CGF.EmitVarDecl(*PD); 5339 DependenciesArray = CGF.GetAddrOfLocalVar(PD); 5340 } else { 5341 KmpDependInfoArrayTy = C.getConstantArrayType( 5342 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies), 5343 nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 5344 DependenciesArray = 5345 CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr"); 5346 NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies, 5347 /*isSigned=*/false); 5348 } 5349 if (ForDepobj) { 5350 // Write number of elements in the first element of array for depobj. 5351 llvm::Value *NumVal = 5352 llvm::ConstantInt::get(CGF.IntPtrTy, NumDependencies); 5353 LValue Base = CGF.MakeAddrLValue( 5354 CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0), 5355 KmpDependInfoTy); 5356 // deps[i].base_addr = NumDependencies; 5357 LValue BaseAddrLVal = CGF.EmitLValueForField( 5358 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 5359 CGF.EmitStoreOfScalar(NumVal, BaseAddrLVal); 5360 } 5361 unsigned Pos = ForDepobj ? 1 : 0; 5362 for (unsigned I = 0; I < NumDependencies; ++I) { 5363 if (Dependencies[I].first == OMPC_DEPEND_depobj) 5364 continue; 5365 const Expr *E = Dependencies[I].second; 5366 const auto *OASE = dyn_cast<OMPArrayShapingExpr>(E); 5367 LValue Addr; 5368 if (OASE) { 5369 const Expr *Base = OASE->getBase()->IgnoreParenImpCasts(); 5370 Addr = 5371 CGF.EmitLoadOfPointerLValue(CGF.EmitLValue(Base).getAddress(CGF), 5372 Base->getType()->castAs<PointerType>()); 5373 } else { 5374 Addr = CGF.EmitLValue(E); 5375 } 5376 llvm::Value *Size; 5377 QualType Ty = E->getType(); 5378 if (OASE) { 5379 Size = llvm::ConstantInt::get(CGF.SizeTy,/*V=*/1); 5380 for (const Expr *SE : OASE->getDimensions()) { 5381 llvm::Value *Sz = CGF.EmitScalarExpr(SE); 5382 Sz = CGF.EmitScalarConversion(Sz, SE->getType(), 5383 CGF.getContext().getSizeType(), 5384 SE->getExprLoc()); 5385 Size = CGF.Builder.CreateNUWMul(Size, Sz); 5386 } 5387 } else if (const auto *ASE = 5388 dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) { 5389 LValue UpAddrLVal = 5390 CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false); 5391 llvm::Value *UpAddr = CGF.Builder.CreateConstGEP1_32( 5392 UpAddrLVal.getPointer(CGF), /*Idx0=*/1); 5393 llvm::Value *LowIntPtr = 5394 CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGM.SizeTy); 5395 llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGM.SizeTy); 5396 Size = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr); 5397 } else { 5398 Size = CGF.getTypeSize(Ty); 5399 } 5400 LValue Base; 5401 if (NumDepobjDependecies > 0) { 5402 Base = CGF.MakeAddrLValue( 5403 CGF.Builder.CreateConstGEP(DependenciesArray, Pos), 5404 KmpDependInfoTy); 5405 } else { 5406 Base = CGF.MakeAddrLValue( 5407 CGF.Builder.CreateConstArrayGEP(DependenciesArray, Pos), 5408 KmpDependInfoTy); 5409 } 5410 // deps[i].base_addr = &<Dependencies[i].second>; 5411 LValue BaseAddrLVal = CGF.EmitLValueForField( 5412 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 5413 CGF.EmitStoreOfScalar( 5414 CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGF.IntPtrTy), 5415 BaseAddrLVal); 5416 // deps[i].len = sizeof(<Dependencies[i].second>); 5417 LValue LenLVal = CGF.EmitLValueForField( 5418 Base, *std::next(KmpDependInfoRD->field_begin(), Len)); 5419 CGF.EmitStoreOfScalar(Size, LenLVal); 5420 // deps[i].flags = <Dependencies[i].first>; 5421 RTLDependenceKindTy DepKind = 5422 translateDependencyKind(Dependencies[I].first); 5423 LValue FlagsLVal = CGF.EmitLValueForField( 5424 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 5425 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 5426 FlagsLVal); 5427 ++Pos; 5428 } 5429 // Copy final depobj arrays. 5430 if (NumDepobjDependecies > 0) { 5431 llvm::Value *ElSize = CGF.getTypeSize(KmpDependInfoTy); 5432 Address Addr = CGF.Builder.CreateConstGEP(DependenciesArray, Pos); 5433 for (const std::pair<llvm::Value *, LValue> &Pair : Depobjs) { 5434 llvm::Value *Size = CGF.Builder.CreateNUWMul(ElSize, Pair.first); 5435 CGF.Builder.CreateMemCpy(Addr, Pair.second.getAddress(CGF), Size); 5436 Addr = 5437 Address(CGF.Builder.CreateGEP( 5438 Addr.getElementType(), Addr.getPointer(), Pair.first), 5439 DependenciesArray.getAlignment().alignmentOfArrayElement( 5440 C.getTypeSizeInChars(KmpDependInfoTy))); 5441 } 5442 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5443 DependenciesArray, CGF.VoidPtrTy); 5444 } else { 5445 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5446 CGF.Builder.CreateConstArrayGEP(DependenciesArray, ForDepobj ? 1 : 0), 5447 CGF.VoidPtrTy); 5448 } 5449 } 5450 return std::make_pair(NumOfElements, DependenciesArray); 5451 } 5452 5453 void CGOpenMPRuntime::emitDestroyClause(CodeGenFunction &CGF, LValue DepobjLVal, 5454 SourceLocation Loc) { 5455 ASTContext &C = CGM.getContext(); 5456 QualType FlagsTy; 5457 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5458 LValue Base = CGF.EmitLoadOfPointerLValue( 5459 DepobjLVal.getAddress(CGF), 5460 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 5461 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 5462 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5463 Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy)); 5464 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP( 5465 Addr.getPointer(), 5466 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 5467 DepObjAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(DepObjAddr, 5468 CGF.VoidPtrTy); 5469 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5470 // Use default allocator. 5471 llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5472 llvm::Value *Args[] = {ThreadID, DepObjAddr, Allocator}; 5473 5474 // _kmpc_free(gtid, addr, nullptr); 5475 (void)CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_free), Args); 5476 } 5477 5478 void CGOpenMPRuntime::emitUpdateClause(CodeGenFunction &CGF, LValue DepobjLVal, 5479 OpenMPDependClauseKind NewDepKind, 5480 SourceLocation Loc) { 5481 ASTContext &C = CGM.getContext(); 5482 QualType FlagsTy; 5483 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5484 RecordDecl *KmpDependInfoRD = 5485 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5486 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 5487 llvm::Value *NumDeps; 5488 LValue Base; 5489 std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc); 5490 5491 Address Begin = Base.getAddress(CGF); 5492 // Cast from pointer to array type to pointer to single element. 5493 llvm::Value *End = CGF.Builder.CreateGEP(Begin.getPointer(), NumDeps); 5494 // The basic structure here is a while-do loop. 5495 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.body"); 5496 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.done"); 5497 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5498 CGF.EmitBlock(BodyBB); 5499 llvm::PHINode *ElementPHI = 5500 CGF.Builder.CreatePHI(Begin.getType(), 2, "omp.elementPast"); 5501 ElementPHI->addIncoming(Begin.getPointer(), EntryBB); 5502 Begin = Address(ElementPHI, Begin.getAlignment()); 5503 Base = CGF.MakeAddrLValue(Begin, KmpDependInfoTy, Base.getBaseInfo(), 5504 Base.getTBAAInfo()); 5505 // deps[i].flags = NewDepKind; 5506 RTLDependenceKindTy DepKind = translateDependencyKind(NewDepKind); 5507 LValue FlagsLVal = CGF.EmitLValueForField( 5508 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 5509 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 5510 FlagsLVal); 5511 5512 // Shift the address forward by one element. 5513 Address ElementNext = 5514 CGF.Builder.CreateConstGEP(Begin, /*Index=*/1, "omp.elementNext"); 5515 ElementPHI->addIncoming(ElementNext.getPointer(), 5516 CGF.Builder.GetInsertBlock()); 5517 llvm::Value *IsEmpty = 5518 CGF.Builder.CreateICmpEQ(ElementNext.getPointer(), End, "omp.isempty"); 5519 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5520 // Done. 5521 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5522 } 5523 5524 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 5525 const OMPExecutableDirective &D, 5526 llvm::Function *TaskFunction, 5527 QualType SharedsTy, Address Shareds, 5528 const Expr *IfCond, 5529 const OMPTaskDataTy &Data) { 5530 if (!CGF.HaveInsertPoint()) 5531 return; 5532 5533 TaskResultTy Result = 5534 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5535 llvm::Value *NewTask = Result.NewTask; 5536 llvm::Function *TaskEntry = Result.TaskEntry; 5537 llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy; 5538 LValue TDBase = Result.TDBase; 5539 const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD; 5540 // Process list of dependences. 5541 Address DependenciesArray = Address::invalid(); 5542 llvm::Value *NumOfElements; 5543 std::tie(NumOfElements, DependenciesArray) = 5544 emitDependClause(CGF, Data.Dependences, /*ForDepobj=*/false, Loc); 5545 5546 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5547 // libcall. 5548 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 5549 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 5550 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence 5551 // list is not empty 5552 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5553 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5554 llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask }; 5555 llvm::Value *DepTaskArgs[7]; 5556 if (!Data.Dependences.empty()) { 5557 DepTaskArgs[0] = UpLoc; 5558 DepTaskArgs[1] = ThreadID; 5559 DepTaskArgs[2] = NewTask; 5560 DepTaskArgs[3] = NumOfElements; 5561 DepTaskArgs[4] = DependenciesArray.getPointer(); 5562 DepTaskArgs[5] = CGF.Builder.getInt32(0); 5563 DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5564 } 5565 auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, &TaskArgs, 5566 &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) { 5567 if (!Data.Tied) { 5568 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 5569 LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI); 5570 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal); 5571 } 5572 if (!Data.Dependences.empty()) { 5573 CGF.EmitRuntimeCall( 5574 createRuntimeFunction(OMPRTL__kmpc_omp_task_with_deps), DepTaskArgs); 5575 } else { 5576 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), 5577 TaskArgs); 5578 } 5579 // Check if parent region is untied and build return for untied task; 5580 if (auto *Region = 5581 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 5582 Region->emitUntiedSwitch(CGF); 5583 }; 5584 5585 llvm::Value *DepWaitTaskArgs[6]; 5586 if (!Data.Dependences.empty()) { 5587 DepWaitTaskArgs[0] = UpLoc; 5588 DepWaitTaskArgs[1] = ThreadID; 5589 DepWaitTaskArgs[2] = NumOfElements; 5590 DepWaitTaskArgs[3] = DependenciesArray.getPointer(); 5591 DepWaitTaskArgs[4] = CGF.Builder.getInt32(0); 5592 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5593 } 5594 auto &&ElseCodeGen = [&TaskArgs, ThreadID, NewTaskNewTaskTTy, TaskEntry, 5595 &Data, &DepWaitTaskArgs, 5596 Loc](CodeGenFunction &CGF, PrePostActionTy &) { 5597 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5598 CodeGenFunction::RunCleanupsScope LocalScope(CGF); 5599 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 5600 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 5601 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info 5602 // is specified. 5603 if (!Data.Dependences.empty()) 5604 CGF.EmitRuntimeCall(RT.createRuntimeFunction(OMPRTL__kmpc_omp_wait_deps), 5605 DepWaitTaskArgs); 5606 // Call proxy_task_entry(gtid, new_task); 5607 auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy, 5608 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 5609 Action.Enter(CGF); 5610 llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy}; 5611 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry, 5612 OutlinedFnArgs); 5613 }; 5614 5615 // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 5616 // kmp_task_t *new_task); 5617 // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 5618 // kmp_task_t *new_task); 5619 RegionCodeGenTy RCG(CodeGen); 5620 CommonActionTy Action( 5621 RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_begin_if0), TaskArgs, 5622 RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_complete_if0), TaskArgs); 5623 RCG.setAction(Action); 5624 RCG(CGF); 5625 }; 5626 5627 if (IfCond) { 5628 emitIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen); 5629 } else { 5630 RegionCodeGenTy ThenRCG(ThenCodeGen); 5631 ThenRCG(CGF); 5632 } 5633 } 5634 5635 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc, 5636 const OMPLoopDirective &D, 5637 llvm::Function *TaskFunction, 5638 QualType SharedsTy, Address Shareds, 5639 const Expr *IfCond, 5640 const OMPTaskDataTy &Data) { 5641 if (!CGF.HaveInsertPoint()) 5642 return; 5643 TaskResultTy Result = 5644 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5645 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5646 // libcall. 5647 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 5648 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 5649 // sched, kmp_uint64 grainsize, void *task_dup); 5650 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5651 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5652 llvm::Value *IfVal; 5653 if (IfCond) { 5654 IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy, 5655 /*isSigned=*/true); 5656 } else { 5657 IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1); 5658 } 5659 5660 LValue LBLVal = CGF.EmitLValueForField( 5661 Result.TDBase, 5662 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound)); 5663 const auto *LBVar = 5664 cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl()); 5665 CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(CGF), 5666 LBLVal.getQuals(), 5667 /*IsInitializer=*/true); 5668 LValue UBLVal = CGF.EmitLValueForField( 5669 Result.TDBase, 5670 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound)); 5671 const auto *UBVar = 5672 cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl()); 5673 CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(CGF), 5674 UBLVal.getQuals(), 5675 /*IsInitializer=*/true); 5676 LValue StLVal = CGF.EmitLValueForField( 5677 Result.TDBase, 5678 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride)); 5679 const auto *StVar = 5680 cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl()); 5681 CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(CGF), 5682 StLVal.getQuals(), 5683 /*IsInitializer=*/true); 5684 // Store reductions address. 5685 LValue RedLVal = CGF.EmitLValueForField( 5686 Result.TDBase, 5687 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions)); 5688 if (Data.Reductions) { 5689 CGF.EmitStoreOfScalar(Data.Reductions, RedLVal); 5690 } else { 5691 CGF.EmitNullInitialization(RedLVal.getAddress(CGF), 5692 CGF.getContext().VoidPtrTy); 5693 } 5694 enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 }; 5695 llvm::Value *TaskArgs[] = { 5696 UpLoc, 5697 ThreadID, 5698 Result.NewTask, 5699 IfVal, 5700 LBLVal.getPointer(CGF), 5701 UBLVal.getPointer(CGF), 5702 CGF.EmitLoadOfScalar(StLVal, Loc), 5703 llvm::ConstantInt::getSigned( 5704 CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler 5705 llvm::ConstantInt::getSigned( 5706 CGF.IntTy, Data.Schedule.getPointer() 5707 ? Data.Schedule.getInt() ? NumTasks : Grainsize 5708 : NoSchedule), 5709 Data.Schedule.getPointer() 5710 ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty, 5711 /*isSigned=*/false) 5712 : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0), 5713 Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5714 Result.TaskDupFn, CGF.VoidPtrTy) 5715 : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)}; 5716 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_taskloop), TaskArgs); 5717 } 5718 5719 /// Emit reduction operation for each element of array (required for 5720 /// array sections) LHS op = RHS. 5721 /// \param Type Type of array. 5722 /// \param LHSVar Variable on the left side of the reduction operation 5723 /// (references element of array in original variable). 5724 /// \param RHSVar Variable on the right side of the reduction operation 5725 /// (references element of array in original variable). 5726 /// \param RedOpGen Generator of reduction operation with use of LHSVar and 5727 /// RHSVar. 5728 static void EmitOMPAggregateReduction( 5729 CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar, 5730 const VarDecl *RHSVar, 5731 const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *, 5732 const Expr *, const Expr *)> &RedOpGen, 5733 const Expr *XExpr = nullptr, const Expr *EExpr = nullptr, 5734 const Expr *UpExpr = nullptr) { 5735 // Perform element-by-element initialization. 5736 QualType ElementTy; 5737 Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar); 5738 Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar); 5739 5740 // Drill down to the base element type on both arrays. 5741 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 5742 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr); 5743 5744 llvm::Value *RHSBegin = RHSAddr.getPointer(); 5745 llvm::Value *LHSBegin = LHSAddr.getPointer(); 5746 // Cast from pointer to array type to pointer to single element. 5747 llvm::Value *LHSEnd = CGF.Builder.CreateGEP(LHSBegin, NumElements); 5748 // The basic structure here is a while-do loop. 5749 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body"); 5750 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done"); 5751 llvm::Value *IsEmpty = 5752 CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty"); 5753 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5754 5755 // Enter the loop body, making that address the current address. 5756 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5757 CGF.EmitBlock(BodyBB); 5758 5759 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 5760 5761 llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI( 5762 RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast"); 5763 RHSElementPHI->addIncoming(RHSBegin, EntryBB); 5764 Address RHSElementCurrent = 5765 Address(RHSElementPHI, 5766 RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5767 5768 llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI( 5769 LHSBegin->getType(), 2, "omp.arraycpy.destElementPast"); 5770 LHSElementPHI->addIncoming(LHSBegin, EntryBB); 5771 Address LHSElementCurrent = 5772 Address(LHSElementPHI, 5773 LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5774 5775 // Emit copy. 5776 CodeGenFunction::OMPPrivateScope Scope(CGF); 5777 Scope.addPrivate(LHSVar, [=]() { return LHSElementCurrent; }); 5778 Scope.addPrivate(RHSVar, [=]() { return RHSElementCurrent; }); 5779 Scope.Privatize(); 5780 RedOpGen(CGF, XExpr, EExpr, UpExpr); 5781 Scope.ForceCleanup(); 5782 5783 // Shift the address forward by one element. 5784 llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32( 5785 LHSElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 5786 llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32( 5787 RHSElementPHI, /*Idx0=*/1, "omp.arraycpy.src.element"); 5788 // Check whether we've reached the end. 5789 llvm::Value *Done = 5790 CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done"); 5791 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 5792 LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock()); 5793 RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock()); 5794 5795 // Done. 5796 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5797 } 5798 5799 /// Emit reduction combiner. If the combiner is a simple expression emit it as 5800 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of 5801 /// UDR combiner function. 5802 static void emitReductionCombiner(CodeGenFunction &CGF, 5803 const Expr *ReductionOp) { 5804 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 5805 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 5806 if (const auto *DRE = 5807 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 5808 if (const auto *DRD = 5809 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) { 5810 std::pair<llvm::Function *, llvm::Function *> Reduction = 5811 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 5812 RValue Func = RValue::get(Reduction.first); 5813 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 5814 CGF.EmitIgnoredExpr(ReductionOp); 5815 return; 5816 } 5817 CGF.EmitIgnoredExpr(ReductionOp); 5818 } 5819 5820 llvm::Function *CGOpenMPRuntime::emitReductionFunction( 5821 SourceLocation Loc, llvm::Type *ArgsType, ArrayRef<const Expr *> Privates, 5822 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 5823 ArrayRef<const Expr *> ReductionOps) { 5824 ASTContext &C = CGM.getContext(); 5825 5826 // void reduction_func(void *LHSArg, void *RHSArg); 5827 FunctionArgList Args; 5828 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5829 ImplicitParamDecl::Other); 5830 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5831 ImplicitParamDecl::Other); 5832 Args.push_back(&LHSArg); 5833 Args.push_back(&RHSArg); 5834 const auto &CGFI = 5835 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5836 std::string Name = getName({"omp", "reduction", "reduction_func"}); 5837 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 5838 llvm::GlobalValue::InternalLinkage, Name, 5839 &CGM.getModule()); 5840 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 5841 Fn->setDoesNotRecurse(); 5842 CodeGenFunction CGF(CGM); 5843 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 5844 5845 // Dst = (void*[n])(LHSArg); 5846 // Src = (void*[n])(RHSArg); 5847 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5848 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 5849 ArgsType), CGF.getPointerAlign()); 5850 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5851 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 5852 ArgsType), CGF.getPointerAlign()); 5853 5854 // ... 5855 // *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]); 5856 // ... 5857 CodeGenFunction::OMPPrivateScope Scope(CGF); 5858 auto IPriv = Privates.begin(); 5859 unsigned Idx = 0; 5860 for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) { 5861 const auto *RHSVar = 5862 cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl()); 5863 Scope.addPrivate(RHSVar, [&CGF, RHS, Idx, RHSVar]() { 5864 return emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar); 5865 }); 5866 const auto *LHSVar = 5867 cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl()); 5868 Scope.addPrivate(LHSVar, [&CGF, LHS, Idx, LHSVar]() { 5869 return emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar); 5870 }); 5871 QualType PrivTy = (*IPriv)->getType(); 5872 if (PrivTy->isVariablyModifiedType()) { 5873 // Get array size and emit VLA type. 5874 ++Idx; 5875 Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx); 5876 llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem); 5877 const VariableArrayType *VLA = 5878 CGF.getContext().getAsVariableArrayType(PrivTy); 5879 const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr()); 5880 CodeGenFunction::OpaqueValueMapping OpaqueMap( 5881 CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy))); 5882 CGF.EmitVariablyModifiedType(PrivTy); 5883 } 5884 } 5885 Scope.Privatize(); 5886 IPriv = Privates.begin(); 5887 auto ILHS = LHSExprs.begin(); 5888 auto IRHS = RHSExprs.begin(); 5889 for (const Expr *E : ReductionOps) { 5890 if ((*IPriv)->getType()->isArrayType()) { 5891 // Emit reduction for array section. 5892 const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5893 const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5894 EmitOMPAggregateReduction( 5895 CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5896 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5897 emitReductionCombiner(CGF, E); 5898 }); 5899 } else { 5900 // Emit reduction for array subscript or single variable. 5901 emitReductionCombiner(CGF, E); 5902 } 5903 ++IPriv; 5904 ++ILHS; 5905 ++IRHS; 5906 } 5907 Scope.ForceCleanup(); 5908 CGF.FinishFunction(); 5909 return Fn; 5910 } 5911 5912 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF, 5913 const Expr *ReductionOp, 5914 const Expr *PrivateRef, 5915 const DeclRefExpr *LHS, 5916 const DeclRefExpr *RHS) { 5917 if (PrivateRef->getType()->isArrayType()) { 5918 // Emit reduction for array section. 5919 const auto *LHSVar = cast<VarDecl>(LHS->getDecl()); 5920 const auto *RHSVar = cast<VarDecl>(RHS->getDecl()); 5921 EmitOMPAggregateReduction( 5922 CGF, PrivateRef->getType(), LHSVar, RHSVar, 5923 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5924 emitReductionCombiner(CGF, ReductionOp); 5925 }); 5926 } else { 5927 // Emit reduction for array subscript or single variable. 5928 emitReductionCombiner(CGF, ReductionOp); 5929 } 5930 } 5931 5932 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc, 5933 ArrayRef<const Expr *> Privates, 5934 ArrayRef<const Expr *> LHSExprs, 5935 ArrayRef<const Expr *> RHSExprs, 5936 ArrayRef<const Expr *> ReductionOps, 5937 ReductionOptionsTy Options) { 5938 if (!CGF.HaveInsertPoint()) 5939 return; 5940 5941 bool WithNowait = Options.WithNowait; 5942 bool SimpleReduction = Options.SimpleReduction; 5943 5944 // Next code should be emitted for reduction: 5945 // 5946 // static kmp_critical_name lock = { 0 }; 5947 // 5948 // void reduce_func(void *lhs[<n>], void *rhs[<n>]) { 5949 // *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]); 5950 // ... 5951 // *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1], 5952 // *(Type<n>-1*)rhs[<n>-1]); 5953 // } 5954 // 5955 // ... 5956 // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]}; 5957 // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5958 // RedList, reduce_func, &<lock>)) { 5959 // case 1: 5960 // ... 5961 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5962 // ... 5963 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5964 // break; 5965 // case 2: 5966 // ... 5967 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5968 // ... 5969 // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);] 5970 // break; 5971 // default:; 5972 // } 5973 // 5974 // if SimpleReduction is true, only the next code is generated: 5975 // ... 5976 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5977 // ... 5978 5979 ASTContext &C = CGM.getContext(); 5980 5981 if (SimpleReduction) { 5982 CodeGenFunction::RunCleanupsScope Scope(CGF); 5983 auto IPriv = Privates.begin(); 5984 auto ILHS = LHSExprs.begin(); 5985 auto IRHS = RHSExprs.begin(); 5986 for (const Expr *E : ReductionOps) { 5987 emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5988 cast<DeclRefExpr>(*IRHS)); 5989 ++IPriv; 5990 ++ILHS; 5991 ++IRHS; 5992 } 5993 return; 5994 } 5995 5996 // 1. Build a list of reduction variables. 5997 // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]}; 5998 auto Size = RHSExprs.size(); 5999 for (const Expr *E : Privates) { 6000 if (E->getType()->isVariablyModifiedType()) 6001 // Reserve place for array size. 6002 ++Size; 6003 } 6004 llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size); 6005 QualType ReductionArrayTy = 6006 C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 6007 /*IndexTypeQuals=*/0); 6008 Address ReductionList = 6009 CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list"); 6010 auto IPriv = Privates.begin(); 6011 unsigned Idx = 0; 6012 for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) { 6013 Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 6014 CGF.Builder.CreateStore( 6015 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6016 CGF.EmitLValue(RHSExprs[I]).getPointer(CGF), CGF.VoidPtrTy), 6017 Elem); 6018 if ((*IPriv)->getType()->isVariablyModifiedType()) { 6019 // Store array size. 6020 ++Idx; 6021 Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 6022 llvm::Value *Size = CGF.Builder.CreateIntCast( 6023 CGF.getVLASize( 6024 CGF.getContext().getAsVariableArrayType((*IPriv)->getType())) 6025 .NumElts, 6026 CGF.SizeTy, /*isSigned=*/false); 6027 CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy), 6028 Elem); 6029 } 6030 } 6031 6032 // 2. Emit reduce_func(). 6033 llvm::Function *ReductionFn = emitReductionFunction( 6034 Loc, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), Privates, 6035 LHSExprs, RHSExprs, ReductionOps); 6036 6037 // 3. Create static kmp_critical_name lock = { 0 }; 6038 std::string Name = getName({"reduction"}); 6039 llvm::Value *Lock = getCriticalRegionLock(Name); 6040 6041 // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 6042 // RedList, reduce_func, &<lock>); 6043 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE); 6044 llvm::Value *ThreadId = getThreadID(CGF, Loc); 6045 llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy); 6046 llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6047 ReductionList.getPointer(), CGF.VoidPtrTy); 6048 llvm::Value *Args[] = { 6049 IdentTLoc, // ident_t *<loc> 6050 ThreadId, // i32 <gtid> 6051 CGF.Builder.getInt32(RHSExprs.size()), // i32 <n> 6052 ReductionArrayTySize, // size_type sizeof(RedList) 6053 RL, // void *RedList 6054 ReductionFn, // void (*) (void *, void *) <reduce_func> 6055 Lock // kmp_critical_name *&<lock> 6056 }; 6057 llvm::Value *Res = CGF.EmitRuntimeCall( 6058 createRuntimeFunction(WithNowait ? OMPRTL__kmpc_reduce_nowait 6059 : OMPRTL__kmpc_reduce), 6060 Args); 6061 6062 // 5. Build switch(res) 6063 llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default"); 6064 llvm::SwitchInst *SwInst = 6065 CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2); 6066 6067 // 6. Build case 1: 6068 // ... 6069 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 6070 // ... 6071 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 6072 // break; 6073 llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1"); 6074 SwInst->addCase(CGF.Builder.getInt32(1), Case1BB); 6075 CGF.EmitBlock(Case1BB); 6076 6077 // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 6078 llvm::Value *EndArgs[] = { 6079 IdentTLoc, // ident_t *<loc> 6080 ThreadId, // i32 <gtid> 6081 Lock // kmp_critical_name *&<lock> 6082 }; 6083 auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps]( 6084 CodeGenFunction &CGF, PrePostActionTy &Action) { 6085 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6086 auto IPriv = Privates.begin(); 6087 auto ILHS = LHSExprs.begin(); 6088 auto IRHS = RHSExprs.begin(); 6089 for (const Expr *E : ReductionOps) { 6090 RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 6091 cast<DeclRefExpr>(*IRHS)); 6092 ++IPriv; 6093 ++ILHS; 6094 ++IRHS; 6095 } 6096 }; 6097 RegionCodeGenTy RCG(CodeGen); 6098 CommonActionTy Action( 6099 nullptr, llvm::None, 6100 createRuntimeFunction(WithNowait ? OMPRTL__kmpc_end_reduce_nowait 6101 : OMPRTL__kmpc_end_reduce), 6102 EndArgs); 6103 RCG.setAction(Action); 6104 RCG(CGF); 6105 6106 CGF.EmitBranch(DefaultBB); 6107 6108 // 7. Build case 2: 6109 // ... 6110 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 6111 // ... 6112 // break; 6113 llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2"); 6114 SwInst->addCase(CGF.Builder.getInt32(2), Case2BB); 6115 CGF.EmitBlock(Case2BB); 6116 6117 auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps]( 6118 CodeGenFunction &CGF, PrePostActionTy &Action) { 6119 auto ILHS = LHSExprs.begin(); 6120 auto IRHS = RHSExprs.begin(); 6121 auto IPriv = Privates.begin(); 6122 for (const Expr *E : ReductionOps) { 6123 const Expr *XExpr = nullptr; 6124 const Expr *EExpr = nullptr; 6125 const Expr *UpExpr = nullptr; 6126 BinaryOperatorKind BO = BO_Comma; 6127 if (const auto *BO = dyn_cast<BinaryOperator>(E)) { 6128 if (BO->getOpcode() == BO_Assign) { 6129 XExpr = BO->getLHS(); 6130 UpExpr = BO->getRHS(); 6131 } 6132 } 6133 // Try to emit update expression as a simple atomic. 6134 const Expr *RHSExpr = UpExpr; 6135 if (RHSExpr) { 6136 // Analyze RHS part of the whole expression. 6137 if (const auto *ACO = dyn_cast<AbstractConditionalOperator>( 6138 RHSExpr->IgnoreParenImpCasts())) { 6139 // If this is a conditional operator, analyze its condition for 6140 // min/max reduction operator. 6141 RHSExpr = ACO->getCond(); 6142 } 6143 if (const auto *BORHS = 6144 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) { 6145 EExpr = BORHS->getRHS(); 6146 BO = BORHS->getOpcode(); 6147 } 6148 } 6149 if (XExpr) { 6150 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 6151 auto &&AtomicRedGen = [BO, VD, 6152 Loc](CodeGenFunction &CGF, const Expr *XExpr, 6153 const Expr *EExpr, const Expr *UpExpr) { 6154 LValue X = CGF.EmitLValue(XExpr); 6155 RValue E; 6156 if (EExpr) 6157 E = CGF.EmitAnyExpr(EExpr); 6158 CGF.EmitOMPAtomicSimpleUpdateExpr( 6159 X, E, BO, /*IsXLHSInRHSPart=*/true, 6160 llvm::AtomicOrdering::Monotonic, Loc, 6161 [&CGF, UpExpr, VD, Loc](RValue XRValue) { 6162 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 6163 PrivateScope.addPrivate( 6164 VD, [&CGF, VD, XRValue, Loc]() { 6165 Address LHSTemp = CGF.CreateMemTemp(VD->getType()); 6166 CGF.emitOMPSimpleStore( 6167 CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue, 6168 VD->getType().getNonReferenceType(), Loc); 6169 return LHSTemp; 6170 }); 6171 (void)PrivateScope.Privatize(); 6172 return CGF.EmitAnyExpr(UpExpr); 6173 }); 6174 }; 6175 if ((*IPriv)->getType()->isArrayType()) { 6176 // Emit atomic reduction for array section. 6177 const auto *RHSVar = 6178 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 6179 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar, 6180 AtomicRedGen, XExpr, EExpr, UpExpr); 6181 } else { 6182 // Emit atomic reduction for array subscript or single variable. 6183 AtomicRedGen(CGF, XExpr, EExpr, UpExpr); 6184 } 6185 } else { 6186 // Emit as a critical region. 6187 auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *, 6188 const Expr *, const Expr *) { 6189 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6190 std::string Name = RT.getName({"atomic_reduction"}); 6191 RT.emitCriticalRegion( 6192 CGF, Name, 6193 [=](CodeGenFunction &CGF, PrePostActionTy &Action) { 6194 Action.Enter(CGF); 6195 emitReductionCombiner(CGF, E); 6196 }, 6197 Loc); 6198 }; 6199 if ((*IPriv)->getType()->isArrayType()) { 6200 const auto *LHSVar = 6201 cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 6202 const auto *RHSVar = 6203 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 6204 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar, 6205 CritRedGen); 6206 } else { 6207 CritRedGen(CGF, nullptr, nullptr, nullptr); 6208 } 6209 } 6210 ++ILHS; 6211 ++IRHS; 6212 ++IPriv; 6213 } 6214 }; 6215 RegionCodeGenTy AtomicRCG(AtomicCodeGen); 6216 if (!WithNowait) { 6217 // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>); 6218 llvm::Value *EndArgs[] = { 6219 IdentTLoc, // ident_t *<loc> 6220 ThreadId, // i32 <gtid> 6221 Lock // kmp_critical_name *&<lock> 6222 }; 6223 CommonActionTy Action(nullptr, llvm::None, 6224 createRuntimeFunction(OMPRTL__kmpc_end_reduce), 6225 EndArgs); 6226 AtomicRCG.setAction(Action); 6227 AtomicRCG(CGF); 6228 } else { 6229 AtomicRCG(CGF); 6230 } 6231 6232 CGF.EmitBranch(DefaultBB); 6233 CGF.EmitBlock(DefaultBB, /*IsFinished=*/true); 6234 } 6235 6236 /// Generates unique name for artificial threadprivate variables. 6237 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>" 6238 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix, 6239 const Expr *Ref) { 6240 SmallString<256> Buffer; 6241 llvm::raw_svector_ostream Out(Buffer); 6242 const clang::DeclRefExpr *DE; 6243 const VarDecl *D = ::getBaseDecl(Ref, DE); 6244 if (!D) 6245 D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl()); 6246 D = D->getCanonicalDecl(); 6247 std::string Name = CGM.getOpenMPRuntime().getName( 6248 {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)}); 6249 Out << Prefix << Name << "_" 6250 << D->getCanonicalDecl()->getBeginLoc().getRawEncoding(); 6251 return std::string(Out.str()); 6252 } 6253 6254 /// Emits reduction initializer function: 6255 /// \code 6256 /// void @.red_init(void* %arg) { 6257 /// %0 = bitcast void* %arg to <type>* 6258 /// store <type> <init>, <type>* %0 6259 /// ret void 6260 /// } 6261 /// \endcode 6262 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM, 6263 SourceLocation Loc, 6264 ReductionCodeGen &RCG, unsigned N) { 6265 ASTContext &C = CGM.getContext(); 6266 FunctionArgList Args; 6267 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6268 ImplicitParamDecl::Other); 6269 Args.emplace_back(&Param); 6270 const auto &FnInfo = 6271 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6272 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6273 std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""}); 6274 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6275 Name, &CGM.getModule()); 6276 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6277 Fn->setDoesNotRecurse(); 6278 CodeGenFunction CGF(CGM); 6279 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6280 Address PrivateAddr = CGF.EmitLoadOfPointer( 6281 CGF.GetAddrOfLocalVar(&Param), 6282 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6283 llvm::Value *Size = nullptr; 6284 // If the size of the reduction item is non-constant, load it from global 6285 // threadprivate variable. 6286 if (RCG.getSizes(N).second) { 6287 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6288 CGF, CGM.getContext().getSizeType(), 6289 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6290 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6291 CGM.getContext().getSizeType(), Loc); 6292 } 6293 RCG.emitAggregateType(CGF, N, Size); 6294 LValue SharedLVal; 6295 // If initializer uses initializer from declare reduction construct, emit a 6296 // pointer to the address of the original reduction item (reuired by reduction 6297 // initializer) 6298 if (RCG.usesReductionInitializer(N)) { 6299 Address SharedAddr = 6300 CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6301 CGF, CGM.getContext().VoidPtrTy, 6302 generateUniqueName(CGM, "reduction", RCG.getRefExpr(N))); 6303 SharedAddr = CGF.EmitLoadOfPointer( 6304 SharedAddr, 6305 CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr()); 6306 SharedLVal = CGF.MakeAddrLValue(SharedAddr, CGM.getContext().VoidPtrTy); 6307 } else { 6308 SharedLVal = CGF.MakeNaturalAlignAddrLValue( 6309 llvm::ConstantPointerNull::get(CGM.VoidPtrTy), 6310 CGM.getContext().VoidPtrTy); 6311 } 6312 // Emit the initializer: 6313 // %0 = bitcast void* %arg to <type>* 6314 // store <type> <init>, <type>* %0 6315 RCG.emitInitialization(CGF, N, PrivateAddr, SharedLVal, 6316 [](CodeGenFunction &) { return false; }); 6317 CGF.FinishFunction(); 6318 return Fn; 6319 } 6320 6321 /// Emits reduction combiner function: 6322 /// \code 6323 /// void @.red_comb(void* %arg0, void* %arg1) { 6324 /// %lhs = bitcast void* %arg0 to <type>* 6325 /// %rhs = bitcast void* %arg1 to <type>* 6326 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs) 6327 /// store <type> %2, <type>* %lhs 6328 /// ret void 6329 /// } 6330 /// \endcode 6331 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM, 6332 SourceLocation Loc, 6333 ReductionCodeGen &RCG, unsigned N, 6334 const Expr *ReductionOp, 6335 const Expr *LHS, const Expr *RHS, 6336 const Expr *PrivateRef) { 6337 ASTContext &C = CGM.getContext(); 6338 const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl()); 6339 const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl()); 6340 FunctionArgList Args; 6341 ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 6342 C.VoidPtrTy, ImplicitParamDecl::Other); 6343 ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6344 ImplicitParamDecl::Other); 6345 Args.emplace_back(&ParamInOut); 6346 Args.emplace_back(&ParamIn); 6347 const auto &FnInfo = 6348 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6349 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6350 std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""}); 6351 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6352 Name, &CGM.getModule()); 6353 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6354 Fn->setDoesNotRecurse(); 6355 CodeGenFunction CGF(CGM); 6356 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6357 llvm::Value *Size = nullptr; 6358 // If the size of the reduction item is non-constant, load it from global 6359 // threadprivate variable. 6360 if (RCG.getSizes(N).second) { 6361 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6362 CGF, CGM.getContext().getSizeType(), 6363 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6364 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6365 CGM.getContext().getSizeType(), Loc); 6366 } 6367 RCG.emitAggregateType(CGF, N, Size); 6368 // Remap lhs and rhs variables to the addresses of the function arguments. 6369 // %lhs = bitcast void* %arg0 to <type>* 6370 // %rhs = bitcast void* %arg1 to <type>* 6371 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 6372 PrivateScope.addPrivate(LHSVD, [&C, &CGF, &ParamInOut, LHSVD]() { 6373 // Pull out the pointer to the variable. 6374 Address PtrAddr = CGF.EmitLoadOfPointer( 6375 CGF.GetAddrOfLocalVar(&ParamInOut), 6376 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6377 return CGF.Builder.CreateElementBitCast( 6378 PtrAddr, CGF.ConvertTypeForMem(LHSVD->getType())); 6379 }); 6380 PrivateScope.addPrivate(RHSVD, [&C, &CGF, &ParamIn, RHSVD]() { 6381 // Pull out the pointer to the variable. 6382 Address PtrAddr = CGF.EmitLoadOfPointer( 6383 CGF.GetAddrOfLocalVar(&ParamIn), 6384 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6385 return CGF.Builder.CreateElementBitCast( 6386 PtrAddr, CGF.ConvertTypeForMem(RHSVD->getType())); 6387 }); 6388 PrivateScope.Privatize(); 6389 // Emit the combiner body: 6390 // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs) 6391 // store <type> %2, <type>* %lhs 6392 CGM.getOpenMPRuntime().emitSingleReductionCombiner( 6393 CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS), 6394 cast<DeclRefExpr>(RHS)); 6395 CGF.FinishFunction(); 6396 return Fn; 6397 } 6398 6399 /// Emits reduction finalizer function: 6400 /// \code 6401 /// void @.red_fini(void* %arg) { 6402 /// %0 = bitcast void* %arg to <type>* 6403 /// <destroy>(<type>* %0) 6404 /// ret void 6405 /// } 6406 /// \endcode 6407 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM, 6408 SourceLocation Loc, 6409 ReductionCodeGen &RCG, unsigned N) { 6410 if (!RCG.needCleanups(N)) 6411 return nullptr; 6412 ASTContext &C = CGM.getContext(); 6413 FunctionArgList Args; 6414 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6415 ImplicitParamDecl::Other); 6416 Args.emplace_back(&Param); 6417 const auto &FnInfo = 6418 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6419 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6420 std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""}); 6421 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6422 Name, &CGM.getModule()); 6423 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6424 Fn->setDoesNotRecurse(); 6425 CodeGenFunction CGF(CGM); 6426 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6427 Address PrivateAddr = CGF.EmitLoadOfPointer( 6428 CGF.GetAddrOfLocalVar(&Param), 6429 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6430 llvm::Value *Size = nullptr; 6431 // If the size of the reduction item is non-constant, load it from global 6432 // threadprivate variable. 6433 if (RCG.getSizes(N).second) { 6434 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6435 CGF, CGM.getContext().getSizeType(), 6436 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6437 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6438 CGM.getContext().getSizeType(), Loc); 6439 } 6440 RCG.emitAggregateType(CGF, N, Size); 6441 // Emit the finalizer body: 6442 // <destroy>(<type>* %0) 6443 RCG.emitCleanups(CGF, N, PrivateAddr); 6444 CGF.FinishFunction(Loc); 6445 return Fn; 6446 } 6447 6448 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit( 6449 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 6450 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 6451 if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty()) 6452 return nullptr; 6453 6454 // Build typedef struct: 6455 // kmp_task_red_input { 6456 // void *reduce_shar; // shared reduction item 6457 // size_t reduce_size; // size of data item 6458 // void *reduce_init; // data initialization routine 6459 // void *reduce_fini; // data finalization routine 6460 // void *reduce_comb; // data combiner routine 6461 // kmp_task_red_flags_t flags; // flags for additional info from compiler 6462 // } kmp_task_red_input_t; 6463 ASTContext &C = CGM.getContext(); 6464 RecordDecl *RD = C.buildImplicitRecord("kmp_task_red_input_t"); 6465 RD->startDefinition(); 6466 const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6467 const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType()); 6468 const FieldDecl *InitFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6469 const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6470 const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6471 const FieldDecl *FlagsFD = addFieldToRecordDecl( 6472 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false)); 6473 RD->completeDefinition(); 6474 QualType RDType = C.getRecordType(RD); 6475 unsigned Size = Data.ReductionVars.size(); 6476 llvm::APInt ArraySize(/*numBits=*/64, Size); 6477 QualType ArrayRDType = C.getConstantArrayType( 6478 RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 6479 // kmp_task_red_input_t .rd_input.[Size]; 6480 Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input."); 6481 ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionCopies, 6482 Data.ReductionOps); 6483 for (unsigned Cnt = 0; Cnt < Size; ++Cnt) { 6484 // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt]; 6485 llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0), 6486 llvm::ConstantInt::get(CGM.SizeTy, Cnt)}; 6487 llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP( 6488 TaskRedInput.getPointer(), Idxs, 6489 /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc, 6490 ".rd_input.gep."); 6491 LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType); 6492 // ElemLVal.reduce_shar = &Shareds[Cnt]; 6493 LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD); 6494 RCG.emitSharedLValue(CGF, Cnt); 6495 llvm::Value *CastedShared = 6496 CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer(CGF)); 6497 CGF.EmitStoreOfScalar(CastedShared, SharedLVal); 6498 RCG.emitAggregateType(CGF, Cnt); 6499 llvm::Value *SizeValInChars; 6500 llvm::Value *SizeVal; 6501 std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt); 6502 // We use delayed creation/initialization for VLAs, array sections and 6503 // custom reduction initializations. It is required because runtime does not 6504 // provide the way to pass the sizes of VLAs/array sections to 6505 // initializer/combiner/finalizer functions and does not pass the pointer to 6506 // original reduction item to the initializer. Instead threadprivate global 6507 // variables are used to store these values and use them in the functions. 6508 bool DelayedCreation = !!SizeVal; 6509 SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy, 6510 /*isSigned=*/false); 6511 LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD); 6512 CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal); 6513 // ElemLVal.reduce_init = init; 6514 LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD); 6515 llvm::Value *InitAddr = 6516 CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt)); 6517 CGF.EmitStoreOfScalar(InitAddr, InitLVal); 6518 DelayedCreation = DelayedCreation || RCG.usesReductionInitializer(Cnt); 6519 // ElemLVal.reduce_fini = fini; 6520 LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD); 6521 llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt); 6522 llvm::Value *FiniAddr = Fini 6523 ? CGF.EmitCastToVoidPtr(Fini) 6524 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy); 6525 CGF.EmitStoreOfScalar(FiniAddr, FiniLVal); 6526 // ElemLVal.reduce_comb = comb; 6527 LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD); 6528 llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction( 6529 CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt], 6530 RHSExprs[Cnt], Data.ReductionCopies[Cnt])); 6531 CGF.EmitStoreOfScalar(CombAddr, CombLVal); 6532 // ElemLVal.flags = 0; 6533 LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD); 6534 if (DelayedCreation) { 6535 CGF.EmitStoreOfScalar( 6536 llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true), 6537 FlagsLVal); 6538 } else 6539 CGF.EmitNullInitialization(FlagsLVal.getAddress(CGF), 6540 FlagsLVal.getType()); 6541 } 6542 // Build call void *__kmpc_task_reduction_init(int gtid, int num_data, void 6543 // *data); 6544 llvm::Value *Args[] = { 6545 CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy, 6546 /*isSigned=*/true), 6547 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true), 6548 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(), 6549 CGM.VoidPtrTy)}; 6550 return CGF.EmitRuntimeCall( 6551 createRuntimeFunction(OMPRTL__kmpc_task_reduction_init), Args); 6552 } 6553 6554 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 6555 SourceLocation Loc, 6556 ReductionCodeGen &RCG, 6557 unsigned N) { 6558 auto Sizes = RCG.getSizes(N); 6559 // Emit threadprivate global variable if the type is non-constant 6560 // (Sizes.second = nullptr). 6561 if (Sizes.second) { 6562 llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy, 6563 /*isSigned=*/false); 6564 Address SizeAddr = getAddrOfArtificialThreadPrivate( 6565 CGF, CGM.getContext().getSizeType(), 6566 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6567 CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false); 6568 } 6569 // Store address of the original reduction item if custom initializer is used. 6570 if (RCG.usesReductionInitializer(N)) { 6571 Address SharedAddr = getAddrOfArtificialThreadPrivate( 6572 CGF, CGM.getContext().VoidPtrTy, 6573 generateUniqueName(CGM, "reduction", RCG.getRefExpr(N))); 6574 CGF.Builder.CreateStore( 6575 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6576 RCG.getSharedLValue(N).getPointer(CGF), CGM.VoidPtrTy), 6577 SharedAddr, /*IsVolatile=*/false); 6578 } 6579 } 6580 6581 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF, 6582 SourceLocation Loc, 6583 llvm::Value *ReductionsPtr, 6584 LValue SharedLVal) { 6585 // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 6586 // *d); 6587 llvm::Value *Args[] = {CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), 6588 CGM.IntTy, 6589 /*isSigned=*/true), 6590 ReductionsPtr, 6591 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6592 SharedLVal.getPointer(CGF), CGM.VoidPtrTy)}; 6593 return Address( 6594 CGF.EmitRuntimeCall( 6595 createRuntimeFunction(OMPRTL__kmpc_task_reduction_get_th_data), Args), 6596 SharedLVal.getAlignment()); 6597 } 6598 6599 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 6600 SourceLocation Loc) { 6601 if (!CGF.HaveInsertPoint()) 6602 return; 6603 6604 llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder(); 6605 if (OMPBuilder) { 6606 OMPBuilder->CreateTaskwait(CGF.Builder); 6607 } else { 6608 // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 6609 // global_tid); 6610 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 6611 // Ignore return result until untied tasks are supported. 6612 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskwait), Args); 6613 } 6614 6615 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 6616 Region->emitUntiedSwitch(CGF); 6617 } 6618 6619 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF, 6620 OpenMPDirectiveKind InnerKind, 6621 const RegionCodeGenTy &CodeGen, 6622 bool HasCancel) { 6623 if (!CGF.HaveInsertPoint()) 6624 return; 6625 InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel); 6626 CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr); 6627 } 6628 6629 namespace { 6630 enum RTCancelKind { 6631 CancelNoreq = 0, 6632 CancelParallel = 1, 6633 CancelLoop = 2, 6634 CancelSections = 3, 6635 CancelTaskgroup = 4 6636 }; 6637 } // anonymous namespace 6638 6639 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) { 6640 RTCancelKind CancelKind = CancelNoreq; 6641 if (CancelRegion == OMPD_parallel) 6642 CancelKind = CancelParallel; 6643 else if (CancelRegion == OMPD_for) 6644 CancelKind = CancelLoop; 6645 else if (CancelRegion == OMPD_sections) 6646 CancelKind = CancelSections; 6647 else { 6648 assert(CancelRegion == OMPD_taskgroup); 6649 CancelKind = CancelTaskgroup; 6650 } 6651 return CancelKind; 6652 } 6653 6654 void CGOpenMPRuntime::emitCancellationPointCall( 6655 CodeGenFunction &CGF, SourceLocation Loc, 6656 OpenMPDirectiveKind CancelRegion) { 6657 if (!CGF.HaveInsertPoint()) 6658 return; 6659 // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 6660 // global_tid, kmp_int32 cncl_kind); 6661 if (auto *OMPRegionInfo = 6662 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6663 // For 'cancellation point taskgroup', the task region info may not have a 6664 // cancel. This may instead happen in another adjacent task. 6665 if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) { 6666 llvm::Value *Args[] = { 6667 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 6668 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6669 // Ignore return result until untied tasks are supported. 6670 llvm::Value *Result = CGF.EmitRuntimeCall( 6671 createRuntimeFunction(OMPRTL__kmpc_cancellationpoint), Args); 6672 // if (__kmpc_cancellationpoint()) { 6673 // exit from construct; 6674 // } 6675 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6676 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6677 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6678 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6679 CGF.EmitBlock(ExitBB); 6680 // exit from construct; 6681 CodeGenFunction::JumpDest CancelDest = 6682 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6683 CGF.EmitBranchThroughCleanup(CancelDest); 6684 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6685 } 6686 } 6687 } 6688 6689 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc, 6690 const Expr *IfCond, 6691 OpenMPDirectiveKind CancelRegion) { 6692 if (!CGF.HaveInsertPoint()) 6693 return; 6694 // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 6695 // kmp_int32 cncl_kind); 6696 if (auto *OMPRegionInfo = 6697 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6698 auto &&ThenGen = [Loc, CancelRegion, OMPRegionInfo](CodeGenFunction &CGF, 6699 PrePostActionTy &) { 6700 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6701 llvm::Value *Args[] = { 6702 RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc), 6703 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6704 // Ignore return result until untied tasks are supported. 6705 llvm::Value *Result = CGF.EmitRuntimeCall( 6706 RT.createRuntimeFunction(OMPRTL__kmpc_cancel), Args); 6707 // if (__kmpc_cancel()) { 6708 // exit from construct; 6709 // } 6710 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6711 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6712 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6713 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6714 CGF.EmitBlock(ExitBB); 6715 // exit from construct; 6716 CodeGenFunction::JumpDest CancelDest = 6717 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6718 CGF.EmitBranchThroughCleanup(CancelDest); 6719 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6720 }; 6721 if (IfCond) { 6722 emitIfClause(CGF, IfCond, ThenGen, 6723 [](CodeGenFunction &, PrePostActionTy &) {}); 6724 } else { 6725 RegionCodeGenTy ThenRCG(ThenGen); 6726 ThenRCG(CGF); 6727 } 6728 } 6729 } 6730 6731 void CGOpenMPRuntime::emitTargetOutlinedFunction( 6732 const OMPExecutableDirective &D, StringRef ParentName, 6733 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6734 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6735 assert(!ParentName.empty() && "Invalid target region parent name!"); 6736 HasEmittedTargetRegion = true; 6737 emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID, 6738 IsOffloadEntry, CodeGen); 6739 } 6740 6741 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper( 6742 const OMPExecutableDirective &D, StringRef ParentName, 6743 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6744 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6745 // Create a unique name for the entry function using the source location 6746 // information of the current target region. The name will be something like: 6747 // 6748 // __omp_offloading_DD_FFFF_PP_lBB 6749 // 6750 // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the 6751 // mangled name of the function that encloses the target region and BB is the 6752 // line number of the target region. 6753 6754 unsigned DeviceID; 6755 unsigned FileID; 6756 unsigned Line; 6757 getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID, 6758 Line); 6759 SmallString<64> EntryFnName; 6760 { 6761 llvm::raw_svector_ostream OS(EntryFnName); 6762 OS << "__omp_offloading" << llvm::format("_%x", DeviceID) 6763 << llvm::format("_%x_", FileID) << ParentName << "_l" << Line; 6764 } 6765 6766 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 6767 6768 CodeGenFunction CGF(CGM, true); 6769 CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName); 6770 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6771 6772 OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS, D.getBeginLoc()); 6773 6774 // If this target outline function is not an offload entry, we don't need to 6775 // register it. 6776 if (!IsOffloadEntry) 6777 return; 6778 6779 // The target region ID is used by the runtime library to identify the current 6780 // target region, so it only has to be unique and not necessarily point to 6781 // anything. It could be the pointer to the outlined function that implements 6782 // the target region, but we aren't using that so that the compiler doesn't 6783 // need to keep that, and could therefore inline the host function if proven 6784 // worthwhile during optimization. In the other hand, if emitting code for the 6785 // device, the ID has to be the function address so that it can retrieved from 6786 // the offloading entry and launched by the runtime library. We also mark the 6787 // outlined function to have external linkage in case we are emitting code for 6788 // the device, because these functions will be entry points to the device. 6789 6790 if (CGM.getLangOpts().OpenMPIsDevice) { 6791 OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy); 6792 OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 6793 OutlinedFn->setDSOLocal(false); 6794 } else { 6795 std::string Name = getName({EntryFnName, "region_id"}); 6796 OutlinedFnID = new llvm::GlobalVariable( 6797 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 6798 llvm::GlobalValue::WeakAnyLinkage, 6799 llvm::Constant::getNullValue(CGM.Int8Ty), Name); 6800 } 6801 6802 // Register the information for the entry associated with this target region. 6803 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 6804 DeviceID, FileID, ParentName, Line, OutlinedFn, OutlinedFnID, 6805 OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion); 6806 } 6807 6808 /// Checks if the expression is constant or does not have non-trivial function 6809 /// calls. 6810 static bool isTrivial(ASTContext &Ctx, const Expr * E) { 6811 // We can skip constant expressions. 6812 // We can skip expressions with trivial calls or simple expressions. 6813 return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) || 6814 !E->hasNonTrivialCall(Ctx)) && 6815 !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true); 6816 } 6817 6818 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx, 6819 const Stmt *Body) { 6820 const Stmt *Child = Body->IgnoreContainers(); 6821 while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) { 6822 Child = nullptr; 6823 for (const Stmt *S : C->body()) { 6824 if (const auto *E = dyn_cast<Expr>(S)) { 6825 if (isTrivial(Ctx, E)) 6826 continue; 6827 } 6828 // Some of the statements can be ignored. 6829 if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) || 6830 isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S)) 6831 continue; 6832 // Analyze declarations. 6833 if (const auto *DS = dyn_cast<DeclStmt>(S)) { 6834 if (llvm::all_of(DS->decls(), [&Ctx](const Decl *D) { 6835 if (isa<EmptyDecl>(D) || isa<DeclContext>(D) || 6836 isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) || 6837 isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) || 6838 isa<UsingDirectiveDecl>(D) || 6839 isa<OMPDeclareReductionDecl>(D) || 6840 isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D)) 6841 return true; 6842 const auto *VD = dyn_cast<VarDecl>(D); 6843 if (!VD) 6844 return false; 6845 return VD->isConstexpr() || 6846 ((VD->getType().isTrivialType(Ctx) || 6847 VD->getType()->isReferenceType()) && 6848 (!VD->hasInit() || isTrivial(Ctx, VD->getInit()))); 6849 })) 6850 continue; 6851 } 6852 // Found multiple children - cannot get the one child only. 6853 if (Child) 6854 return nullptr; 6855 Child = S; 6856 } 6857 if (Child) 6858 Child = Child->IgnoreContainers(); 6859 } 6860 return Child; 6861 } 6862 6863 /// Emit the number of teams for a target directive. Inspect the num_teams 6864 /// clause associated with a teams construct combined or closely nested 6865 /// with the target directive. 6866 /// 6867 /// Emit a team of size one for directives such as 'target parallel' that 6868 /// have no associated teams construct. 6869 /// 6870 /// Otherwise, return nullptr. 6871 static llvm::Value * 6872 emitNumTeamsForTargetDirective(CodeGenFunction &CGF, 6873 const OMPExecutableDirective &D) { 6874 assert(!CGF.getLangOpts().OpenMPIsDevice && 6875 "Clauses associated with the teams directive expected to be emitted " 6876 "only for the host!"); 6877 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6878 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 6879 "Expected target-based executable directive."); 6880 CGBuilderTy &Bld = CGF.Builder; 6881 switch (DirectiveKind) { 6882 case OMPD_target: { 6883 const auto *CS = D.getInnermostCapturedStmt(); 6884 const auto *Body = 6885 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 6886 const Stmt *ChildStmt = 6887 CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body); 6888 if (const auto *NestedDir = 6889 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 6890 if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) { 6891 if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) { 6892 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6893 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6894 const Expr *NumTeams = 6895 NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6896 llvm::Value *NumTeamsVal = 6897 CGF.EmitScalarExpr(NumTeams, 6898 /*IgnoreResultAssign*/ true); 6899 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6900 /*isSigned=*/true); 6901 } 6902 return Bld.getInt32(0); 6903 } 6904 if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) || 6905 isOpenMPSimdDirective(NestedDir->getDirectiveKind())) 6906 return Bld.getInt32(1); 6907 return Bld.getInt32(0); 6908 } 6909 return nullptr; 6910 } 6911 case OMPD_target_teams: 6912 case OMPD_target_teams_distribute: 6913 case OMPD_target_teams_distribute_simd: 6914 case OMPD_target_teams_distribute_parallel_for: 6915 case OMPD_target_teams_distribute_parallel_for_simd: { 6916 if (D.hasClausesOfKind<OMPNumTeamsClause>()) { 6917 CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF); 6918 const Expr *NumTeams = 6919 D.getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6920 llvm::Value *NumTeamsVal = 6921 CGF.EmitScalarExpr(NumTeams, 6922 /*IgnoreResultAssign*/ true); 6923 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6924 /*isSigned=*/true); 6925 } 6926 return Bld.getInt32(0); 6927 } 6928 case OMPD_target_parallel: 6929 case OMPD_target_parallel_for: 6930 case OMPD_target_parallel_for_simd: 6931 case OMPD_target_simd: 6932 return Bld.getInt32(1); 6933 case OMPD_parallel: 6934 case OMPD_for: 6935 case OMPD_parallel_for: 6936 case OMPD_parallel_master: 6937 case OMPD_parallel_sections: 6938 case OMPD_for_simd: 6939 case OMPD_parallel_for_simd: 6940 case OMPD_cancel: 6941 case OMPD_cancellation_point: 6942 case OMPD_ordered: 6943 case OMPD_threadprivate: 6944 case OMPD_allocate: 6945 case OMPD_task: 6946 case OMPD_simd: 6947 case OMPD_sections: 6948 case OMPD_section: 6949 case OMPD_single: 6950 case OMPD_master: 6951 case OMPD_critical: 6952 case OMPD_taskyield: 6953 case OMPD_barrier: 6954 case OMPD_taskwait: 6955 case OMPD_taskgroup: 6956 case OMPD_atomic: 6957 case OMPD_flush: 6958 case OMPD_depobj: 6959 case OMPD_scan: 6960 case OMPD_teams: 6961 case OMPD_target_data: 6962 case OMPD_target_exit_data: 6963 case OMPD_target_enter_data: 6964 case OMPD_distribute: 6965 case OMPD_distribute_simd: 6966 case OMPD_distribute_parallel_for: 6967 case OMPD_distribute_parallel_for_simd: 6968 case OMPD_teams_distribute: 6969 case OMPD_teams_distribute_simd: 6970 case OMPD_teams_distribute_parallel_for: 6971 case OMPD_teams_distribute_parallel_for_simd: 6972 case OMPD_target_update: 6973 case OMPD_declare_simd: 6974 case OMPD_declare_variant: 6975 case OMPD_begin_declare_variant: 6976 case OMPD_end_declare_variant: 6977 case OMPD_declare_target: 6978 case OMPD_end_declare_target: 6979 case OMPD_declare_reduction: 6980 case OMPD_declare_mapper: 6981 case OMPD_taskloop: 6982 case OMPD_taskloop_simd: 6983 case OMPD_master_taskloop: 6984 case OMPD_master_taskloop_simd: 6985 case OMPD_parallel_master_taskloop: 6986 case OMPD_parallel_master_taskloop_simd: 6987 case OMPD_requires: 6988 case OMPD_unknown: 6989 break; 6990 } 6991 llvm_unreachable("Unexpected directive kind."); 6992 } 6993 6994 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS, 6995 llvm::Value *DefaultThreadLimitVal) { 6996 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6997 CGF.getContext(), CS->getCapturedStmt()); 6998 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6999 if (isOpenMPParallelDirective(Dir->getDirectiveKind())) { 7000 llvm::Value *NumThreads = nullptr; 7001 llvm::Value *CondVal = nullptr; 7002 // Handle if clause. If if clause present, the number of threads is 7003 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 7004 if (Dir->hasClausesOfKind<OMPIfClause>()) { 7005 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 7006 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 7007 const OMPIfClause *IfClause = nullptr; 7008 for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) { 7009 if (C->getNameModifier() == OMPD_unknown || 7010 C->getNameModifier() == OMPD_parallel) { 7011 IfClause = C; 7012 break; 7013 } 7014 } 7015 if (IfClause) { 7016 const Expr *Cond = IfClause->getCondition(); 7017 bool Result; 7018 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 7019 if (!Result) 7020 return CGF.Builder.getInt32(1); 7021 } else { 7022 CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange()); 7023 if (const auto *PreInit = 7024 cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) { 7025 for (const auto *I : PreInit->decls()) { 7026 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 7027 CGF.EmitVarDecl(cast<VarDecl>(*I)); 7028 } else { 7029 CodeGenFunction::AutoVarEmission Emission = 7030 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 7031 CGF.EmitAutoVarCleanups(Emission); 7032 } 7033 } 7034 } 7035 CondVal = CGF.EvaluateExprAsBool(Cond); 7036 } 7037 } 7038 } 7039 // Check the value of num_threads clause iff if clause was not specified 7040 // or is not evaluated to false. 7041 if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) { 7042 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 7043 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 7044 const auto *NumThreadsClause = 7045 Dir->getSingleClause<OMPNumThreadsClause>(); 7046 CodeGenFunction::LexicalScope Scope( 7047 CGF, NumThreadsClause->getNumThreads()->getSourceRange()); 7048 if (const auto *PreInit = 7049 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) { 7050 for (const auto *I : PreInit->decls()) { 7051 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 7052 CGF.EmitVarDecl(cast<VarDecl>(*I)); 7053 } else { 7054 CodeGenFunction::AutoVarEmission Emission = 7055 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 7056 CGF.EmitAutoVarCleanups(Emission); 7057 } 7058 } 7059 } 7060 NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads()); 7061 NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, 7062 /*isSigned=*/false); 7063 if (DefaultThreadLimitVal) 7064 NumThreads = CGF.Builder.CreateSelect( 7065 CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads), 7066 DefaultThreadLimitVal, NumThreads); 7067 } else { 7068 NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal 7069 : CGF.Builder.getInt32(0); 7070 } 7071 // Process condition of the if clause. 7072 if (CondVal) { 7073 NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads, 7074 CGF.Builder.getInt32(1)); 7075 } 7076 return NumThreads; 7077 } 7078 if (isOpenMPSimdDirective(Dir->getDirectiveKind())) 7079 return CGF.Builder.getInt32(1); 7080 return DefaultThreadLimitVal; 7081 } 7082 return DefaultThreadLimitVal ? DefaultThreadLimitVal 7083 : CGF.Builder.getInt32(0); 7084 } 7085 7086 /// Emit the number of threads for a target directive. Inspect the 7087 /// thread_limit clause associated with a teams construct combined or closely 7088 /// nested with the target directive. 7089 /// 7090 /// Emit the num_threads clause for directives such as 'target parallel' that 7091 /// have no associated teams construct. 7092 /// 7093 /// Otherwise, return nullptr. 7094 static llvm::Value * 7095 emitNumThreadsForTargetDirective(CodeGenFunction &CGF, 7096 const OMPExecutableDirective &D) { 7097 assert(!CGF.getLangOpts().OpenMPIsDevice && 7098 "Clauses associated with the teams directive expected to be emitted " 7099 "only for the host!"); 7100 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 7101 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 7102 "Expected target-based executable directive."); 7103 CGBuilderTy &Bld = CGF.Builder; 7104 llvm::Value *ThreadLimitVal = nullptr; 7105 llvm::Value *NumThreadsVal = nullptr; 7106 switch (DirectiveKind) { 7107 case OMPD_target: { 7108 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 7109 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7110 return NumThreads; 7111 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7112 CGF.getContext(), CS->getCapturedStmt()); 7113 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 7114 if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) { 7115 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 7116 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 7117 const auto *ThreadLimitClause = 7118 Dir->getSingleClause<OMPThreadLimitClause>(); 7119 CodeGenFunction::LexicalScope Scope( 7120 CGF, ThreadLimitClause->getThreadLimit()->getSourceRange()); 7121 if (const auto *PreInit = 7122 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) { 7123 for (const auto *I : PreInit->decls()) { 7124 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 7125 CGF.EmitVarDecl(cast<VarDecl>(*I)); 7126 } else { 7127 CodeGenFunction::AutoVarEmission Emission = 7128 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 7129 CGF.EmitAutoVarCleanups(Emission); 7130 } 7131 } 7132 } 7133 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7134 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7135 ThreadLimitVal = 7136 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7137 } 7138 if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) && 7139 !isOpenMPDistributeDirective(Dir->getDirectiveKind())) { 7140 CS = Dir->getInnermostCapturedStmt(); 7141 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7142 CGF.getContext(), CS->getCapturedStmt()); 7143 Dir = dyn_cast_or_null<OMPExecutableDirective>(Child); 7144 } 7145 if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) && 7146 !isOpenMPSimdDirective(Dir->getDirectiveKind())) { 7147 CS = Dir->getInnermostCapturedStmt(); 7148 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7149 return NumThreads; 7150 } 7151 if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind())) 7152 return Bld.getInt32(1); 7153 } 7154 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 7155 } 7156 case OMPD_target_teams: { 7157 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7158 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7159 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7160 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7161 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7162 ThreadLimitVal = 7163 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7164 } 7165 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 7166 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7167 return NumThreads; 7168 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7169 CGF.getContext(), CS->getCapturedStmt()); 7170 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 7171 if (Dir->getDirectiveKind() == OMPD_distribute) { 7172 CS = Dir->getInnermostCapturedStmt(); 7173 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7174 return NumThreads; 7175 } 7176 } 7177 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 7178 } 7179 case OMPD_target_teams_distribute: 7180 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7181 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7182 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7183 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7184 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7185 ThreadLimitVal = 7186 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7187 } 7188 return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal); 7189 case OMPD_target_parallel: 7190 case OMPD_target_parallel_for: 7191 case OMPD_target_parallel_for_simd: 7192 case OMPD_target_teams_distribute_parallel_for: 7193 case OMPD_target_teams_distribute_parallel_for_simd: { 7194 llvm::Value *CondVal = nullptr; 7195 // Handle if clause. If if clause present, the number of threads is 7196 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 7197 if (D.hasClausesOfKind<OMPIfClause>()) { 7198 const OMPIfClause *IfClause = nullptr; 7199 for (const auto *C : D.getClausesOfKind<OMPIfClause>()) { 7200 if (C->getNameModifier() == OMPD_unknown || 7201 C->getNameModifier() == OMPD_parallel) { 7202 IfClause = C; 7203 break; 7204 } 7205 } 7206 if (IfClause) { 7207 const Expr *Cond = IfClause->getCondition(); 7208 bool Result; 7209 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 7210 if (!Result) 7211 return Bld.getInt32(1); 7212 } else { 7213 CodeGenFunction::RunCleanupsScope Scope(CGF); 7214 CondVal = CGF.EvaluateExprAsBool(Cond); 7215 } 7216 } 7217 } 7218 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7219 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7220 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7221 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7222 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7223 ThreadLimitVal = 7224 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7225 } 7226 if (D.hasClausesOfKind<OMPNumThreadsClause>()) { 7227 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF); 7228 const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>(); 7229 llvm::Value *NumThreads = CGF.EmitScalarExpr( 7230 NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true); 7231 NumThreadsVal = 7232 Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false); 7233 ThreadLimitVal = ThreadLimitVal 7234 ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal, 7235 ThreadLimitVal), 7236 NumThreadsVal, ThreadLimitVal) 7237 : NumThreadsVal; 7238 } 7239 if (!ThreadLimitVal) 7240 ThreadLimitVal = Bld.getInt32(0); 7241 if (CondVal) 7242 return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1)); 7243 return ThreadLimitVal; 7244 } 7245 case OMPD_target_teams_distribute_simd: 7246 case OMPD_target_simd: 7247 return Bld.getInt32(1); 7248 case OMPD_parallel: 7249 case OMPD_for: 7250 case OMPD_parallel_for: 7251 case OMPD_parallel_master: 7252 case OMPD_parallel_sections: 7253 case OMPD_for_simd: 7254 case OMPD_parallel_for_simd: 7255 case OMPD_cancel: 7256 case OMPD_cancellation_point: 7257 case OMPD_ordered: 7258 case OMPD_threadprivate: 7259 case OMPD_allocate: 7260 case OMPD_task: 7261 case OMPD_simd: 7262 case OMPD_sections: 7263 case OMPD_section: 7264 case OMPD_single: 7265 case OMPD_master: 7266 case OMPD_critical: 7267 case OMPD_taskyield: 7268 case OMPD_barrier: 7269 case OMPD_taskwait: 7270 case OMPD_taskgroup: 7271 case OMPD_atomic: 7272 case OMPD_flush: 7273 case OMPD_depobj: 7274 case OMPD_scan: 7275 case OMPD_teams: 7276 case OMPD_target_data: 7277 case OMPD_target_exit_data: 7278 case OMPD_target_enter_data: 7279 case OMPD_distribute: 7280 case OMPD_distribute_simd: 7281 case OMPD_distribute_parallel_for: 7282 case OMPD_distribute_parallel_for_simd: 7283 case OMPD_teams_distribute: 7284 case OMPD_teams_distribute_simd: 7285 case OMPD_teams_distribute_parallel_for: 7286 case OMPD_teams_distribute_parallel_for_simd: 7287 case OMPD_target_update: 7288 case OMPD_declare_simd: 7289 case OMPD_declare_variant: 7290 case OMPD_begin_declare_variant: 7291 case OMPD_end_declare_variant: 7292 case OMPD_declare_target: 7293 case OMPD_end_declare_target: 7294 case OMPD_declare_reduction: 7295 case OMPD_declare_mapper: 7296 case OMPD_taskloop: 7297 case OMPD_taskloop_simd: 7298 case OMPD_master_taskloop: 7299 case OMPD_master_taskloop_simd: 7300 case OMPD_parallel_master_taskloop: 7301 case OMPD_parallel_master_taskloop_simd: 7302 case OMPD_requires: 7303 case OMPD_unknown: 7304 break; 7305 } 7306 llvm_unreachable("Unsupported directive kind."); 7307 } 7308 7309 namespace { 7310 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 7311 7312 // Utility to handle information from clauses associated with a given 7313 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause). 7314 // It provides a convenient interface to obtain the information and generate 7315 // code for that information. 7316 class MappableExprsHandler { 7317 public: 7318 /// Values for bit flags used to specify the mapping type for 7319 /// offloading. 7320 enum OpenMPOffloadMappingFlags : uint64_t { 7321 /// No flags 7322 OMP_MAP_NONE = 0x0, 7323 /// Allocate memory on the device and move data from host to device. 7324 OMP_MAP_TO = 0x01, 7325 /// Allocate memory on the device and move data from device to host. 7326 OMP_MAP_FROM = 0x02, 7327 /// Always perform the requested mapping action on the element, even 7328 /// if it was already mapped before. 7329 OMP_MAP_ALWAYS = 0x04, 7330 /// Delete the element from the device environment, ignoring the 7331 /// current reference count associated with the element. 7332 OMP_MAP_DELETE = 0x08, 7333 /// The element being mapped is a pointer-pointee pair; both the 7334 /// pointer and the pointee should be mapped. 7335 OMP_MAP_PTR_AND_OBJ = 0x10, 7336 /// This flags signals that the base address of an entry should be 7337 /// passed to the target kernel as an argument. 7338 OMP_MAP_TARGET_PARAM = 0x20, 7339 /// Signal that the runtime library has to return the device pointer 7340 /// in the current position for the data being mapped. Used when we have the 7341 /// use_device_ptr clause. 7342 OMP_MAP_RETURN_PARAM = 0x40, 7343 /// This flag signals that the reference being passed is a pointer to 7344 /// private data. 7345 OMP_MAP_PRIVATE = 0x80, 7346 /// Pass the element to the device by value. 7347 OMP_MAP_LITERAL = 0x100, 7348 /// Implicit map 7349 OMP_MAP_IMPLICIT = 0x200, 7350 /// Close is a hint to the runtime to allocate memory close to 7351 /// the target device. 7352 OMP_MAP_CLOSE = 0x400, 7353 /// The 16 MSBs of the flags indicate whether the entry is member of some 7354 /// struct/class. 7355 OMP_MAP_MEMBER_OF = 0xffff000000000000, 7356 LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF), 7357 }; 7358 7359 /// Get the offset of the OMP_MAP_MEMBER_OF field. 7360 static unsigned getFlagMemberOffset() { 7361 unsigned Offset = 0; 7362 for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1); 7363 Remain = Remain >> 1) 7364 Offset++; 7365 return Offset; 7366 } 7367 7368 /// Class that associates information with a base pointer to be passed to the 7369 /// runtime library. 7370 class BasePointerInfo { 7371 /// The base pointer. 7372 llvm::Value *Ptr = nullptr; 7373 /// The base declaration that refers to this device pointer, or null if 7374 /// there is none. 7375 const ValueDecl *DevPtrDecl = nullptr; 7376 7377 public: 7378 BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr) 7379 : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {} 7380 llvm::Value *operator*() const { return Ptr; } 7381 const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; } 7382 void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; } 7383 }; 7384 7385 using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>; 7386 using MapValuesArrayTy = SmallVector<llvm::Value *, 4>; 7387 using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>; 7388 7389 /// Map between a struct and the its lowest & highest elements which have been 7390 /// mapped. 7391 /// [ValueDecl *] --> {LE(FieldIndex, Pointer), 7392 /// HE(FieldIndex, Pointer)} 7393 struct StructRangeInfoTy { 7394 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = { 7395 0, Address::invalid()}; 7396 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = { 7397 0, Address::invalid()}; 7398 Address Base = Address::invalid(); 7399 }; 7400 7401 private: 7402 /// Kind that defines how a device pointer has to be returned. 7403 struct MapInfo { 7404 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 7405 OpenMPMapClauseKind MapType = OMPC_MAP_unknown; 7406 ArrayRef<OpenMPMapModifierKind> MapModifiers; 7407 bool ReturnDevicePointer = false; 7408 bool IsImplicit = false; 7409 7410 MapInfo() = default; 7411 MapInfo( 7412 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7413 OpenMPMapClauseKind MapType, 7414 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7415 bool ReturnDevicePointer, bool IsImplicit) 7416 : Components(Components), MapType(MapType), MapModifiers(MapModifiers), 7417 ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit) {} 7418 }; 7419 7420 /// If use_device_ptr is used on a pointer which is a struct member and there 7421 /// is no map information about it, then emission of that entry is deferred 7422 /// until the whole struct has been processed. 7423 struct DeferredDevicePtrEntryTy { 7424 const Expr *IE = nullptr; 7425 const ValueDecl *VD = nullptr; 7426 7427 DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD) 7428 : IE(IE), VD(VD) {} 7429 }; 7430 7431 /// The target directive from where the mappable clauses were extracted. It 7432 /// is either a executable directive or a user-defined mapper directive. 7433 llvm::PointerUnion<const OMPExecutableDirective *, 7434 const OMPDeclareMapperDecl *> 7435 CurDir; 7436 7437 /// Function the directive is being generated for. 7438 CodeGenFunction &CGF; 7439 7440 /// Set of all first private variables in the current directive. 7441 /// bool data is set to true if the variable is implicitly marked as 7442 /// firstprivate, false otherwise. 7443 llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls; 7444 7445 /// Map between device pointer declarations and their expression components. 7446 /// The key value for declarations in 'this' is null. 7447 llvm::DenseMap< 7448 const ValueDecl *, 7449 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>> 7450 DevPointersMap; 7451 7452 llvm::Value *getExprTypeSize(const Expr *E) const { 7453 QualType ExprTy = E->getType().getCanonicalType(); 7454 7455 // Reference types are ignored for mapping purposes. 7456 if (const auto *RefTy = ExprTy->getAs<ReferenceType>()) 7457 ExprTy = RefTy->getPointeeType().getCanonicalType(); 7458 7459 // Given that an array section is considered a built-in type, we need to 7460 // do the calculation based on the length of the section instead of relying 7461 // on CGF.getTypeSize(E->getType()). 7462 if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) { 7463 QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType( 7464 OAE->getBase()->IgnoreParenImpCasts()) 7465 .getCanonicalType(); 7466 7467 // If there is no length associated with the expression and lower bound is 7468 // not specified too, that means we are using the whole length of the 7469 // base. 7470 if (!OAE->getLength() && OAE->getColonLoc().isValid() && 7471 !OAE->getLowerBound()) 7472 return CGF.getTypeSize(BaseTy); 7473 7474 llvm::Value *ElemSize; 7475 if (const auto *PTy = BaseTy->getAs<PointerType>()) { 7476 ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType()); 7477 } else { 7478 const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr()); 7479 assert(ATy && "Expecting array type if not a pointer type."); 7480 ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType()); 7481 } 7482 7483 // If we don't have a length at this point, that is because we have an 7484 // array section with a single element. 7485 if (!OAE->getLength() && OAE->getColonLoc().isInvalid()) 7486 return ElemSize; 7487 7488 if (const Expr *LenExpr = OAE->getLength()) { 7489 llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr); 7490 LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(), 7491 CGF.getContext().getSizeType(), 7492 LenExpr->getExprLoc()); 7493 return CGF.Builder.CreateNUWMul(LengthVal, ElemSize); 7494 } 7495 assert(!OAE->getLength() && OAE->getColonLoc().isValid() && 7496 OAE->getLowerBound() && "expected array_section[lb:]."); 7497 // Size = sizetype - lb * elemtype; 7498 llvm::Value *LengthVal = CGF.getTypeSize(BaseTy); 7499 llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound()); 7500 LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(), 7501 CGF.getContext().getSizeType(), 7502 OAE->getLowerBound()->getExprLoc()); 7503 LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize); 7504 llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal); 7505 llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal); 7506 LengthVal = CGF.Builder.CreateSelect( 7507 Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0)); 7508 return LengthVal; 7509 } 7510 return CGF.getTypeSize(ExprTy); 7511 } 7512 7513 /// Return the corresponding bits for a given map clause modifier. Add 7514 /// a flag marking the map as a pointer if requested. Add a flag marking the 7515 /// map as the first one of a series of maps that relate to the same map 7516 /// expression. 7517 OpenMPOffloadMappingFlags getMapTypeBits( 7518 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers, 7519 bool IsImplicit, bool AddPtrFlag, bool AddIsTargetParamFlag) const { 7520 OpenMPOffloadMappingFlags Bits = 7521 IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE; 7522 switch (MapType) { 7523 case OMPC_MAP_alloc: 7524 case OMPC_MAP_release: 7525 // alloc and release is the default behavior in the runtime library, i.e. 7526 // if we don't pass any bits alloc/release that is what the runtime is 7527 // going to do. Therefore, we don't need to signal anything for these two 7528 // type modifiers. 7529 break; 7530 case OMPC_MAP_to: 7531 Bits |= OMP_MAP_TO; 7532 break; 7533 case OMPC_MAP_from: 7534 Bits |= OMP_MAP_FROM; 7535 break; 7536 case OMPC_MAP_tofrom: 7537 Bits |= OMP_MAP_TO | OMP_MAP_FROM; 7538 break; 7539 case OMPC_MAP_delete: 7540 Bits |= OMP_MAP_DELETE; 7541 break; 7542 case OMPC_MAP_unknown: 7543 llvm_unreachable("Unexpected map type!"); 7544 } 7545 if (AddPtrFlag) 7546 Bits |= OMP_MAP_PTR_AND_OBJ; 7547 if (AddIsTargetParamFlag) 7548 Bits |= OMP_MAP_TARGET_PARAM; 7549 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_always) 7550 != MapModifiers.end()) 7551 Bits |= OMP_MAP_ALWAYS; 7552 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_close) 7553 != MapModifiers.end()) 7554 Bits |= OMP_MAP_CLOSE; 7555 return Bits; 7556 } 7557 7558 /// Return true if the provided expression is a final array section. A 7559 /// final array section, is one whose length can't be proved to be one. 7560 bool isFinalArraySectionExpression(const Expr *E) const { 7561 const auto *OASE = dyn_cast<OMPArraySectionExpr>(E); 7562 7563 // It is not an array section and therefore not a unity-size one. 7564 if (!OASE) 7565 return false; 7566 7567 // An array section with no colon always refer to a single element. 7568 if (OASE->getColonLoc().isInvalid()) 7569 return false; 7570 7571 const Expr *Length = OASE->getLength(); 7572 7573 // If we don't have a length we have to check if the array has size 1 7574 // for this dimension. Also, we should always expect a length if the 7575 // base type is pointer. 7576 if (!Length) { 7577 QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType( 7578 OASE->getBase()->IgnoreParenImpCasts()) 7579 .getCanonicalType(); 7580 if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr())) 7581 return ATy->getSize().getSExtValue() != 1; 7582 // If we don't have a constant dimension length, we have to consider 7583 // the current section as having any size, so it is not necessarily 7584 // unitary. If it happen to be unity size, that's user fault. 7585 return true; 7586 } 7587 7588 // Check if the length evaluates to 1. 7589 Expr::EvalResult Result; 7590 if (!Length->EvaluateAsInt(Result, CGF.getContext())) 7591 return true; // Can have more that size 1. 7592 7593 llvm::APSInt ConstLength = Result.Val.getInt(); 7594 return ConstLength.getSExtValue() != 1; 7595 } 7596 7597 /// Generate the base pointers, section pointers, sizes and map type 7598 /// bits for the provided map type, map modifier, and expression components. 7599 /// \a IsFirstComponent should be set to true if the provided set of 7600 /// components is the first associated with a capture. 7601 void generateInfoForComponentList( 7602 OpenMPMapClauseKind MapType, 7603 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7604 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7605 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 7606 MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types, 7607 StructRangeInfoTy &PartialStruct, bool IsFirstComponentList, 7608 bool IsImplicit, 7609 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 7610 OverlappedElements = llvm::None) const { 7611 // The following summarizes what has to be generated for each map and the 7612 // types below. The generated information is expressed in this order: 7613 // base pointer, section pointer, size, flags 7614 // (to add to the ones that come from the map type and modifier). 7615 // 7616 // double d; 7617 // int i[100]; 7618 // float *p; 7619 // 7620 // struct S1 { 7621 // int i; 7622 // float f[50]; 7623 // } 7624 // struct S2 { 7625 // int i; 7626 // float f[50]; 7627 // S1 s; 7628 // double *p; 7629 // struct S2 *ps; 7630 // } 7631 // S2 s; 7632 // S2 *ps; 7633 // 7634 // map(d) 7635 // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM 7636 // 7637 // map(i) 7638 // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM 7639 // 7640 // map(i[1:23]) 7641 // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM 7642 // 7643 // map(p) 7644 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM 7645 // 7646 // map(p[1:24]) 7647 // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM 7648 // 7649 // map(s) 7650 // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM 7651 // 7652 // map(s.i) 7653 // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM 7654 // 7655 // map(s.s.f) 7656 // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7657 // 7658 // map(s.p) 7659 // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM 7660 // 7661 // map(to: s.p[:22]) 7662 // &s, &(s.p), sizeof(double*), TARGET_PARAM (*) 7663 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**) 7664 // &(s.p), &(s.p[0]), 22*sizeof(double), 7665 // MEMBER_OF(1) | PTR_AND_OBJ | TO (***) 7666 // (*) alloc space for struct members, only this is a target parameter 7667 // (**) map the pointer (nothing to be mapped in this example) (the compiler 7668 // optimizes this entry out, same in the examples below) 7669 // (***) map the pointee (map: to) 7670 // 7671 // map(s.ps) 7672 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7673 // 7674 // map(from: s.ps->s.i) 7675 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7676 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7677 // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7678 // 7679 // map(to: s.ps->ps) 7680 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7681 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7682 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | TO 7683 // 7684 // map(s.ps->ps->ps) 7685 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7686 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7687 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7688 // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7689 // 7690 // map(to: s.ps->ps->s.f[:22]) 7691 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7692 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7693 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7694 // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7695 // 7696 // map(ps) 7697 // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM 7698 // 7699 // map(ps->i) 7700 // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM 7701 // 7702 // map(ps->s.f) 7703 // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7704 // 7705 // map(from: ps->p) 7706 // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM 7707 // 7708 // map(to: ps->p[:22]) 7709 // ps, &(ps->p), sizeof(double*), TARGET_PARAM 7710 // ps, &(ps->p), sizeof(double*), MEMBER_OF(1) 7711 // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO 7712 // 7713 // map(ps->ps) 7714 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7715 // 7716 // map(from: ps->ps->s.i) 7717 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7718 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7719 // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7720 // 7721 // map(from: ps->ps->ps) 7722 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7723 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7724 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7725 // 7726 // map(ps->ps->ps->ps) 7727 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7728 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7729 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7730 // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7731 // 7732 // map(to: ps->ps->ps->s.f[:22]) 7733 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7734 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7735 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7736 // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7737 // 7738 // map(to: s.f[:22]) map(from: s.p[:33]) 7739 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) + 7740 // sizeof(double*) (**), TARGET_PARAM 7741 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO 7742 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) 7743 // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7744 // (*) allocate contiguous space needed to fit all mapped members even if 7745 // we allocate space for members not mapped (in this example, 7746 // s.f[22..49] and s.s are not mapped, yet we must allocate space for 7747 // them as well because they fall between &s.f[0] and &s.p) 7748 // 7749 // map(from: s.f[:22]) map(to: ps->p[:33]) 7750 // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM 7751 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7752 // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*) 7753 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO 7754 // (*) the struct this entry pertains to is the 2nd element in the list of 7755 // arguments, hence MEMBER_OF(2) 7756 // 7757 // map(from: s.f[:22], s.s) map(to: ps->p[:33]) 7758 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM 7759 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM 7760 // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM 7761 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7762 // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*) 7763 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO 7764 // (*) the struct this entry pertains to is the 4th element in the list 7765 // of arguments, hence MEMBER_OF(4) 7766 7767 // Track if the map information being generated is the first for a capture. 7768 bool IsCaptureFirstInfo = IsFirstComponentList; 7769 // When the variable is on a declare target link or in a to clause with 7770 // unified memory, a reference is needed to hold the host/device address 7771 // of the variable. 7772 bool RequiresReference = false; 7773 7774 // Scan the components from the base to the complete expression. 7775 auto CI = Components.rbegin(); 7776 auto CE = Components.rend(); 7777 auto I = CI; 7778 7779 // Track if the map information being generated is the first for a list of 7780 // components. 7781 bool IsExpressionFirstInfo = true; 7782 Address BP = Address::invalid(); 7783 const Expr *AssocExpr = I->getAssociatedExpression(); 7784 const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr); 7785 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 7786 7787 if (isa<MemberExpr>(AssocExpr)) { 7788 // The base is the 'this' pointer. The content of the pointer is going 7789 // to be the base of the field being mapped. 7790 BP = CGF.LoadCXXThisAddress(); 7791 } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) || 7792 (OASE && 7793 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) { 7794 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF); 7795 } else { 7796 // The base is the reference to the variable. 7797 // BP = &Var. 7798 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF); 7799 if (const auto *VD = 7800 dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) { 7801 if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 7802 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) { 7803 if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 7804 (*Res == OMPDeclareTargetDeclAttr::MT_To && 7805 CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) { 7806 RequiresReference = true; 7807 BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 7808 } 7809 } 7810 } 7811 7812 // If the variable is a pointer and is being dereferenced (i.e. is not 7813 // the last component), the base has to be the pointer itself, not its 7814 // reference. References are ignored for mapping purposes. 7815 QualType Ty = 7816 I->getAssociatedDeclaration()->getType().getNonReferenceType(); 7817 if (Ty->isAnyPointerType() && std::next(I) != CE) { 7818 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>()); 7819 7820 // We do not need to generate individual map information for the 7821 // pointer, it can be associated with the combined storage. 7822 ++I; 7823 } 7824 } 7825 7826 // Track whether a component of the list should be marked as MEMBER_OF some 7827 // combined entry (for partial structs). Only the first PTR_AND_OBJ entry 7828 // in a component list should be marked as MEMBER_OF, all subsequent entries 7829 // do not belong to the base struct. E.g. 7830 // struct S2 s; 7831 // s.ps->ps->ps->f[:] 7832 // (1) (2) (3) (4) 7833 // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a 7834 // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3) 7835 // is the pointee of ps(2) which is not member of struct s, so it should not 7836 // be marked as such (it is still PTR_AND_OBJ). 7837 // The variable is initialized to false so that PTR_AND_OBJ entries which 7838 // are not struct members are not considered (e.g. array of pointers to 7839 // data). 7840 bool ShouldBeMemberOf = false; 7841 7842 // Variable keeping track of whether or not we have encountered a component 7843 // in the component list which is a member expression. Useful when we have a 7844 // pointer or a final array section, in which case it is the previous 7845 // component in the list which tells us whether we have a member expression. 7846 // E.g. X.f[:] 7847 // While processing the final array section "[:]" it is "f" which tells us 7848 // whether we are dealing with a member of a declared struct. 7849 const MemberExpr *EncounteredME = nullptr; 7850 7851 for (; I != CE; ++I) { 7852 // If the current component is member of a struct (parent struct) mark it. 7853 if (!EncounteredME) { 7854 EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression()); 7855 // If we encounter a PTR_AND_OBJ entry from now on it should be marked 7856 // as MEMBER_OF the parent struct. 7857 if (EncounteredME) 7858 ShouldBeMemberOf = true; 7859 } 7860 7861 auto Next = std::next(I); 7862 7863 // We need to generate the addresses and sizes if this is the last 7864 // component, if the component is a pointer or if it is an array section 7865 // whose length can't be proved to be one. If this is a pointer, it 7866 // becomes the base address for the following components. 7867 7868 // A final array section, is one whose length can't be proved to be one. 7869 bool IsFinalArraySection = 7870 isFinalArraySectionExpression(I->getAssociatedExpression()); 7871 7872 // Get information on whether the element is a pointer. Have to do a 7873 // special treatment for array sections given that they are built-in 7874 // types. 7875 const auto *OASE = 7876 dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression()); 7877 const auto *UO = dyn_cast<UnaryOperator>(I->getAssociatedExpression()); 7878 const auto *BO = dyn_cast<BinaryOperator>(I->getAssociatedExpression()); 7879 bool IsPointer = 7880 (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE) 7881 .getCanonicalType() 7882 ->isAnyPointerType()) || 7883 I->getAssociatedExpression()->getType()->isAnyPointerType(); 7884 bool IsNonDerefPointer = IsPointer && !UO && !BO; 7885 7886 if (Next == CE || IsNonDerefPointer || IsFinalArraySection) { 7887 // If this is not the last component, we expect the pointer to be 7888 // associated with an array expression or member expression. 7889 assert((Next == CE || 7890 isa<MemberExpr>(Next->getAssociatedExpression()) || 7891 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) || 7892 isa<OMPArraySectionExpr>(Next->getAssociatedExpression()) || 7893 isa<UnaryOperator>(Next->getAssociatedExpression()) || 7894 isa<BinaryOperator>(Next->getAssociatedExpression())) && 7895 "Unexpected expression"); 7896 7897 Address LB = CGF.EmitOMPSharedLValue(I->getAssociatedExpression()) 7898 .getAddress(CGF); 7899 7900 // If this component is a pointer inside the base struct then we don't 7901 // need to create any entry for it - it will be combined with the object 7902 // it is pointing to into a single PTR_AND_OBJ entry. 7903 bool IsMemberPointer = 7904 IsPointer && EncounteredME && 7905 (dyn_cast<MemberExpr>(I->getAssociatedExpression()) == 7906 EncounteredME); 7907 if (!OverlappedElements.empty()) { 7908 // Handle base element with the info for overlapped elements. 7909 assert(!PartialStruct.Base.isValid() && "The base element is set."); 7910 assert(Next == CE && 7911 "Expected last element for the overlapped elements."); 7912 assert(!IsPointer && 7913 "Unexpected base element with the pointer type."); 7914 // Mark the whole struct as the struct that requires allocation on the 7915 // device. 7916 PartialStruct.LowestElem = {0, LB}; 7917 CharUnits TypeSize = CGF.getContext().getTypeSizeInChars( 7918 I->getAssociatedExpression()->getType()); 7919 Address HB = CGF.Builder.CreateConstGEP( 7920 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(LB, 7921 CGF.VoidPtrTy), 7922 TypeSize.getQuantity() - 1); 7923 PartialStruct.HighestElem = { 7924 std::numeric_limits<decltype( 7925 PartialStruct.HighestElem.first)>::max(), 7926 HB}; 7927 PartialStruct.Base = BP; 7928 // Emit data for non-overlapped data. 7929 OpenMPOffloadMappingFlags Flags = 7930 OMP_MAP_MEMBER_OF | 7931 getMapTypeBits(MapType, MapModifiers, IsImplicit, 7932 /*AddPtrFlag=*/false, 7933 /*AddIsTargetParamFlag=*/false); 7934 LB = BP; 7935 llvm::Value *Size = nullptr; 7936 // Do bitcopy of all non-overlapped structure elements. 7937 for (OMPClauseMappableExprCommon::MappableExprComponentListRef 7938 Component : OverlappedElements) { 7939 Address ComponentLB = Address::invalid(); 7940 for (const OMPClauseMappableExprCommon::MappableComponent &MC : 7941 Component) { 7942 if (MC.getAssociatedDeclaration()) { 7943 ComponentLB = 7944 CGF.EmitOMPSharedLValue(MC.getAssociatedExpression()) 7945 .getAddress(CGF); 7946 Size = CGF.Builder.CreatePtrDiff( 7947 CGF.EmitCastToVoidPtr(ComponentLB.getPointer()), 7948 CGF.EmitCastToVoidPtr(LB.getPointer())); 7949 break; 7950 } 7951 } 7952 BasePointers.push_back(BP.getPointer()); 7953 Pointers.push_back(LB.getPointer()); 7954 Sizes.push_back(CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, 7955 /*isSigned=*/true)); 7956 Types.push_back(Flags); 7957 LB = CGF.Builder.CreateConstGEP(ComponentLB, 1); 7958 } 7959 BasePointers.push_back(BP.getPointer()); 7960 Pointers.push_back(LB.getPointer()); 7961 Size = CGF.Builder.CreatePtrDiff( 7962 CGF.EmitCastToVoidPtr( 7963 CGF.Builder.CreateConstGEP(HB, 1).getPointer()), 7964 CGF.EmitCastToVoidPtr(LB.getPointer())); 7965 Sizes.push_back( 7966 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 7967 Types.push_back(Flags); 7968 break; 7969 } 7970 llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression()); 7971 if (!IsMemberPointer) { 7972 BasePointers.push_back(BP.getPointer()); 7973 Pointers.push_back(LB.getPointer()); 7974 Sizes.push_back( 7975 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 7976 7977 // We need to add a pointer flag for each map that comes from the 7978 // same expression except for the first one. We also need to signal 7979 // this map is the first one that relates with the current capture 7980 // (there is a set of entries for each capture). 7981 OpenMPOffloadMappingFlags Flags = getMapTypeBits( 7982 MapType, MapModifiers, IsImplicit, 7983 !IsExpressionFirstInfo || RequiresReference, 7984 IsCaptureFirstInfo && !RequiresReference); 7985 7986 if (!IsExpressionFirstInfo) { 7987 // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well, 7988 // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags. 7989 if (IsPointer) 7990 Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS | 7991 OMP_MAP_DELETE | OMP_MAP_CLOSE); 7992 7993 if (ShouldBeMemberOf) { 7994 // Set placeholder value MEMBER_OF=FFFF to indicate that the flag 7995 // should be later updated with the correct value of MEMBER_OF. 7996 Flags |= OMP_MAP_MEMBER_OF; 7997 // From now on, all subsequent PTR_AND_OBJ entries should not be 7998 // marked as MEMBER_OF. 7999 ShouldBeMemberOf = false; 8000 } 8001 } 8002 8003 Types.push_back(Flags); 8004 } 8005 8006 // If we have encountered a member expression so far, keep track of the 8007 // mapped member. If the parent is "*this", then the value declaration 8008 // is nullptr. 8009 if (EncounteredME) { 8010 const auto *FD = cast<FieldDecl>(EncounteredME->getMemberDecl()); 8011 unsigned FieldIndex = FD->getFieldIndex(); 8012 8013 // Update info about the lowest and highest elements for this struct 8014 if (!PartialStruct.Base.isValid()) { 8015 PartialStruct.LowestElem = {FieldIndex, LB}; 8016 PartialStruct.HighestElem = {FieldIndex, LB}; 8017 PartialStruct.Base = BP; 8018 } else if (FieldIndex < PartialStruct.LowestElem.first) { 8019 PartialStruct.LowestElem = {FieldIndex, LB}; 8020 } else if (FieldIndex > PartialStruct.HighestElem.first) { 8021 PartialStruct.HighestElem = {FieldIndex, LB}; 8022 } 8023 } 8024 8025 // If we have a final array section, we are done with this expression. 8026 if (IsFinalArraySection) 8027 break; 8028 8029 // The pointer becomes the base for the next element. 8030 if (Next != CE) 8031 BP = LB; 8032 8033 IsExpressionFirstInfo = false; 8034 IsCaptureFirstInfo = false; 8035 } 8036 } 8037 } 8038 8039 /// Return the adjusted map modifiers if the declaration a capture refers to 8040 /// appears in a first-private clause. This is expected to be used only with 8041 /// directives that start with 'target'. 8042 MappableExprsHandler::OpenMPOffloadMappingFlags 8043 getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const { 8044 assert(Cap.capturesVariable() && "Expected capture by reference only!"); 8045 8046 // A first private variable captured by reference will use only the 8047 // 'private ptr' and 'map to' flag. Return the right flags if the captured 8048 // declaration is known as first-private in this handler. 8049 if (FirstPrivateDecls.count(Cap.getCapturedVar())) { 8050 if (Cap.getCapturedVar()->getType().isConstant(CGF.getContext()) && 8051 Cap.getCaptureKind() == CapturedStmt::VCK_ByRef) 8052 return MappableExprsHandler::OMP_MAP_ALWAYS | 8053 MappableExprsHandler::OMP_MAP_TO; 8054 if (Cap.getCapturedVar()->getType()->isAnyPointerType()) 8055 return MappableExprsHandler::OMP_MAP_TO | 8056 MappableExprsHandler::OMP_MAP_PTR_AND_OBJ; 8057 return MappableExprsHandler::OMP_MAP_PRIVATE | 8058 MappableExprsHandler::OMP_MAP_TO; 8059 } 8060 return MappableExprsHandler::OMP_MAP_TO | 8061 MappableExprsHandler::OMP_MAP_FROM; 8062 } 8063 8064 static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) { 8065 // Rotate by getFlagMemberOffset() bits. 8066 return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1) 8067 << getFlagMemberOffset()); 8068 } 8069 8070 static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags, 8071 OpenMPOffloadMappingFlags MemberOfFlag) { 8072 // If the entry is PTR_AND_OBJ but has not been marked with the special 8073 // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be 8074 // marked as MEMBER_OF. 8075 if ((Flags & OMP_MAP_PTR_AND_OBJ) && 8076 ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF)) 8077 return; 8078 8079 // Reset the placeholder value to prepare the flag for the assignment of the 8080 // proper MEMBER_OF value. 8081 Flags &= ~OMP_MAP_MEMBER_OF; 8082 Flags |= MemberOfFlag; 8083 } 8084 8085 void getPlainLayout(const CXXRecordDecl *RD, 8086 llvm::SmallVectorImpl<const FieldDecl *> &Layout, 8087 bool AsBase) const { 8088 const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD); 8089 8090 llvm::StructType *St = 8091 AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType(); 8092 8093 unsigned NumElements = St->getNumElements(); 8094 llvm::SmallVector< 8095 llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4> 8096 RecordLayout(NumElements); 8097 8098 // Fill bases. 8099 for (const auto &I : RD->bases()) { 8100 if (I.isVirtual()) 8101 continue; 8102 const auto *Base = I.getType()->getAsCXXRecordDecl(); 8103 // Ignore empty bases. 8104 if (Base->isEmpty() || CGF.getContext() 8105 .getASTRecordLayout(Base) 8106 .getNonVirtualSize() 8107 .isZero()) 8108 continue; 8109 8110 unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base); 8111 RecordLayout[FieldIndex] = Base; 8112 } 8113 // Fill in virtual bases. 8114 for (const auto &I : RD->vbases()) { 8115 const auto *Base = I.getType()->getAsCXXRecordDecl(); 8116 // Ignore empty bases. 8117 if (Base->isEmpty()) 8118 continue; 8119 unsigned FieldIndex = RL.getVirtualBaseIndex(Base); 8120 if (RecordLayout[FieldIndex]) 8121 continue; 8122 RecordLayout[FieldIndex] = Base; 8123 } 8124 // Fill in all the fields. 8125 assert(!RD->isUnion() && "Unexpected union."); 8126 for (const auto *Field : RD->fields()) { 8127 // Fill in non-bitfields. (Bitfields always use a zero pattern, which we 8128 // will fill in later.) 8129 if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) { 8130 unsigned FieldIndex = RL.getLLVMFieldNo(Field); 8131 RecordLayout[FieldIndex] = Field; 8132 } 8133 } 8134 for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *> 8135 &Data : RecordLayout) { 8136 if (Data.isNull()) 8137 continue; 8138 if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>()) 8139 getPlainLayout(Base, Layout, /*AsBase=*/true); 8140 else 8141 Layout.push_back(Data.get<const FieldDecl *>()); 8142 } 8143 } 8144 8145 public: 8146 MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF) 8147 : CurDir(&Dir), CGF(CGF) { 8148 // Extract firstprivate clause information. 8149 for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>()) 8150 for (const auto *D : C->varlists()) 8151 FirstPrivateDecls.try_emplace( 8152 cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit()); 8153 // Extract device pointer clause information. 8154 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>()) 8155 for (auto L : C->component_lists()) 8156 DevPointersMap[L.first].push_back(L.second); 8157 } 8158 8159 /// Constructor for the declare mapper directive. 8160 MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF) 8161 : CurDir(&Dir), CGF(CGF) {} 8162 8163 /// Generate code for the combined entry if we have a partially mapped struct 8164 /// and take care of the mapping flags of the arguments corresponding to 8165 /// individual struct members. 8166 void emitCombinedEntry(MapBaseValuesArrayTy &BasePointers, 8167 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 8168 MapFlagsArrayTy &Types, MapFlagsArrayTy &CurTypes, 8169 const StructRangeInfoTy &PartialStruct) const { 8170 // Base is the base of the struct 8171 BasePointers.push_back(PartialStruct.Base.getPointer()); 8172 // Pointer is the address of the lowest element 8173 llvm::Value *LB = PartialStruct.LowestElem.second.getPointer(); 8174 Pointers.push_back(LB); 8175 // Size is (addr of {highest+1} element) - (addr of lowest element) 8176 llvm::Value *HB = PartialStruct.HighestElem.second.getPointer(); 8177 llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(HB, /*Idx0=*/1); 8178 llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy); 8179 llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy); 8180 llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr); 8181 llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty, 8182 /*isSigned=*/false); 8183 Sizes.push_back(Size); 8184 // Map type is always TARGET_PARAM 8185 Types.push_back(OMP_MAP_TARGET_PARAM); 8186 // Remove TARGET_PARAM flag from the first element 8187 (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM; 8188 8189 // All other current entries will be MEMBER_OF the combined entry 8190 // (except for PTR_AND_OBJ entries which do not have a placeholder value 8191 // 0xFFFF in the MEMBER_OF field). 8192 OpenMPOffloadMappingFlags MemberOfFlag = 8193 getMemberOfFlag(BasePointers.size() - 1); 8194 for (auto &M : CurTypes) 8195 setCorrectMemberOfFlag(M, MemberOfFlag); 8196 } 8197 8198 /// Generate all the base pointers, section pointers, sizes and map 8199 /// types for the extracted mappable expressions. Also, for each item that 8200 /// relates with a device pointer, a pair of the relevant declaration and 8201 /// index where it occurs is appended to the device pointers info array. 8202 void generateAllInfo(MapBaseValuesArrayTy &BasePointers, 8203 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 8204 MapFlagsArrayTy &Types) const { 8205 // We have to process the component lists that relate with the same 8206 // declaration in a single chunk so that we can generate the map flags 8207 // correctly. Therefore, we organize all lists in a map. 8208 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 8209 8210 // Helper function to fill the information map for the different supported 8211 // clauses. 8212 auto &&InfoGen = [&Info]( 8213 const ValueDecl *D, 8214 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 8215 OpenMPMapClauseKind MapType, 8216 ArrayRef<OpenMPMapModifierKind> MapModifiers, 8217 bool ReturnDevicePointer, bool IsImplicit) { 8218 const ValueDecl *VD = 8219 D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr; 8220 Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer, 8221 IsImplicit); 8222 }; 8223 8224 assert(CurDir.is<const OMPExecutableDirective *>() && 8225 "Expect a executable directive"); 8226 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8227 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) 8228 for (const auto L : C->component_lists()) { 8229 InfoGen(L.first, L.second, C->getMapType(), C->getMapTypeModifiers(), 8230 /*ReturnDevicePointer=*/false, C->isImplicit()); 8231 } 8232 for (const auto *C : CurExecDir->getClausesOfKind<OMPToClause>()) 8233 for (const auto L : C->component_lists()) { 8234 InfoGen(L.first, L.second, OMPC_MAP_to, llvm::None, 8235 /*ReturnDevicePointer=*/false, C->isImplicit()); 8236 } 8237 for (const auto *C : CurExecDir->getClausesOfKind<OMPFromClause>()) 8238 for (const auto L : C->component_lists()) { 8239 InfoGen(L.first, L.second, OMPC_MAP_from, llvm::None, 8240 /*ReturnDevicePointer=*/false, C->isImplicit()); 8241 } 8242 8243 // Look at the use_device_ptr clause information and mark the existing map 8244 // entries as such. If there is no map information for an entry in the 8245 // use_device_ptr list, we create one with map type 'alloc' and zero size 8246 // section. It is the user fault if that was not mapped before. If there is 8247 // no map information and the pointer is a struct member, then we defer the 8248 // emission of that entry until the whole struct has been processed. 8249 llvm::MapVector<const ValueDecl *, SmallVector<DeferredDevicePtrEntryTy, 4>> 8250 DeferredInfo; 8251 8252 for (const auto *C : 8253 CurExecDir->getClausesOfKind<OMPUseDevicePtrClause>()) { 8254 for (const auto L : C->component_lists()) { 8255 assert(!L.second.empty() && "Not expecting empty list of components!"); 8256 const ValueDecl *VD = L.second.back().getAssociatedDeclaration(); 8257 VD = cast<ValueDecl>(VD->getCanonicalDecl()); 8258 const Expr *IE = L.second.back().getAssociatedExpression(); 8259 // If the first component is a member expression, we have to look into 8260 // 'this', which maps to null in the map of map information. Otherwise 8261 // look directly for the information. 8262 auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD); 8263 8264 // We potentially have map information for this declaration already. 8265 // Look for the first set of components that refer to it. 8266 if (It != Info.end()) { 8267 auto CI = std::find_if( 8268 It->second.begin(), It->second.end(), [VD](const MapInfo &MI) { 8269 return MI.Components.back().getAssociatedDeclaration() == VD; 8270 }); 8271 // If we found a map entry, signal that the pointer has to be returned 8272 // and move on to the next declaration. 8273 if (CI != It->second.end()) { 8274 CI->ReturnDevicePointer = true; 8275 continue; 8276 } 8277 } 8278 8279 // We didn't find any match in our map information - generate a zero 8280 // size array section - if the pointer is a struct member we defer this 8281 // action until the whole struct has been processed. 8282 if (isa<MemberExpr>(IE)) { 8283 // Insert the pointer into Info to be processed by 8284 // generateInfoForComponentList. Because it is a member pointer 8285 // without a pointee, no entry will be generated for it, therefore 8286 // we need to generate one after the whole struct has been processed. 8287 // Nonetheless, generateInfoForComponentList must be called to take 8288 // the pointer into account for the calculation of the range of the 8289 // partial struct. 8290 InfoGen(nullptr, L.second, OMPC_MAP_unknown, llvm::None, 8291 /*ReturnDevicePointer=*/false, C->isImplicit()); 8292 DeferredInfo[nullptr].emplace_back(IE, VD); 8293 } else { 8294 llvm::Value *Ptr = 8295 CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc()); 8296 BasePointers.emplace_back(Ptr, VD); 8297 Pointers.push_back(Ptr); 8298 Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty)); 8299 Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_TARGET_PARAM); 8300 } 8301 } 8302 } 8303 8304 for (const auto &M : Info) { 8305 // We need to know when we generate information for the first component 8306 // associated with a capture, because the mapping flags depend on it. 8307 bool IsFirstComponentList = true; 8308 8309 // Temporary versions of arrays 8310 MapBaseValuesArrayTy CurBasePointers; 8311 MapValuesArrayTy CurPointers; 8312 MapValuesArrayTy CurSizes; 8313 MapFlagsArrayTy CurTypes; 8314 StructRangeInfoTy PartialStruct; 8315 8316 for (const MapInfo &L : M.second) { 8317 assert(!L.Components.empty() && 8318 "Not expecting declaration with no component lists."); 8319 8320 // Remember the current base pointer index. 8321 unsigned CurrentBasePointersIdx = CurBasePointers.size(); 8322 generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components, 8323 CurBasePointers, CurPointers, CurSizes, 8324 CurTypes, PartialStruct, 8325 IsFirstComponentList, L.IsImplicit); 8326 8327 // If this entry relates with a device pointer, set the relevant 8328 // declaration and add the 'return pointer' flag. 8329 if (L.ReturnDevicePointer) { 8330 assert(CurBasePointers.size() > CurrentBasePointersIdx && 8331 "Unexpected number of mapped base pointers."); 8332 8333 const ValueDecl *RelevantVD = 8334 L.Components.back().getAssociatedDeclaration(); 8335 assert(RelevantVD && 8336 "No relevant declaration related with device pointer??"); 8337 8338 CurBasePointers[CurrentBasePointersIdx].setDevicePtrDecl(RelevantVD); 8339 CurTypes[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM; 8340 } 8341 IsFirstComponentList = false; 8342 } 8343 8344 // Append any pending zero-length pointers which are struct members and 8345 // used with use_device_ptr. 8346 auto CI = DeferredInfo.find(M.first); 8347 if (CI != DeferredInfo.end()) { 8348 for (const DeferredDevicePtrEntryTy &L : CI->second) { 8349 llvm::Value *BasePtr = this->CGF.EmitLValue(L.IE).getPointer(CGF); 8350 llvm::Value *Ptr = this->CGF.EmitLoadOfScalar( 8351 this->CGF.EmitLValue(L.IE), L.IE->getExprLoc()); 8352 CurBasePointers.emplace_back(BasePtr, L.VD); 8353 CurPointers.push_back(Ptr); 8354 CurSizes.push_back(llvm::Constant::getNullValue(this->CGF.Int64Ty)); 8355 // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the placeholder 8356 // value MEMBER_OF=FFFF so that the entry is later updated with the 8357 // correct value of MEMBER_OF. 8358 CurTypes.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM | 8359 OMP_MAP_MEMBER_OF); 8360 } 8361 } 8362 8363 // If there is an entry in PartialStruct it means we have a struct with 8364 // individual members mapped. Emit an extra combined entry. 8365 if (PartialStruct.Base.isValid()) 8366 emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes, 8367 PartialStruct); 8368 8369 // We need to append the results of this capture to what we already have. 8370 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 8371 Pointers.append(CurPointers.begin(), CurPointers.end()); 8372 Sizes.append(CurSizes.begin(), CurSizes.end()); 8373 Types.append(CurTypes.begin(), CurTypes.end()); 8374 } 8375 } 8376 8377 /// Generate all the base pointers, section pointers, sizes and map types for 8378 /// the extracted map clauses of user-defined mapper. 8379 void generateAllInfoForMapper(MapBaseValuesArrayTy &BasePointers, 8380 MapValuesArrayTy &Pointers, 8381 MapValuesArrayTy &Sizes, 8382 MapFlagsArrayTy &Types) const { 8383 assert(CurDir.is<const OMPDeclareMapperDecl *>() && 8384 "Expect a declare mapper directive"); 8385 const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>(); 8386 // We have to process the component lists that relate with the same 8387 // declaration in a single chunk so that we can generate the map flags 8388 // correctly. Therefore, we organize all lists in a map. 8389 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 8390 8391 // Helper function to fill the information map for the different supported 8392 // clauses. 8393 auto &&InfoGen = [&Info]( 8394 const ValueDecl *D, 8395 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 8396 OpenMPMapClauseKind MapType, 8397 ArrayRef<OpenMPMapModifierKind> MapModifiers, 8398 bool ReturnDevicePointer, bool IsImplicit) { 8399 const ValueDecl *VD = 8400 D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr; 8401 Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer, 8402 IsImplicit); 8403 }; 8404 8405 for (const auto *C : CurMapperDir->clauselists()) { 8406 const auto *MC = cast<OMPMapClause>(C); 8407 for (const auto L : MC->component_lists()) { 8408 InfoGen(L.first, L.second, MC->getMapType(), MC->getMapTypeModifiers(), 8409 /*ReturnDevicePointer=*/false, MC->isImplicit()); 8410 } 8411 } 8412 8413 for (const auto &M : Info) { 8414 // We need to know when we generate information for the first component 8415 // associated with a capture, because the mapping flags depend on it. 8416 bool IsFirstComponentList = true; 8417 8418 // Temporary versions of arrays 8419 MapBaseValuesArrayTy CurBasePointers; 8420 MapValuesArrayTy CurPointers; 8421 MapValuesArrayTy CurSizes; 8422 MapFlagsArrayTy CurTypes; 8423 StructRangeInfoTy PartialStruct; 8424 8425 for (const MapInfo &L : M.second) { 8426 assert(!L.Components.empty() && 8427 "Not expecting declaration with no component lists."); 8428 generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components, 8429 CurBasePointers, CurPointers, CurSizes, 8430 CurTypes, PartialStruct, 8431 IsFirstComponentList, L.IsImplicit); 8432 IsFirstComponentList = false; 8433 } 8434 8435 // If there is an entry in PartialStruct it means we have a struct with 8436 // individual members mapped. Emit an extra combined entry. 8437 if (PartialStruct.Base.isValid()) 8438 emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes, 8439 PartialStruct); 8440 8441 // We need to append the results of this capture to what we already have. 8442 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 8443 Pointers.append(CurPointers.begin(), CurPointers.end()); 8444 Sizes.append(CurSizes.begin(), CurSizes.end()); 8445 Types.append(CurTypes.begin(), CurTypes.end()); 8446 } 8447 } 8448 8449 /// Emit capture info for lambdas for variables captured by reference. 8450 void generateInfoForLambdaCaptures( 8451 const ValueDecl *VD, llvm::Value *Arg, MapBaseValuesArrayTy &BasePointers, 8452 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 8453 MapFlagsArrayTy &Types, 8454 llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const { 8455 const auto *RD = VD->getType() 8456 .getCanonicalType() 8457 .getNonReferenceType() 8458 ->getAsCXXRecordDecl(); 8459 if (!RD || !RD->isLambda()) 8460 return; 8461 Address VDAddr = Address(Arg, CGF.getContext().getDeclAlign(VD)); 8462 LValue VDLVal = CGF.MakeAddrLValue( 8463 VDAddr, VD->getType().getCanonicalType().getNonReferenceType()); 8464 llvm::DenseMap<const VarDecl *, FieldDecl *> Captures; 8465 FieldDecl *ThisCapture = nullptr; 8466 RD->getCaptureFields(Captures, ThisCapture); 8467 if (ThisCapture) { 8468 LValue ThisLVal = 8469 CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture); 8470 LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture); 8471 LambdaPointers.try_emplace(ThisLVal.getPointer(CGF), 8472 VDLVal.getPointer(CGF)); 8473 BasePointers.push_back(ThisLVal.getPointer(CGF)); 8474 Pointers.push_back(ThisLValVal.getPointer(CGF)); 8475 Sizes.push_back( 8476 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy), 8477 CGF.Int64Ty, /*isSigned=*/true)); 8478 Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8479 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 8480 } 8481 for (const LambdaCapture &LC : RD->captures()) { 8482 if (!LC.capturesVariable()) 8483 continue; 8484 const VarDecl *VD = LC.getCapturedVar(); 8485 if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType()) 8486 continue; 8487 auto It = Captures.find(VD); 8488 assert(It != Captures.end() && "Found lambda capture without field."); 8489 LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second); 8490 if (LC.getCaptureKind() == LCK_ByRef) { 8491 LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second); 8492 LambdaPointers.try_emplace(VarLVal.getPointer(CGF), 8493 VDLVal.getPointer(CGF)); 8494 BasePointers.push_back(VarLVal.getPointer(CGF)); 8495 Pointers.push_back(VarLValVal.getPointer(CGF)); 8496 Sizes.push_back(CGF.Builder.CreateIntCast( 8497 CGF.getTypeSize( 8498 VD->getType().getCanonicalType().getNonReferenceType()), 8499 CGF.Int64Ty, /*isSigned=*/true)); 8500 } else { 8501 RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation()); 8502 LambdaPointers.try_emplace(VarLVal.getPointer(CGF), 8503 VDLVal.getPointer(CGF)); 8504 BasePointers.push_back(VarLVal.getPointer(CGF)); 8505 Pointers.push_back(VarRVal.getScalarVal()); 8506 Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0)); 8507 } 8508 Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8509 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 8510 } 8511 } 8512 8513 /// Set correct indices for lambdas captures. 8514 void adjustMemberOfForLambdaCaptures( 8515 const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers, 8516 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 8517 MapFlagsArrayTy &Types) const { 8518 for (unsigned I = 0, E = Types.size(); I < E; ++I) { 8519 // Set correct member_of idx for all implicit lambda captures. 8520 if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8521 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT)) 8522 continue; 8523 llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]); 8524 assert(BasePtr && "Unable to find base lambda address."); 8525 int TgtIdx = -1; 8526 for (unsigned J = I; J > 0; --J) { 8527 unsigned Idx = J - 1; 8528 if (Pointers[Idx] != BasePtr) 8529 continue; 8530 TgtIdx = Idx; 8531 break; 8532 } 8533 assert(TgtIdx != -1 && "Unable to find parent lambda."); 8534 // All other current entries will be MEMBER_OF the combined entry 8535 // (except for PTR_AND_OBJ entries which do not have a placeholder value 8536 // 0xFFFF in the MEMBER_OF field). 8537 OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx); 8538 setCorrectMemberOfFlag(Types[I], MemberOfFlag); 8539 } 8540 } 8541 8542 /// Generate the base pointers, section pointers, sizes and map types 8543 /// associated to a given capture. 8544 void generateInfoForCapture(const CapturedStmt::Capture *Cap, 8545 llvm::Value *Arg, 8546 MapBaseValuesArrayTy &BasePointers, 8547 MapValuesArrayTy &Pointers, 8548 MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types, 8549 StructRangeInfoTy &PartialStruct) const { 8550 assert(!Cap->capturesVariableArrayType() && 8551 "Not expecting to generate map info for a variable array type!"); 8552 8553 // We need to know when we generating information for the first component 8554 const ValueDecl *VD = Cap->capturesThis() 8555 ? nullptr 8556 : Cap->getCapturedVar()->getCanonicalDecl(); 8557 8558 // If this declaration appears in a is_device_ptr clause we just have to 8559 // pass the pointer by value. If it is a reference to a declaration, we just 8560 // pass its value. 8561 if (DevPointersMap.count(VD)) { 8562 BasePointers.emplace_back(Arg, VD); 8563 Pointers.push_back(Arg); 8564 Sizes.push_back( 8565 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy), 8566 CGF.Int64Ty, /*isSigned=*/true)); 8567 Types.push_back(OMP_MAP_LITERAL | OMP_MAP_TARGET_PARAM); 8568 return; 8569 } 8570 8571 using MapData = 8572 std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef, 8573 OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool>; 8574 SmallVector<MapData, 4> DeclComponentLists; 8575 assert(CurDir.is<const OMPExecutableDirective *>() && 8576 "Expect a executable directive"); 8577 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8578 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) { 8579 for (const auto L : C->decl_component_lists(VD)) { 8580 assert(L.first == VD && 8581 "We got information for the wrong declaration??"); 8582 assert(!L.second.empty() && 8583 "Not expecting declaration with no component lists."); 8584 DeclComponentLists.emplace_back(L.second, C->getMapType(), 8585 C->getMapTypeModifiers(), 8586 C->isImplicit()); 8587 } 8588 } 8589 8590 // Find overlapping elements (including the offset from the base element). 8591 llvm::SmallDenseMap< 8592 const MapData *, 8593 llvm::SmallVector< 8594 OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>, 8595 4> 8596 OverlappedData; 8597 size_t Count = 0; 8598 for (const MapData &L : DeclComponentLists) { 8599 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8600 OpenMPMapClauseKind MapType; 8601 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8602 bool IsImplicit; 8603 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8604 ++Count; 8605 for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) { 8606 OMPClauseMappableExprCommon::MappableExprComponentListRef Components1; 8607 std::tie(Components1, MapType, MapModifiers, IsImplicit) = L1; 8608 auto CI = Components.rbegin(); 8609 auto CE = Components.rend(); 8610 auto SI = Components1.rbegin(); 8611 auto SE = Components1.rend(); 8612 for (; CI != CE && SI != SE; ++CI, ++SI) { 8613 if (CI->getAssociatedExpression()->getStmtClass() != 8614 SI->getAssociatedExpression()->getStmtClass()) 8615 break; 8616 // Are we dealing with different variables/fields? 8617 if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration()) 8618 break; 8619 } 8620 // Found overlapping if, at least for one component, reached the head of 8621 // the components list. 8622 if (CI == CE || SI == SE) { 8623 assert((CI != CE || SI != SE) && 8624 "Unexpected full match of the mapping components."); 8625 const MapData &BaseData = CI == CE ? L : L1; 8626 OMPClauseMappableExprCommon::MappableExprComponentListRef SubData = 8627 SI == SE ? Components : Components1; 8628 auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData); 8629 OverlappedElements.getSecond().push_back(SubData); 8630 } 8631 } 8632 } 8633 // Sort the overlapped elements for each item. 8634 llvm::SmallVector<const FieldDecl *, 4> Layout; 8635 if (!OverlappedData.empty()) { 8636 if (const auto *CRD = 8637 VD->getType().getCanonicalType()->getAsCXXRecordDecl()) 8638 getPlainLayout(CRD, Layout, /*AsBase=*/false); 8639 else { 8640 const auto *RD = VD->getType().getCanonicalType()->getAsRecordDecl(); 8641 Layout.append(RD->field_begin(), RD->field_end()); 8642 } 8643 } 8644 for (auto &Pair : OverlappedData) { 8645 llvm::sort( 8646 Pair.getSecond(), 8647 [&Layout]( 8648 OMPClauseMappableExprCommon::MappableExprComponentListRef First, 8649 OMPClauseMappableExprCommon::MappableExprComponentListRef 8650 Second) { 8651 auto CI = First.rbegin(); 8652 auto CE = First.rend(); 8653 auto SI = Second.rbegin(); 8654 auto SE = Second.rend(); 8655 for (; CI != CE && SI != SE; ++CI, ++SI) { 8656 if (CI->getAssociatedExpression()->getStmtClass() != 8657 SI->getAssociatedExpression()->getStmtClass()) 8658 break; 8659 // Are we dealing with different variables/fields? 8660 if (CI->getAssociatedDeclaration() != 8661 SI->getAssociatedDeclaration()) 8662 break; 8663 } 8664 8665 // Lists contain the same elements. 8666 if (CI == CE && SI == SE) 8667 return false; 8668 8669 // List with less elements is less than list with more elements. 8670 if (CI == CE || SI == SE) 8671 return CI == CE; 8672 8673 const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration()); 8674 const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration()); 8675 if (FD1->getParent() == FD2->getParent()) 8676 return FD1->getFieldIndex() < FD2->getFieldIndex(); 8677 const auto It = 8678 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) { 8679 return FD == FD1 || FD == FD2; 8680 }); 8681 return *It == FD1; 8682 }); 8683 } 8684 8685 // Associated with a capture, because the mapping flags depend on it. 8686 // Go through all of the elements with the overlapped elements. 8687 for (const auto &Pair : OverlappedData) { 8688 const MapData &L = *Pair.getFirst(); 8689 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8690 OpenMPMapClauseKind MapType; 8691 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8692 bool IsImplicit; 8693 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8694 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 8695 OverlappedComponents = Pair.getSecond(); 8696 bool IsFirstComponentList = true; 8697 generateInfoForComponentList(MapType, MapModifiers, Components, 8698 BasePointers, Pointers, Sizes, Types, 8699 PartialStruct, IsFirstComponentList, 8700 IsImplicit, OverlappedComponents); 8701 } 8702 // Go through other elements without overlapped elements. 8703 bool IsFirstComponentList = OverlappedData.empty(); 8704 for (const MapData &L : DeclComponentLists) { 8705 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8706 OpenMPMapClauseKind MapType; 8707 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8708 bool IsImplicit; 8709 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8710 auto It = OverlappedData.find(&L); 8711 if (It == OverlappedData.end()) 8712 generateInfoForComponentList(MapType, MapModifiers, Components, 8713 BasePointers, Pointers, Sizes, Types, 8714 PartialStruct, IsFirstComponentList, 8715 IsImplicit); 8716 IsFirstComponentList = false; 8717 } 8718 } 8719 8720 /// Generate the base pointers, section pointers, sizes and map types 8721 /// associated with the declare target link variables. 8722 void generateInfoForDeclareTargetLink(MapBaseValuesArrayTy &BasePointers, 8723 MapValuesArrayTy &Pointers, 8724 MapValuesArrayTy &Sizes, 8725 MapFlagsArrayTy &Types) const { 8726 assert(CurDir.is<const OMPExecutableDirective *>() && 8727 "Expect a executable directive"); 8728 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8729 // Map other list items in the map clause which are not captured variables 8730 // but "declare target link" global variables. 8731 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) { 8732 for (const auto L : C->component_lists()) { 8733 if (!L.first) 8734 continue; 8735 const auto *VD = dyn_cast<VarDecl>(L.first); 8736 if (!VD) 8737 continue; 8738 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 8739 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 8740 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() || 8741 !Res || *Res != OMPDeclareTargetDeclAttr::MT_Link) 8742 continue; 8743 StructRangeInfoTy PartialStruct; 8744 generateInfoForComponentList( 8745 C->getMapType(), C->getMapTypeModifiers(), L.second, BasePointers, 8746 Pointers, Sizes, Types, PartialStruct, 8747 /*IsFirstComponentList=*/true, C->isImplicit()); 8748 assert(!PartialStruct.Base.isValid() && 8749 "No partial structs for declare target link expected."); 8750 } 8751 } 8752 } 8753 8754 /// Generate the default map information for a given capture \a CI, 8755 /// record field declaration \a RI and captured value \a CV. 8756 void generateDefaultMapInfo(const CapturedStmt::Capture &CI, 8757 const FieldDecl &RI, llvm::Value *CV, 8758 MapBaseValuesArrayTy &CurBasePointers, 8759 MapValuesArrayTy &CurPointers, 8760 MapValuesArrayTy &CurSizes, 8761 MapFlagsArrayTy &CurMapTypes) const { 8762 bool IsImplicit = true; 8763 // Do the default mapping. 8764 if (CI.capturesThis()) { 8765 CurBasePointers.push_back(CV); 8766 CurPointers.push_back(CV); 8767 const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr()); 8768 CurSizes.push_back( 8769 CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()), 8770 CGF.Int64Ty, /*isSigned=*/true)); 8771 // Default map type. 8772 CurMapTypes.push_back(OMP_MAP_TO | OMP_MAP_FROM); 8773 } else if (CI.capturesVariableByCopy()) { 8774 CurBasePointers.push_back(CV); 8775 CurPointers.push_back(CV); 8776 if (!RI.getType()->isAnyPointerType()) { 8777 // We have to signal to the runtime captures passed by value that are 8778 // not pointers. 8779 CurMapTypes.push_back(OMP_MAP_LITERAL); 8780 CurSizes.push_back(CGF.Builder.CreateIntCast( 8781 CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true)); 8782 } else { 8783 // Pointers are implicitly mapped with a zero size and no flags 8784 // (other than first map that is added for all implicit maps). 8785 CurMapTypes.push_back(OMP_MAP_NONE); 8786 CurSizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty)); 8787 } 8788 const VarDecl *VD = CI.getCapturedVar(); 8789 auto I = FirstPrivateDecls.find(VD); 8790 if (I != FirstPrivateDecls.end()) 8791 IsImplicit = I->getSecond(); 8792 } else { 8793 assert(CI.capturesVariable() && "Expected captured reference."); 8794 const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr()); 8795 QualType ElementType = PtrTy->getPointeeType(); 8796 CurSizes.push_back(CGF.Builder.CreateIntCast( 8797 CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true)); 8798 // The default map type for a scalar/complex type is 'to' because by 8799 // default the value doesn't have to be retrieved. For an aggregate 8800 // type, the default is 'tofrom'. 8801 CurMapTypes.push_back(getMapModifiersForPrivateClauses(CI)); 8802 const VarDecl *VD = CI.getCapturedVar(); 8803 auto I = FirstPrivateDecls.find(VD); 8804 if (I != FirstPrivateDecls.end() && 8805 VD->getType().isConstant(CGF.getContext())) { 8806 llvm::Constant *Addr = 8807 CGF.CGM.getOpenMPRuntime().registerTargetFirstprivateCopy(CGF, VD); 8808 // Copy the value of the original variable to the new global copy. 8809 CGF.Builder.CreateMemCpy( 8810 CGF.MakeNaturalAlignAddrLValue(Addr, ElementType).getAddress(CGF), 8811 Address(CV, CGF.getContext().getTypeAlignInChars(ElementType)), 8812 CurSizes.back(), /*IsVolatile=*/false); 8813 // Use new global variable as the base pointers. 8814 CurBasePointers.push_back(Addr); 8815 CurPointers.push_back(Addr); 8816 } else { 8817 CurBasePointers.push_back(CV); 8818 if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) { 8819 Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue( 8820 CV, ElementType, CGF.getContext().getDeclAlign(VD), 8821 AlignmentSource::Decl)); 8822 CurPointers.push_back(PtrAddr.getPointer()); 8823 } else { 8824 CurPointers.push_back(CV); 8825 } 8826 } 8827 if (I != FirstPrivateDecls.end()) 8828 IsImplicit = I->getSecond(); 8829 } 8830 // Every default map produces a single argument which is a target parameter. 8831 CurMapTypes.back() |= OMP_MAP_TARGET_PARAM; 8832 8833 // Add flag stating this is an implicit map. 8834 if (IsImplicit) 8835 CurMapTypes.back() |= OMP_MAP_IMPLICIT; 8836 } 8837 }; 8838 } // anonymous namespace 8839 8840 /// Emit the arrays used to pass the captures and map information to the 8841 /// offloading runtime library. If there is no map or capture information, 8842 /// return nullptr by reference. 8843 static void 8844 emitOffloadingArrays(CodeGenFunction &CGF, 8845 MappableExprsHandler::MapBaseValuesArrayTy &BasePointers, 8846 MappableExprsHandler::MapValuesArrayTy &Pointers, 8847 MappableExprsHandler::MapValuesArrayTy &Sizes, 8848 MappableExprsHandler::MapFlagsArrayTy &MapTypes, 8849 CGOpenMPRuntime::TargetDataInfo &Info) { 8850 CodeGenModule &CGM = CGF.CGM; 8851 ASTContext &Ctx = CGF.getContext(); 8852 8853 // Reset the array information. 8854 Info.clearArrayInfo(); 8855 Info.NumberOfPtrs = BasePointers.size(); 8856 8857 if (Info.NumberOfPtrs) { 8858 // Detect if we have any capture size requiring runtime evaluation of the 8859 // size so that a constant array could be eventually used. 8860 bool hasRuntimeEvaluationCaptureSize = false; 8861 for (llvm::Value *S : Sizes) 8862 if (!isa<llvm::Constant>(S)) { 8863 hasRuntimeEvaluationCaptureSize = true; 8864 break; 8865 } 8866 8867 llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true); 8868 QualType PointerArrayType = Ctx.getConstantArrayType( 8869 Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal, 8870 /*IndexTypeQuals=*/0); 8871 8872 Info.BasePointersArray = 8873 CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer(); 8874 Info.PointersArray = 8875 CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer(); 8876 8877 // If we don't have any VLA types or other types that require runtime 8878 // evaluation, we can use a constant array for the map sizes, otherwise we 8879 // need to fill up the arrays as we do for the pointers. 8880 QualType Int64Ty = 8881 Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 8882 if (hasRuntimeEvaluationCaptureSize) { 8883 QualType SizeArrayType = Ctx.getConstantArrayType( 8884 Int64Ty, PointerNumAP, nullptr, ArrayType::Normal, 8885 /*IndexTypeQuals=*/0); 8886 Info.SizesArray = 8887 CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer(); 8888 } else { 8889 // We expect all the sizes to be constant, so we collect them to create 8890 // a constant array. 8891 SmallVector<llvm::Constant *, 16> ConstSizes; 8892 for (llvm::Value *S : Sizes) 8893 ConstSizes.push_back(cast<llvm::Constant>(S)); 8894 8895 auto *SizesArrayInit = llvm::ConstantArray::get( 8896 llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes); 8897 std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"}); 8898 auto *SizesArrayGbl = new llvm::GlobalVariable( 8899 CGM.getModule(), SizesArrayInit->getType(), 8900 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 8901 SizesArrayInit, Name); 8902 SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 8903 Info.SizesArray = SizesArrayGbl; 8904 } 8905 8906 // The map types are always constant so we don't need to generate code to 8907 // fill arrays. Instead, we create an array constant. 8908 SmallVector<uint64_t, 4> Mapping(MapTypes.size(), 0); 8909 llvm::copy(MapTypes, Mapping.begin()); 8910 llvm::Constant *MapTypesArrayInit = 8911 llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping); 8912 std::string MaptypesName = 8913 CGM.getOpenMPRuntime().getName({"offload_maptypes"}); 8914 auto *MapTypesArrayGbl = new llvm::GlobalVariable( 8915 CGM.getModule(), MapTypesArrayInit->getType(), 8916 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 8917 MapTypesArrayInit, MaptypesName); 8918 MapTypesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 8919 Info.MapTypesArray = MapTypesArrayGbl; 8920 8921 for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) { 8922 llvm::Value *BPVal = *BasePointers[I]; 8923 llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32( 8924 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8925 Info.BasePointersArray, 0, I); 8926 BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8927 BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0)); 8928 Address BPAddr(BP, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 8929 CGF.Builder.CreateStore(BPVal, BPAddr); 8930 8931 if (Info.requiresDevicePointerInfo()) 8932 if (const ValueDecl *DevVD = BasePointers[I].getDevicePtrDecl()) 8933 Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr); 8934 8935 llvm::Value *PVal = Pointers[I]; 8936 llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32( 8937 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8938 Info.PointersArray, 0, I); 8939 P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8940 P, PVal->getType()->getPointerTo(/*AddrSpace=*/0)); 8941 Address PAddr(P, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 8942 CGF.Builder.CreateStore(PVal, PAddr); 8943 8944 if (hasRuntimeEvaluationCaptureSize) { 8945 llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32( 8946 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 8947 Info.SizesArray, 8948 /*Idx0=*/0, 8949 /*Idx1=*/I); 8950 Address SAddr(S, Ctx.getTypeAlignInChars(Int64Ty)); 8951 CGF.Builder.CreateStore( 8952 CGF.Builder.CreateIntCast(Sizes[I], CGM.Int64Ty, /*isSigned=*/true), 8953 SAddr); 8954 } 8955 } 8956 } 8957 } 8958 8959 /// Emit the arguments to be passed to the runtime library based on the 8960 /// arrays of pointers, sizes and map types. 8961 static void emitOffloadingArraysArgument( 8962 CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg, 8963 llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg, 8964 llvm::Value *&MapTypesArrayArg, CGOpenMPRuntime::TargetDataInfo &Info) { 8965 CodeGenModule &CGM = CGF.CGM; 8966 if (Info.NumberOfPtrs) { 8967 BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8968 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8969 Info.BasePointersArray, 8970 /*Idx0=*/0, /*Idx1=*/0); 8971 PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8972 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8973 Info.PointersArray, 8974 /*Idx0=*/0, 8975 /*Idx1=*/0); 8976 SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8977 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray, 8978 /*Idx0=*/0, /*Idx1=*/0); 8979 MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8980 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 8981 Info.MapTypesArray, 8982 /*Idx0=*/0, 8983 /*Idx1=*/0); 8984 } else { 8985 BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 8986 PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 8987 SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 8988 MapTypesArrayArg = 8989 llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 8990 } 8991 } 8992 8993 /// Check for inner distribute directive. 8994 static const OMPExecutableDirective * 8995 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) { 8996 const auto *CS = D.getInnermostCapturedStmt(); 8997 const auto *Body = 8998 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 8999 const Stmt *ChildStmt = 9000 CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 9001 9002 if (const auto *NestedDir = 9003 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 9004 OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind(); 9005 switch (D.getDirectiveKind()) { 9006 case OMPD_target: 9007 if (isOpenMPDistributeDirective(DKind)) 9008 return NestedDir; 9009 if (DKind == OMPD_teams) { 9010 Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers( 9011 /*IgnoreCaptured=*/true); 9012 if (!Body) 9013 return nullptr; 9014 ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 9015 if (const auto *NND = 9016 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 9017 DKind = NND->getDirectiveKind(); 9018 if (isOpenMPDistributeDirective(DKind)) 9019 return NND; 9020 } 9021 } 9022 return nullptr; 9023 case OMPD_target_teams: 9024 if (isOpenMPDistributeDirective(DKind)) 9025 return NestedDir; 9026 return nullptr; 9027 case OMPD_target_parallel: 9028 case OMPD_target_simd: 9029 case OMPD_target_parallel_for: 9030 case OMPD_target_parallel_for_simd: 9031 return nullptr; 9032 case OMPD_target_teams_distribute: 9033 case OMPD_target_teams_distribute_simd: 9034 case OMPD_target_teams_distribute_parallel_for: 9035 case OMPD_target_teams_distribute_parallel_for_simd: 9036 case OMPD_parallel: 9037 case OMPD_for: 9038 case OMPD_parallel_for: 9039 case OMPD_parallel_master: 9040 case OMPD_parallel_sections: 9041 case OMPD_for_simd: 9042 case OMPD_parallel_for_simd: 9043 case OMPD_cancel: 9044 case OMPD_cancellation_point: 9045 case OMPD_ordered: 9046 case OMPD_threadprivate: 9047 case OMPD_allocate: 9048 case OMPD_task: 9049 case OMPD_simd: 9050 case OMPD_sections: 9051 case OMPD_section: 9052 case OMPD_single: 9053 case OMPD_master: 9054 case OMPD_critical: 9055 case OMPD_taskyield: 9056 case OMPD_barrier: 9057 case OMPD_taskwait: 9058 case OMPD_taskgroup: 9059 case OMPD_atomic: 9060 case OMPD_flush: 9061 case OMPD_depobj: 9062 case OMPD_scan: 9063 case OMPD_teams: 9064 case OMPD_target_data: 9065 case OMPD_target_exit_data: 9066 case OMPD_target_enter_data: 9067 case OMPD_distribute: 9068 case OMPD_distribute_simd: 9069 case OMPD_distribute_parallel_for: 9070 case OMPD_distribute_parallel_for_simd: 9071 case OMPD_teams_distribute: 9072 case OMPD_teams_distribute_simd: 9073 case OMPD_teams_distribute_parallel_for: 9074 case OMPD_teams_distribute_parallel_for_simd: 9075 case OMPD_target_update: 9076 case OMPD_declare_simd: 9077 case OMPD_declare_variant: 9078 case OMPD_begin_declare_variant: 9079 case OMPD_end_declare_variant: 9080 case OMPD_declare_target: 9081 case OMPD_end_declare_target: 9082 case OMPD_declare_reduction: 9083 case OMPD_declare_mapper: 9084 case OMPD_taskloop: 9085 case OMPD_taskloop_simd: 9086 case OMPD_master_taskloop: 9087 case OMPD_master_taskloop_simd: 9088 case OMPD_parallel_master_taskloop: 9089 case OMPD_parallel_master_taskloop_simd: 9090 case OMPD_requires: 9091 case OMPD_unknown: 9092 llvm_unreachable("Unexpected directive."); 9093 } 9094 } 9095 9096 return nullptr; 9097 } 9098 9099 /// Emit the user-defined mapper function. The code generation follows the 9100 /// pattern in the example below. 9101 /// \code 9102 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle, 9103 /// void *base, void *begin, 9104 /// int64_t size, int64_t type) { 9105 /// // Allocate space for an array section first. 9106 /// if (size > 1 && !maptype.IsDelete) 9107 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 9108 /// size*sizeof(Ty), clearToFrom(type)); 9109 /// // Map members. 9110 /// for (unsigned i = 0; i < size; i++) { 9111 /// // For each component specified by this mapper: 9112 /// for (auto c : all_components) { 9113 /// if (c.hasMapper()) 9114 /// (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size, 9115 /// c.arg_type); 9116 /// else 9117 /// __tgt_push_mapper_component(rt_mapper_handle, c.arg_base, 9118 /// c.arg_begin, c.arg_size, c.arg_type); 9119 /// } 9120 /// } 9121 /// // Delete the array section. 9122 /// if (size > 1 && maptype.IsDelete) 9123 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 9124 /// size*sizeof(Ty), clearToFrom(type)); 9125 /// } 9126 /// \endcode 9127 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D, 9128 CodeGenFunction *CGF) { 9129 if (UDMMap.count(D) > 0) 9130 return; 9131 ASTContext &C = CGM.getContext(); 9132 QualType Ty = D->getType(); 9133 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 9134 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 9135 auto *MapperVarDecl = 9136 cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl()); 9137 SourceLocation Loc = D->getLocation(); 9138 CharUnits ElementSize = C.getTypeSizeInChars(Ty); 9139 9140 // Prepare mapper function arguments and attributes. 9141 ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 9142 C.VoidPtrTy, ImplicitParamDecl::Other); 9143 ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 9144 ImplicitParamDecl::Other); 9145 ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 9146 C.VoidPtrTy, ImplicitParamDecl::Other); 9147 ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 9148 ImplicitParamDecl::Other); 9149 ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 9150 ImplicitParamDecl::Other); 9151 FunctionArgList Args; 9152 Args.push_back(&HandleArg); 9153 Args.push_back(&BaseArg); 9154 Args.push_back(&BeginArg); 9155 Args.push_back(&SizeArg); 9156 Args.push_back(&TypeArg); 9157 const CGFunctionInfo &FnInfo = 9158 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 9159 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 9160 SmallString<64> TyStr; 9161 llvm::raw_svector_ostream Out(TyStr); 9162 CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out); 9163 std::string Name = getName({"omp_mapper", TyStr, D->getName()}); 9164 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 9165 Name, &CGM.getModule()); 9166 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 9167 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 9168 // Start the mapper function code generation. 9169 CodeGenFunction MapperCGF(CGM); 9170 MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 9171 // Compute the starting and end addreses of array elements. 9172 llvm::Value *Size = MapperCGF.EmitLoadOfScalar( 9173 MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false, 9174 C.getPointerType(Int64Ty), Loc); 9175 llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast( 9176 MapperCGF.GetAddrOfLocalVar(&BeginArg).getPointer(), 9177 CGM.getTypes().ConvertTypeForMem(C.getPointerType(PtrTy))); 9178 llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(PtrBegin, Size); 9179 llvm::Value *MapType = MapperCGF.EmitLoadOfScalar( 9180 MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false, 9181 C.getPointerType(Int64Ty), Loc); 9182 // Prepare common arguments for array initiation and deletion. 9183 llvm::Value *Handle = MapperCGF.EmitLoadOfScalar( 9184 MapperCGF.GetAddrOfLocalVar(&HandleArg), 9185 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9186 llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar( 9187 MapperCGF.GetAddrOfLocalVar(&BaseArg), 9188 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9189 llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar( 9190 MapperCGF.GetAddrOfLocalVar(&BeginArg), 9191 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9192 9193 // Emit array initiation if this is an array section and \p MapType indicates 9194 // that memory allocation is required. 9195 llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head"); 9196 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 9197 ElementSize, HeadBB, /*IsInit=*/true); 9198 9199 // Emit a for loop to iterate through SizeArg of elements and map all of them. 9200 9201 // Emit the loop header block. 9202 MapperCGF.EmitBlock(HeadBB); 9203 llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body"); 9204 llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done"); 9205 // Evaluate whether the initial condition is satisfied. 9206 llvm::Value *IsEmpty = 9207 MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty"); 9208 MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 9209 llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock(); 9210 9211 // Emit the loop body block. 9212 MapperCGF.EmitBlock(BodyBB); 9213 llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI( 9214 PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent"); 9215 PtrPHI->addIncoming(PtrBegin, EntryBB); 9216 Address PtrCurrent = 9217 Address(PtrPHI, MapperCGF.GetAddrOfLocalVar(&BeginArg) 9218 .getAlignment() 9219 .alignmentOfArrayElement(ElementSize)); 9220 // Privatize the declared variable of mapper to be the current array element. 9221 CodeGenFunction::OMPPrivateScope Scope(MapperCGF); 9222 Scope.addPrivate(MapperVarDecl, [&MapperCGF, PtrCurrent, PtrTy]() { 9223 return MapperCGF 9224 .EmitLoadOfPointerLValue(PtrCurrent, PtrTy->castAs<PointerType>()) 9225 .getAddress(MapperCGF); 9226 }); 9227 (void)Scope.Privatize(); 9228 9229 // Get map clause information. Fill up the arrays with all mapped variables. 9230 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 9231 MappableExprsHandler::MapValuesArrayTy Pointers; 9232 MappableExprsHandler::MapValuesArrayTy Sizes; 9233 MappableExprsHandler::MapFlagsArrayTy MapTypes; 9234 MappableExprsHandler MEHandler(*D, MapperCGF); 9235 MEHandler.generateAllInfoForMapper(BasePointers, Pointers, Sizes, MapTypes); 9236 9237 // Call the runtime API __tgt_mapper_num_components to get the number of 9238 // pre-existing components. 9239 llvm::Value *OffloadingArgs[] = {Handle}; 9240 llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall( 9241 createRuntimeFunction(OMPRTL__tgt_mapper_num_components), OffloadingArgs); 9242 llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl( 9243 PreviousSize, 9244 MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset())); 9245 9246 // Fill up the runtime mapper handle for all components. 9247 for (unsigned I = 0; I < BasePointers.size(); ++I) { 9248 llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast( 9249 *BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 9250 llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast( 9251 Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 9252 llvm::Value *CurSizeArg = Sizes[I]; 9253 9254 // Extract the MEMBER_OF field from the map type. 9255 llvm::BasicBlock *MemberBB = MapperCGF.createBasicBlock("omp.member"); 9256 MapperCGF.EmitBlock(MemberBB); 9257 llvm::Value *OriMapType = MapperCGF.Builder.getInt64(MapTypes[I]); 9258 llvm::Value *Member = MapperCGF.Builder.CreateAnd( 9259 OriMapType, 9260 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_MEMBER_OF)); 9261 llvm::BasicBlock *MemberCombineBB = 9262 MapperCGF.createBasicBlock("omp.member.combine"); 9263 llvm::BasicBlock *TypeBB = MapperCGF.createBasicBlock("omp.type"); 9264 llvm::Value *IsMember = MapperCGF.Builder.CreateIsNull(Member); 9265 MapperCGF.Builder.CreateCondBr(IsMember, TypeBB, MemberCombineBB); 9266 // Add the number of pre-existing components to the MEMBER_OF field if it 9267 // is valid. 9268 MapperCGF.EmitBlock(MemberCombineBB); 9269 llvm::Value *CombinedMember = 9270 MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize); 9271 // Do nothing if it is not a member of previous components. 9272 MapperCGF.EmitBlock(TypeBB); 9273 llvm::PHINode *MemberMapType = 9274 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.membermaptype"); 9275 MemberMapType->addIncoming(OriMapType, MemberBB); 9276 MemberMapType->addIncoming(CombinedMember, MemberCombineBB); 9277 9278 // Combine the map type inherited from user-defined mapper with that 9279 // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM 9280 // bits of the \a MapType, which is the input argument of the mapper 9281 // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM 9282 // bits of MemberMapType. 9283 // [OpenMP 5.0], 1.2.6. map-type decay. 9284 // | alloc | to | from | tofrom | release | delete 9285 // ---------------------------------------------------------- 9286 // alloc | alloc | alloc | alloc | alloc | release | delete 9287 // to | alloc | to | alloc | to | release | delete 9288 // from | alloc | alloc | from | from | release | delete 9289 // tofrom | alloc | to | from | tofrom | release | delete 9290 llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd( 9291 MapType, 9292 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO | 9293 MappableExprsHandler::OMP_MAP_FROM)); 9294 llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc"); 9295 llvm::BasicBlock *AllocElseBB = 9296 MapperCGF.createBasicBlock("omp.type.alloc.else"); 9297 llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to"); 9298 llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else"); 9299 llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from"); 9300 llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end"); 9301 llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom); 9302 MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB); 9303 // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM. 9304 MapperCGF.EmitBlock(AllocBB); 9305 llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd( 9306 MemberMapType, 9307 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 9308 MappableExprsHandler::OMP_MAP_FROM))); 9309 MapperCGF.Builder.CreateBr(EndBB); 9310 MapperCGF.EmitBlock(AllocElseBB); 9311 llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ( 9312 LeftToFrom, 9313 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO)); 9314 MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB); 9315 // In case of to, clear OMP_MAP_FROM. 9316 MapperCGF.EmitBlock(ToBB); 9317 llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd( 9318 MemberMapType, 9319 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM)); 9320 MapperCGF.Builder.CreateBr(EndBB); 9321 MapperCGF.EmitBlock(ToElseBB); 9322 llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ( 9323 LeftToFrom, 9324 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM)); 9325 MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB); 9326 // In case of from, clear OMP_MAP_TO. 9327 MapperCGF.EmitBlock(FromBB); 9328 llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd( 9329 MemberMapType, 9330 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO)); 9331 // In case of tofrom, do nothing. 9332 MapperCGF.EmitBlock(EndBB); 9333 llvm::PHINode *CurMapType = 9334 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype"); 9335 CurMapType->addIncoming(AllocMapType, AllocBB); 9336 CurMapType->addIncoming(ToMapType, ToBB); 9337 CurMapType->addIncoming(FromMapType, FromBB); 9338 CurMapType->addIncoming(MemberMapType, ToElseBB); 9339 9340 // TODO: call the corresponding mapper function if a user-defined mapper is 9341 // associated with this map clause. 9342 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 9343 // data structure. 9344 llvm::Value *OffloadingArgs[] = {Handle, CurBaseArg, CurBeginArg, 9345 CurSizeArg, CurMapType}; 9346 MapperCGF.EmitRuntimeCall( 9347 createRuntimeFunction(OMPRTL__tgt_push_mapper_component), 9348 OffloadingArgs); 9349 } 9350 9351 // Update the pointer to point to the next element that needs to be mapped, 9352 // and check whether we have mapped all elements. 9353 llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32( 9354 PtrPHI, /*Idx0=*/1, "omp.arraymap.next"); 9355 PtrPHI->addIncoming(PtrNext, BodyBB); 9356 llvm::Value *IsDone = 9357 MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone"); 9358 llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit"); 9359 MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB); 9360 9361 MapperCGF.EmitBlock(ExitBB); 9362 // Emit array deletion if this is an array section and \p MapType indicates 9363 // that deletion is required. 9364 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 9365 ElementSize, DoneBB, /*IsInit=*/false); 9366 9367 // Emit the function exit block. 9368 MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true); 9369 MapperCGF.FinishFunction(); 9370 UDMMap.try_emplace(D, Fn); 9371 if (CGF) { 9372 auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn); 9373 Decls.second.push_back(D); 9374 } 9375 } 9376 9377 /// Emit the array initialization or deletion portion for user-defined mapper 9378 /// code generation. First, it evaluates whether an array section is mapped and 9379 /// whether the \a MapType instructs to delete this section. If \a IsInit is 9380 /// true, and \a MapType indicates to not delete this array, array 9381 /// initialization code is generated. If \a IsInit is false, and \a MapType 9382 /// indicates to not this array, array deletion code is generated. 9383 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel( 9384 CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base, 9385 llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType, 9386 CharUnits ElementSize, llvm::BasicBlock *ExitBB, bool IsInit) { 9387 StringRef Prefix = IsInit ? ".init" : ".del"; 9388 9389 // Evaluate if this is an array section. 9390 llvm::BasicBlock *IsDeleteBB = 9391 MapperCGF.createBasicBlock(getName({"omp.array", Prefix, ".evaldelete"})); 9392 llvm::BasicBlock *BodyBB = 9393 MapperCGF.createBasicBlock(getName({"omp.array", Prefix})); 9394 llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGE( 9395 Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray"); 9396 MapperCGF.Builder.CreateCondBr(IsArray, IsDeleteBB, ExitBB); 9397 9398 // Evaluate if we are going to delete this section. 9399 MapperCGF.EmitBlock(IsDeleteBB); 9400 llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd( 9401 MapType, 9402 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE)); 9403 llvm::Value *DeleteCond; 9404 if (IsInit) { 9405 DeleteCond = MapperCGF.Builder.CreateIsNull( 9406 DeleteBit, getName({"omp.array", Prefix, ".delete"})); 9407 } else { 9408 DeleteCond = MapperCGF.Builder.CreateIsNotNull( 9409 DeleteBit, getName({"omp.array", Prefix, ".delete"})); 9410 } 9411 MapperCGF.Builder.CreateCondBr(DeleteCond, BodyBB, ExitBB); 9412 9413 MapperCGF.EmitBlock(BodyBB); 9414 // Get the array size by multiplying element size and element number (i.e., \p 9415 // Size). 9416 llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul( 9417 Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity())); 9418 // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves 9419 // memory allocation/deletion purpose only. 9420 llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd( 9421 MapType, 9422 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 9423 MappableExprsHandler::OMP_MAP_FROM))); 9424 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 9425 // data structure. 9426 llvm::Value *OffloadingArgs[] = {Handle, Base, Begin, ArraySize, MapTypeArg}; 9427 MapperCGF.EmitRuntimeCall( 9428 createRuntimeFunction(OMPRTL__tgt_push_mapper_component), OffloadingArgs); 9429 } 9430 9431 void CGOpenMPRuntime::emitTargetNumIterationsCall( 9432 CodeGenFunction &CGF, const OMPExecutableDirective &D, 9433 llvm::Value *DeviceID, 9434 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 9435 const OMPLoopDirective &D)> 9436 SizeEmitter) { 9437 OpenMPDirectiveKind Kind = D.getDirectiveKind(); 9438 const OMPExecutableDirective *TD = &D; 9439 // Get nested teams distribute kind directive, if any. 9440 if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind)) 9441 TD = getNestedDistributeDirective(CGM.getContext(), D); 9442 if (!TD) 9443 return; 9444 const auto *LD = cast<OMPLoopDirective>(TD); 9445 auto &&CodeGen = [LD, DeviceID, SizeEmitter, this](CodeGenFunction &CGF, 9446 PrePostActionTy &) { 9447 if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) { 9448 llvm::Value *Args[] = {DeviceID, NumIterations}; 9449 CGF.EmitRuntimeCall( 9450 createRuntimeFunction(OMPRTL__kmpc_push_target_tripcount), Args); 9451 } 9452 }; 9453 emitInlinedDirective(CGF, OMPD_unknown, CodeGen); 9454 } 9455 9456 void CGOpenMPRuntime::emitTargetCall( 9457 CodeGenFunction &CGF, const OMPExecutableDirective &D, 9458 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 9459 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device, 9460 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 9461 const OMPLoopDirective &D)> 9462 SizeEmitter) { 9463 if (!CGF.HaveInsertPoint()) 9464 return; 9465 9466 assert(OutlinedFn && "Invalid outlined function!"); 9467 9468 const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>(); 9469 llvm::SmallVector<llvm::Value *, 16> CapturedVars; 9470 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 9471 auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF, 9472 PrePostActionTy &) { 9473 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9474 }; 9475 emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen); 9476 9477 CodeGenFunction::OMPTargetDataInfo InputInfo; 9478 llvm::Value *MapTypesArray = nullptr; 9479 // Fill up the pointer arrays and transfer execution to the device. 9480 auto &&ThenGen = [this, Device, OutlinedFn, OutlinedFnID, &D, &InputInfo, 9481 &MapTypesArray, &CS, RequiresOuterTask, &CapturedVars, 9482 SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) { 9483 if (Device.getInt() == OMPC_DEVICE_ancestor) { 9484 // Reverse offloading is not supported, so just execute on the host. 9485 if (RequiresOuterTask) { 9486 CapturedVars.clear(); 9487 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9488 } 9489 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9490 return; 9491 } 9492 9493 // On top of the arrays that were filled up, the target offloading call 9494 // takes as arguments the device id as well as the host pointer. The host 9495 // pointer is used by the runtime library to identify the current target 9496 // region, so it only has to be unique and not necessarily point to 9497 // anything. It could be the pointer to the outlined function that 9498 // implements the target region, but we aren't using that so that the 9499 // compiler doesn't need to keep that, and could therefore inline the host 9500 // function if proven worthwhile during optimization. 9501 9502 // From this point on, we need to have an ID of the target region defined. 9503 assert(OutlinedFnID && "Invalid outlined function ID!"); 9504 9505 // Emit device ID if any. 9506 llvm::Value *DeviceID; 9507 if (Device.getPointer()) { 9508 assert((Device.getInt() == OMPC_DEVICE_unknown || 9509 Device.getInt() == OMPC_DEVICE_device_num) && 9510 "Expected device_num modifier."); 9511 llvm::Value *DevVal = CGF.EmitScalarExpr(Device.getPointer()); 9512 DeviceID = 9513 CGF.Builder.CreateIntCast(DevVal, CGF.Int64Ty, /*isSigned=*/true); 9514 } else { 9515 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 9516 } 9517 9518 // Emit the number of elements in the offloading arrays. 9519 llvm::Value *PointerNum = 9520 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 9521 9522 // Return value of the runtime offloading call. 9523 llvm::Value *Return; 9524 9525 llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D); 9526 llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D); 9527 9528 // Emit tripcount for the target loop-based directive. 9529 emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter); 9530 9531 bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 9532 // The target region is an outlined function launched by the runtime 9533 // via calls __tgt_target() or __tgt_target_teams(). 9534 // 9535 // __tgt_target() launches a target region with one team and one thread, 9536 // executing a serial region. This master thread may in turn launch 9537 // more threads within its team upon encountering a parallel region, 9538 // however, no additional teams can be launched on the device. 9539 // 9540 // __tgt_target_teams() launches a target region with one or more teams, 9541 // each with one or more threads. This call is required for target 9542 // constructs such as: 9543 // 'target teams' 9544 // 'target' / 'teams' 9545 // 'target teams distribute parallel for' 9546 // 'target parallel' 9547 // and so on. 9548 // 9549 // Note that on the host and CPU targets, the runtime implementation of 9550 // these calls simply call the outlined function without forking threads. 9551 // The outlined functions themselves have runtime calls to 9552 // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by 9553 // the compiler in emitTeamsCall() and emitParallelCall(). 9554 // 9555 // In contrast, on the NVPTX target, the implementation of 9556 // __tgt_target_teams() launches a GPU kernel with the requested number 9557 // of teams and threads so no additional calls to the runtime are required. 9558 if (NumTeams) { 9559 // If we have NumTeams defined this means that we have an enclosed teams 9560 // region. Therefore we also expect to have NumThreads defined. These two 9561 // values should be defined in the presence of a teams directive, 9562 // regardless of having any clauses associated. If the user is using teams 9563 // but no clauses, these two values will be the default that should be 9564 // passed to the runtime library - a 32-bit integer with the value zero. 9565 assert(NumThreads && "Thread limit expression should be available along " 9566 "with number of teams."); 9567 llvm::Value *OffloadingArgs[] = {DeviceID, 9568 OutlinedFnID, 9569 PointerNum, 9570 InputInfo.BasePointersArray.getPointer(), 9571 InputInfo.PointersArray.getPointer(), 9572 InputInfo.SizesArray.getPointer(), 9573 MapTypesArray, 9574 NumTeams, 9575 NumThreads}; 9576 Return = CGF.EmitRuntimeCall( 9577 createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_teams_nowait 9578 : OMPRTL__tgt_target_teams), 9579 OffloadingArgs); 9580 } else { 9581 llvm::Value *OffloadingArgs[] = {DeviceID, 9582 OutlinedFnID, 9583 PointerNum, 9584 InputInfo.BasePointersArray.getPointer(), 9585 InputInfo.PointersArray.getPointer(), 9586 InputInfo.SizesArray.getPointer(), 9587 MapTypesArray}; 9588 Return = CGF.EmitRuntimeCall( 9589 createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_nowait 9590 : OMPRTL__tgt_target), 9591 OffloadingArgs); 9592 } 9593 9594 // Check the error code and execute the host version if required. 9595 llvm::BasicBlock *OffloadFailedBlock = 9596 CGF.createBasicBlock("omp_offload.failed"); 9597 llvm::BasicBlock *OffloadContBlock = 9598 CGF.createBasicBlock("omp_offload.cont"); 9599 llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return); 9600 CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock); 9601 9602 CGF.EmitBlock(OffloadFailedBlock); 9603 if (RequiresOuterTask) { 9604 CapturedVars.clear(); 9605 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9606 } 9607 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9608 CGF.EmitBranch(OffloadContBlock); 9609 9610 CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true); 9611 }; 9612 9613 // Notify that the host version must be executed. 9614 auto &&ElseGen = [this, &D, OutlinedFn, &CS, &CapturedVars, 9615 RequiresOuterTask](CodeGenFunction &CGF, 9616 PrePostActionTy &) { 9617 if (RequiresOuterTask) { 9618 CapturedVars.clear(); 9619 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9620 } 9621 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9622 }; 9623 9624 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray, 9625 &CapturedVars, RequiresOuterTask, 9626 &CS](CodeGenFunction &CGF, PrePostActionTy &) { 9627 // Fill up the arrays with all the captured variables. 9628 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 9629 MappableExprsHandler::MapValuesArrayTy Pointers; 9630 MappableExprsHandler::MapValuesArrayTy Sizes; 9631 MappableExprsHandler::MapFlagsArrayTy MapTypes; 9632 9633 // Get mappable expression information. 9634 MappableExprsHandler MEHandler(D, CGF); 9635 llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers; 9636 9637 auto RI = CS.getCapturedRecordDecl()->field_begin(); 9638 auto CV = CapturedVars.begin(); 9639 for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(), 9640 CE = CS.capture_end(); 9641 CI != CE; ++CI, ++RI, ++CV) { 9642 MappableExprsHandler::MapBaseValuesArrayTy CurBasePointers; 9643 MappableExprsHandler::MapValuesArrayTy CurPointers; 9644 MappableExprsHandler::MapValuesArrayTy CurSizes; 9645 MappableExprsHandler::MapFlagsArrayTy CurMapTypes; 9646 MappableExprsHandler::StructRangeInfoTy PartialStruct; 9647 9648 // VLA sizes are passed to the outlined region by copy and do not have map 9649 // information associated. 9650 if (CI->capturesVariableArrayType()) { 9651 CurBasePointers.push_back(*CV); 9652 CurPointers.push_back(*CV); 9653 CurSizes.push_back(CGF.Builder.CreateIntCast( 9654 CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true)); 9655 // Copy to the device as an argument. No need to retrieve it. 9656 CurMapTypes.push_back(MappableExprsHandler::OMP_MAP_LITERAL | 9657 MappableExprsHandler::OMP_MAP_TARGET_PARAM | 9658 MappableExprsHandler::OMP_MAP_IMPLICIT); 9659 } else { 9660 // If we have any information in the map clause, we use it, otherwise we 9661 // just do a default mapping. 9662 MEHandler.generateInfoForCapture(CI, *CV, CurBasePointers, CurPointers, 9663 CurSizes, CurMapTypes, PartialStruct); 9664 if (CurBasePointers.empty()) 9665 MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurBasePointers, 9666 CurPointers, CurSizes, CurMapTypes); 9667 // Generate correct mapping for variables captured by reference in 9668 // lambdas. 9669 if (CI->capturesVariable()) 9670 MEHandler.generateInfoForLambdaCaptures( 9671 CI->getCapturedVar(), *CV, CurBasePointers, CurPointers, CurSizes, 9672 CurMapTypes, LambdaPointers); 9673 } 9674 // We expect to have at least an element of information for this capture. 9675 assert(!CurBasePointers.empty() && 9676 "Non-existing map pointer for capture!"); 9677 assert(CurBasePointers.size() == CurPointers.size() && 9678 CurBasePointers.size() == CurSizes.size() && 9679 CurBasePointers.size() == CurMapTypes.size() && 9680 "Inconsistent map information sizes!"); 9681 9682 // If there is an entry in PartialStruct it means we have a struct with 9683 // individual members mapped. Emit an extra combined entry. 9684 if (PartialStruct.Base.isValid()) 9685 MEHandler.emitCombinedEntry(BasePointers, Pointers, Sizes, MapTypes, 9686 CurMapTypes, PartialStruct); 9687 9688 // We need to append the results of this capture to what we already have. 9689 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 9690 Pointers.append(CurPointers.begin(), CurPointers.end()); 9691 Sizes.append(CurSizes.begin(), CurSizes.end()); 9692 MapTypes.append(CurMapTypes.begin(), CurMapTypes.end()); 9693 } 9694 // Adjust MEMBER_OF flags for the lambdas captures. 9695 MEHandler.adjustMemberOfForLambdaCaptures(LambdaPointers, BasePointers, 9696 Pointers, MapTypes); 9697 // Map other list items in the map clause which are not captured variables 9698 // but "declare target link" global variables. 9699 MEHandler.generateInfoForDeclareTargetLink(BasePointers, Pointers, Sizes, 9700 MapTypes); 9701 9702 TargetDataInfo Info; 9703 // Fill up the arrays and create the arguments. 9704 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 9705 emitOffloadingArraysArgument(CGF, Info.BasePointersArray, 9706 Info.PointersArray, Info.SizesArray, 9707 Info.MapTypesArray, Info); 9708 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 9709 InputInfo.BasePointersArray = 9710 Address(Info.BasePointersArray, CGM.getPointerAlign()); 9711 InputInfo.PointersArray = 9712 Address(Info.PointersArray, CGM.getPointerAlign()); 9713 InputInfo.SizesArray = Address(Info.SizesArray, CGM.getPointerAlign()); 9714 MapTypesArray = Info.MapTypesArray; 9715 if (RequiresOuterTask) 9716 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 9717 else 9718 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 9719 }; 9720 9721 auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask]( 9722 CodeGenFunction &CGF, PrePostActionTy &) { 9723 if (RequiresOuterTask) { 9724 CodeGenFunction::OMPTargetDataInfo InputInfo; 9725 CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo); 9726 } else { 9727 emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen); 9728 } 9729 }; 9730 9731 // If we have a target function ID it means that we need to support 9732 // offloading, otherwise, just execute on the host. We need to execute on host 9733 // regardless of the conditional in the if clause if, e.g., the user do not 9734 // specify target triples. 9735 if (OutlinedFnID) { 9736 if (IfCond) { 9737 emitIfClause(CGF, IfCond, TargetThenGen, TargetElseGen); 9738 } else { 9739 RegionCodeGenTy ThenRCG(TargetThenGen); 9740 ThenRCG(CGF); 9741 } 9742 } else { 9743 RegionCodeGenTy ElseRCG(TargetElseGen); 9744 ElseRCG(CGF); 9745 } 9746 } 9747 9748 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S, 9749 StringRef ParentName) { 9750 if (!S) 9751 return; 9752 9753 // Codegen OMP target directives that offload compute to the device. 9754 bool RequiresDeviceCodegen = 9755 isa<OMPExecutableDirective>(S) && 9756 isOpenMPTargetExecutionDirective( 9757 cast<OMPExecutableDirective>(S)->getDirectiveKind()); 9758 9759 if (RequiresDeviceCodegen) { 9760 const auto &E = *cast<OMPExecutableDirective>(S); 9761 unsigned DeviceID; 9762 unsigned FileID; 9763 unsigned Line; 9764 getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID, 9765 FileID, Line); 9766 9767 // Is this a target region that should not be emitted as an entry point? If 9768 // so just signal we are done with this target region. 9769 if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID, 9770 ParentName, Line)) 9771 return; 9772 9773 switch (E.getDirectiveKind()) { 9774 case OMPD_target: 9775 CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName, 9776 cast<OMPTargetDirective>(E)); 9777 break; 9778 case OMPD_target_parallel: 9779 CodeGenFunction::EmitOMPTargetParallelDeviceFunction( 9780 CGM, ParentName, cast<OMPTargetParallelDirective>(E)); 9781 break; 9782 case OMPD_target_teams: 9783 CodeGenFunction::EmitOMPTargetTeamsDeviceFunction( 9784 CGM, ParentName, cast<OMPTargetTeamsDirective>(E)); 9785 break; 9786 case OMPD_target_teams_distribute: 9787 CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction( 9788 CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E)); 9789 break; 9790 case OMPD_target_teams_distribute_simd: 9791 CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction( 9792 CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E)); 9793 break; 9794 case OMPD_target_parallel_for: 9795 CodeGenFunction::EmitOMPTargetParallelForDeviceFunction( 9796 CGM, ParentName, cast<OMPTargetParallelForDirective>(E)); 9797 break; 9798 case OMPD_target_parallel_for_simd: 9799 CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction( 9800 CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E)); 9801 break; 9802 case OMPD_target_simd: 9803 CodeGenFunction::EmitOMPTargetSimdDeviceFunction( 9804 CGM, ParentName, cast<OMPTargetSimdDirective>(E)); 9805 break; 9806 case OMPD_target_teams_distribute_parallel_for: 9807 CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction( 9808 CGM, ParentName, 9809 cast<OMPTargetTeamsDistributeParallelForDirective>(E)); 9810 break; 9811 case OMPD_target_teams_distribute_parallel_for_simd: 9812 CodeGenFunction:: 9813 EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction( 9814 CGM, ParentName, 9815 cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E)); 9816 break; 9817 case OMPD_parallel: 9818 case OMPD_for: 9819 case OMPD_parallel_for: 9820 case OMPD_parallel_master: 9821 case OMPD_parallel_sections: 9822 case OMPD_for_simd: 9823 case OMPD_parallel_for_simd: 9824 case OMPD_cancel: 9825 case OMPD_cancellation_point: 9826 case OMPD_ordered: 9827 case OMPD_threadprivate: 9828 case OMPD_allocate: 9829 case OMPD_task: 9830 case OMPD_simd: 9831 case OMPD_sections: 9832 case OMPD_section: 9833 case OMPD_single: 9834 case OMPD_master: 9835 case OMPD_critical: 9836 case OMPD_taskyield: 9837 case OMPD_barrier: 9838 case OMPD_taskwait: 9839 case OMPD_taskgroup: 9840 case OMPD_atomic: 9841 case OMPD_flush: 9842 case OMPD_depobj: 9843 case OMPD_scan: 9844 case OMPD_teams: 9845 case OMPD_target_data: 9846 case OMPD_target_exit_data: 9847 case OMPD_target_enter_data: 9848 case OMPD_distribute: 9849 case OMPD_distribute_simd: 9850 case OMPD_distribute_parallel_for: 9851 case OMPD_distribute_parallel_for_simd: 9852 case OMPD_teams_distribute: 9853 case OMPD_teams_distribute_simd: 9854 case OMPD_teams_distribute_parallel_for: 9855 case OMPD_teams_distribute_parallel_for_simd: 9856 case OMPD_target_update: 9857 case OMPD_declare_simd: 9858 case OMPD_declare_variant: 9859 case OMPD_begin_declare_variant: 9860 case OMPD_end_declare_variant: 9861 case OMPD_declare_target: 9862 case OMPD_end_declare_target: 9863 case OMPD_declare_reduction: 9864 case OMPD_declare_mapper: 9865 case OMPD_taskloop: 9866 case OMPD_taskloop_simd: 9867 case OMPD_master_taskloop: 9868 case OMPD_master_taskloop_simd: 9869 case OMPD_parallel_master_taskloop: 9870 case OMPD_parallel_master_taskloop_simd: 9871 case OMPD_requires: 9872 case OMPD_unknown: 9873 llvm_unreachable("Unknown target directive for OpenMP device codegen."); 9874 } 9875 return; 9876 } 9877 9878 if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) { 9879 if (!E->hasAssociatedStmt() || !E->getAssociatedStmt()) 9880 return; 9881 9882 scanForTargetRegionsFunctions( 9883 E->getInnermostCapturedStmt()->getCapturedStmt(), ParentName); 9884 return; 9885 } 9886 9887 // If this is a lambda function, look into its body. 9888 if (const auto *L = dyn_cast<LambdaExpr>(S)) 9889 S = L->getBody(); 9890 9891 // Keep looking for target regions recursively. 9892 for (const Stmt *II : S->children()) 9893 scanForTargetRegionsFunctions(II, ParentName); 9894 } 9895 9896 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) { 9897 // If emitting code for the host, we do not process FD here. Instead we do 9898 // the normal code generation. 9899 if (!CGM.getLangOpts().OpenMPIsDevice) { 9900 if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl())) { 9901 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 9902 OMPDeclareTargetDeclAttr::getDeviceType(FD); 9903 // Do not emit device_type(nohost) functions for the host. 9904 if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_NoHost) 9905 return true; 9906 } 9907 return false; 9908 } 9909 9910 const ValueDecl *VD = cast<ValueDecl>(GD.getDecl()); 9911 // Try to detect target regions in the function. 9912 if (const auto *FD = dyn_cast<FunctionDecl>(VD)) { 9913 StringRef Name = CGM.getMangledName(GD); 9914 scanForTargetRegionsFunctions(FD->getBody(), Name); 9915 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 9916 OMPDeclareTargetDeclAttr::getDeviceType(FD); 9917 // Do not emit device_type(nohost) functions for the host. 9918 if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_Host) 9919 return true; 9920 } 9921 9922 // Do not to emit function if it is not marked as declare target. 9923 return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) && 9924 AlreadyEmittedTargetDecls.count(VD) == 0; 9925 } 9926 9927 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 9928 if (!CGM.getLangOpts().OpenMPIsDevice) 9929 return false; 9930 9931 // Check if there are Ctors/Dtors in this declaration and look for target 9932 // regions in it. We use the complete variant to produce the kernel name 9933 // mangling. 9934 QualType RDTy = cast<VarDecl>(GD.getDecl())->getType(); 9935 if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) { 9936 for (const CXXConstructorDecl *Ctor : RD->ctors()) { 9937 StringRef ParentName = 9938 CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete)); 9939 scanForTargetRegionsFunctions(Ctor->getBody(), ParentName); 9940 } 9941 if (const CXXDestructorDecl *Dtor = RD->getDestructor()) { 9942 StringRef ParentName = 9943 CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete)); 9944 scanForTargetRegionsFunctions(Dtor->getBody(), ParentName); 9945 } 9946 } 9947 9948 // Do not to emit variable if it is not marked as declare target. 9949 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 9950 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration( 9951 cast<VarDecl>(GD.getDecl())); 9952 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 9953 (*Res == OMPDeclareTargetDeclAttr::MT_To && 9954 HasRequiresUnifiedSharedMemory)) { 9955 DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl())); 9956 return true; 9957 } 9958 return false; 9959 } 9960 9961 llvm::Constant * 9962 CGOpenMPRuntime::registerTargetFirstprivateCopy(CodeGenFunction &CGF, 9963 const VarDecl *VD) { 9964 assert(VD->getType().isConstant(CGM.getContext()) && 9965 "Expected constant variable."); 9966 StringRef VarName; 9967 llvm::Constant *Addr; 9968 llvm::GlobalValue::LinkageTypes Linkage; 9969 QualType Ty = VD->getType(); 9970 SmallString<128> Buffer; 9971 { 9972 unsigned DeviceID; 9973 unsigned FileID; 9974 unsigned Line; 9975 getTargetEntryUniqueInfo(CGM.getContext(), VD->getLocation(), DeviceID, 9976 FileID, Line); 9977 llvm::raw_svector_ostream OS(Buffer); 9978 OS << "__omp_offloading_firstprivate_" << llvm::format("_%x", DeviceID) 9979 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 9980 VarName = OS.str(); 9981 } 9982 Linkage = llvm::GlobalValue::InternalLinkage; 9983 Addr = 9984 getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(Ty), VarName, 9985 getDefaultFirstprivateAddressSpace()); 9986 cast<llvm::GlobalValue>(Addr)->setLinkage(Linkage); 9987 CharUnits VarSize = CGM.getContext().getTypeSizeInChars(Ty); 9988 CGM.addCompilerUsedGlobal(cast<llvm::GlobalValue>(Addr)); 9989 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 9990 VarName, Addr, VarSize, 9991 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo, Linkage); 9992 return Addr; 9993 } 9994 9995 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD, 9996 llvm::Constant *Addr) { 9997 if (CGM.getLangOpts().OMPTargetTriples.empty() && 9998 !CGM.getLangOpts().OpenMPIsDevice) 9999 return; 10000 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 10001 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 10002 if (!Res) { 10003 if (CGM.getLangOpts().OpenMPIsDevice) { 10004 // Register non-target variables being emitted in device code (debug info 10005 // may cause this). 10006 StringRef VarName = CGM.getMangledName(VD); 10007 EmittedNonTargetVariables.try_emplace(VarName, Addr); 10008 } 10009 return; 10010 } 10011 // Register declare target variables. 10012 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags; 10013 StringRef VarName; 10014 CharUnits VarSize; 10015 llvm::GlobalValue::LinkageTypes Linkage; 10016 10017 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 10018 !HasRequiresUnifiedSharedMemory) { 10019 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 10020 VarName = CGM.getMangledName(VD); 10021 if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) { 10022 VarSize = CGM.getContext().getTypeSizeInChars(VD->getType()); 10023 assert(!VarSize.isZero() && "Expected non-zero size of the variable"); 10024 } else { 10025 VarSize = CharUnits::Zero(); 10026 } 10027 Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false); 10028 // Temp solution to prevent optimizations of the internal variables. 10029 if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) { 10030 std::string RefName = getName({VarName, "ref"}); 10031 if (!CGM.GetGlobalValue(RefName)) { 10032 llvm::Constant *AddrRef = 10033 getOrCreateInternalVariable(Addr->getType(), RefName); 10034 auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef); 10035 GVAddrRef->setConstant(/*Val=*/true); 10036 GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage); 10037 GVAddrRef->setInitializer(Addr); 10038 CGM.addCompilerUsedGlobal(GVAddrRef); 10039 } 10040 } 10041 } else { 10042 assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 10043 (*Res == OMPDeclareTargetDeclAttr::MT_To && 10044 HasRequiresUnifiedSharedMemory)) && 10045 "Declare target attribute must link or to with unified memory."); 10046 if (*Res == OMPDeclareTargetDeclAttr::MT_Link) 10047 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink; 10048 else 10049 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 10050 10051 if (CGM.getLangOpts().OpenMPIsDevice) { 10052 VarName = Addr->getName(); 10053 Addr = nullptr; 10054 } else { 10055 VarName = getAddrOfDeclareTargetVar(VD).getName(); 10056 Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer()); 10057 } 10058 VarSize = CGM.getPointerSize(); 10059 Linkage = llvm::GlobalValue::WeakAnyLinkage; 10060 } 10061 10062 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 10063 VarName, Addr, VarSize, Flags, Linkage); 10064 } 10065 10066 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) { 10067 if (isa<FunctionDecl>(GD.getDecl()) || 10068 isa<OMPDeclareReductionDecl>(GD.getDecl())) 10069 return emitTargetFunctions(GD); 10070 10071 return emitTargetGlobalVariable(GD); 10072 } 10073 10074 void CGOpenMPRuntime::emitDeferredTargetDecls() const { 10075 for (const VarDecl *VD : DeferredGlobalVariables) { 10076 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 10077 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 10078 if (!Res) 10079 continue; 10080 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 10081 !HasRequiresUnifiedSharedMemory) { 10082 CGM.EmitGlobal(VD); 10083 } else { 10084 assert((*Res == OMPDeclareTargetDeclAttr::MT_Link || 10085 (*Res == OMPDeclareTargetDeclAttr::MT_To && 10086 HasRequiresUnifiedSharedMemory)) && 10087 "Expected link clause or to clause with unified memory."); 10088 (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 10089 } 10090 } 10091 } 10092 10093 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas( 10094 CodeGenFunction &CGF, const OMPExecutableDirective &D) const { 10095 assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) && 10096 " Expected target-based directive."); 10097 } 10098 10099 void CGOpenMPRuntime::processRequiresDirective(const OMPRequiresDecl *D) { 10100 for (const OMPClause *Clause : D->clauselists()) { 10101 if (Clause->getClauseKind() == OMPC_unified_shared_memory) { 10102 HasRequiresUnifiedSharedMemory = true; 10103 } else if (const auto *AC = 10104 dyn_cast<OMPAtomicDefaultMemOrderClause>(Clause)) { 10105 switch (AC->getAtomicDefaultMemOrderKind()) { 10106 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel: 10107 RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease; 10108 break; 10109 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst: 10110 RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent; 10111 break; 10112 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed: 10113 RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic; 10114 break; 10115 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown: 10116 break; 10117 } 10118 } 10119 } 10120 } 10121 10122 llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const { 10123 return RequiresAtomicOrdering; 10124 } 10125 10126 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD, 10127 LangAS &AS) { 10128 if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>()) 10129 return false; 10130 const auto *A = VD->getAttr<OMPAllocateDeclAttr>(); 10131 switch(A->getAllocatorType()) { 10132 case OMPAllocateDeclAttr::OMPDefaultMemAlloc: 10133 // Not supported, fallback to the default mem space. 10134 case OMPAllocateDeclAttr::OMPLargeCapMemAlloc: 10135 case OMPAllocateDeclAttr::OMPCGroupMemAlloc: 10136 case OMPAllocateDeclAttr::OMPHighBWMemAlloc: 10137 case OMPAllocateDeclAttr::OMPLowLatMemAlloc: 10138 case OMPAllocateDeclAttr::OMPThreadMemAlloc: 10139 case OMPAllocateDeclAttr::OMPConstMemAlloc: 10140 case OMPAllocateDeclAttr::OMPPTeamMemAlloc: 10141 AS = LangAS::Default; 10142 return true; 10143 case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc: 10144 llvm_unreachable("Expected predefined allocator for the variables with the " 10145 "static storage."); 10146 } 10147 return false; 10148 } 10149 10150 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const { 10151 return HasRequiresUnifiedSharedMemory; 10152 } 10153 10154 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII( 10155 CodeGenModule &CGM) 10156 : CGM(CGM) { 10157 if (CGM.getLangOpts().OpenMPIsDevice) { 10158 SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal; 10159 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false; 10160 } 10161 } 10162 10163 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() { 10164 if (CGM.getLangOpts().OpenMPIsDevice) 10165 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal; 10166 } 10167 10168 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) { 10169 if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal) 10170 return true; 10171 10172 const auto *D = cast<FunctionDecl>(GD.getDecl()); 10173 // Do not to emit function if it is marked as declare target as it was already 10174 // emitted. 10175 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) { 10176 if (D->hasBody() && AlreadyEmittedTargetDecls.count(D) == 0) { 10177 if (auto *F = dyn_cast_or_null<llvm::Function>( 10178 CGM.GetGlobalValue(CGM.getMangledName(GD)))) 10179 return !F->isDeclaration(); 10180 return false; 10181 } 10182 return true; 10183 } 10184 10185 return !AlreadyEmittedTargetDecls.insert(D).second; 10186 } 10187 10188 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() { 10189 // If we don't have entries or if we are emitting code for the device, we 10190 // don't need to do anything. 10191 if (CGM.getLangOpts().OMPTargetTriples.empty() || 10192 CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice || 10193 (OffloadEntriesInfoManager.empty() && 10194 !HasEmittedDeclareTargetRegion && 10195 !HasEmittedTargetRegion)) 10196 return nullptr; 10197 10198 // Create and register the function that handles the requires directives. 10199 ASTContext &C = CGM.getContext(); 10200 10201 llvm::Function *RequiresRegFn; 10202 { 10203 CodeGenFunction CGF(CGM); 10204 const auto &FI = CGM.getTypes().arrangeNullaryFunction(); 10205 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 10206 std::string ReqName = getName({"omp_offloading", "requires_reg"}); 10207 RequiresRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, ReqName, FI); 10208 CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {}); 10209 OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE; 10210 // TODO: check for other requires clauses. 10211 // The requires directive takes effect only when a target region is 10212 // present in the compilation unit. Otherwise it is ignored and not 10213 // passed to the runtime. This avoids the runtime from throwing an error 10214 // for mismatching requires clauses across compilation units that don't 10215 // contain at least 1 target region. 10216 assert((HasEmittedTargetRegion || 10217 HasEmittedDeclareTargetRegion || 10218 !OffloadEntriesInfoManager.empty()) && 10219 "Target or declare target region expected."); 10220 if (HasRequiresUnifiedSharedMemory) 10221 Flags = OMP_REQ_UNIFIED_SHARED_MEMORY; 10222 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_requires), 10223 llvm::ConstantInt::get(CGM.Int64Ty, Flags)); 10224 CGF.FinishFunction(); 10225 } 10226 return RequiresRegFn; 10227 } 10228 10229 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF, 10230 const OMPExecutableDirective &D, 10231 SourceLocation Loc, 10232 llvm::Function *OutlinedFn, 10233 ArrayRef<llvm::Value *> CapturedVars) { 10234 if (!CGF.HaveInsertPoint()) 10235 return; 10236 10237 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 10238 CodeGenFunction::RunCleanupsScope Scope(CGF); 10239 10240 // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn); 10241 llvm::Value *Args[] = { 10242 RTLoc, 10243 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 10244 CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())}; 10245 llvm::SmallVector<llvm::Value *, 16> RealArgs; 10246 RealArgs.append(std::begin(Args), std::end(Args)); 10247 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 10248 10249 llvm::FunctionCallee RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_teams); 10250 CGF.EmitRuntimeCall(RTLFn, RealArgs); 10251 } 10252 10253 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 10254 const Expr *NumTeams, 10255 const Expr *ThreadLimit, 10256 SourceLocation Loc) { 10257 if (!CGF.HaveInsertPoint()) 10258 return; 10259 10260 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 10261 10262 llvm::Value *NumTeamsVal = 10263 NumTeams 10264 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams), 10265 CGF.CGM.Int32Ty, /* isSigned = */ true) 10266 : CGF.Builder.getInt32(0); 10267 10268 llvm::Value *ThreadLimitVal = 10269 ThreadLimit 10270 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit), 10271 CGF.CGM.Int32Ty, /* isSigned = */ true) 10272 : CGF.Builder.getInt32(0); 10273 10274 // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit) 10275 llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal, 10276 ThreadLimitVal}; 10277 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_teams), 10278 PushNumTeamsArgs); 10279 } 10280 10281 void CGOpenMPRuntime::emitTargetDataCalls( 10282 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 10283 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 10284 if (!CGF.HaveInsertPoint()) 10285 return; 10286 10287 // Action used to replace the default codegen action and turn privatization 10288 // off. 10289 PrePostActionTy NoPrivAction; 10290 10291 // Generate the code for the opening of the data environment. Capture all the 10292 // arguments of the runtime call by reference because they are used in the 10293 // closing of the region. 10294 auto &&BeginThenGen = [this, &D, Device, &Info, 10295 &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) { 10296 // Fill up the arrays with all the mapped variables. 10297 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 10298 MappableExprsHandler::MapValuesArrayTy Pointers; 10299 MappableExprsHandler::MapValuesArrayTy Sizes; 10300 MappableExprsHandler::MapFlagsArrayTy MapTypes; 10301 10302 // Get map clause information. 10303 MappableExprsHandler MCHandler(D, CGF); 10304 MCHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes); 10305 10306 // Fill up the arrays and create the arguments. 10307 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 10308 10309 llvm::Value *BasePointersArrayArg = nullptr; 10310 llvm::Value *PointersArrayArg = nullptr; 10311 llvm::Value *SizesArrayArg = nullptr; 10312 llvm::Value *MapTypesArrayArg = nullptr; 10313 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 10314 SizesArrayArg, MapTypesArrayArg, Info); 10315 10316 // Emit device ID if any. 10317 llvm::Value *DeviceID = nullptr; 10318 if (Device) { 10319 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10320 CGF.Int64Ty, /*isSigned=*/true); 10321 } else { 10322 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10323 } 10324 10325 // Emit the number of elements in the offloading arrays. 10326 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 10327 10328 llvm::Value *OffloadingArgs[] = { 10329 DeviceID, PointerNum, BasePointersArrayArg, 10330 PointersArrayArg, SizesArrayArg, MapTypesArrayArg}; 10331 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_begin), 10332 OffloadingArgs); 10333 10334 // If device pointer privatization is required, emit the body of the region 10335 // here. It will have to be duplicated: with and without privatization. 10336 if (!Info.CaptureDeviceAddrMap.empty()) 10337 CodeGen(CGF); 10338 }; 10339 10340 // Generate code for the closing of the data region. 10341 auto &&EndThenGen = [this, Device, &Info](CodeGenFunction &CGF, 10342 PrePostActionTy &) { 10343 assert(Info.isValid() && "Invalid data environment closing arguments."); 10344 10345 llvm::Value *BasePointersArrayArg = nullptr; 10346 llvm::Value *PointersArrayArg = nullptr; 10347 llvm::Value *SizesArrayArg = nullptr; 10348 llvm::Value *MapTypesArrayArg = nullptr; 10349 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 10350 SizesArrayArg, MapTypesArrayArg, Info); 10351 10352 // Emit device ID if any. 10353 llvm::Value *DeviceID = nullptr; 10354 if (Device) { 10355 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10356 CGF.Int64Ty, /*isSigned=*/true); 10357 } else { 10358 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10359 } 10360 10361 // Emit the number of elements in the offloading arrays. 10362 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 10363 10364 llvm::Value *OffloadingArgs[] = { 10365 DeviceID, PointerNum, BasePointersArrayArg, 10366 PointersArrayArg, SizesArrayArg, MapTypesArrayArg}; 10367 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_end), 10368 OffloadingArgs); 10369 }; 10370 10371 // If we need device pointer privatization, we need to emit the body of the 10372 // region with no privatization in the 'else' branch of the conditional. 10373 // Otherwise, we don't have to do anything. 10374 auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF, 10375 PrePostActionTy &) { 10376 if (!Info.CaptureDeviceAddrMap.empty()) { 10377 CodeGen.setAction(NoPrivAction); 10378 CodeGen(CGF); 10379 } 10380 }; 10381 10382 // We don't have to do anything to close the region if the if clause evaluates 10383 // to false. 10384 auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {}; 10385 10386 if (IfCond) { 10387 emitIfClause(CGF, IfCond, BeginThenGen, BeginElseGen); 10388 } else { 10389 RegionCodeGenTy RCG(BeginThenGen); 10390 RCG(CGF); 10391 } 10392 10393 // If we don't require privatization of device pointers, we emit the body in 10394 // between the runtime calls. This avoids duplicating the body code. 10395 if (Info.CaptureDeviceAddrMap.empty()) { 10396 CodeGen.setAction(NoPrivAction); 10397 CodeGen(CGF); 10398 } 10399 10400 if (IfCond) { 10401 emitIfClause(CGF, IfCond, EndThenGen, EndElseGen); 10402 } else { 10403 RegionCodeGenTy RCG(EndThenGen); 10404 RCG(CGF); 10405 } 10406 } 10407 10408 void CGOpenMPRuntime::emitTargetDataStandAloneCall( 10409 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 10410 const Expr *Device) { 10411 if (!CGF.HaveInsertPoint()) 10412 return; 10413 10414 assert((isa<OMPTargetEnterDataDirective>(D) || 10415 isa<OMPTargetExitDataDirective>(D) || 10416 isa<OMPTargetUpdateDirective>(D)) && 10417 "Expecting either target enter, exit data, or update directives."); 10418 10419 CodeGenFunction::OMPTargetDataInfo InputInfo; 10420 llvm::Value *MapTypesArray = nullptr; 10421 // Generate the code for the opening of the data environment. 10422 auto &&ThenGen = [this, &D, Device, &InputInfo, 10423 &MapTypesArray](CodeGenFunction &CGF, PrePostActionTy &) { 10424 // Emit device ID if any. 10425 llvm::Value *DeviceID = nullptr; 10426 if (Device) { 10427 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10428 CGF.Int64Ty, /*isSigned=*/true); 10429 } else { 10430 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10431 } 10432 10433 // Emit the number of elements in the offloading arrays. 10434 llvm::Constant *PointerNum = 10435 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 10436 10437 llvm::Value *OffloadingArgs[] = {DeviceID, 10438 PointerNum, 10439 InputInfo.BasePointersArray.getPointer(), 10440 InputInfo.PointersArray.getPointer(), 10441 InputInfo.SizesArray.getPointer(), 10442 MapTypesArray}; 10443 10444 // Select the right runtime function call for each expected standalone 10445 // directive. 10446 const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 10447 OpenMPRTLFunction RTLFn; 10448 switch (D.getDirectiveKind()) { 10449 case OMPD_target_enter_data: 10450 RTLFn = HasNowait ? OMPRTL__tgt_target_data_begin_nowait 10451 : OMPRTL__tgt_target_data_begin; 10452 break; 10453 case OMPD_target_exit_data: 10454 RTLFn = HasNowait ? OMPRTL__tgt_target_data_end_nowait 10455 : OMPRTL__tgt_target_data_end; 10456 break; 10457 case OMPD_target_update: 10458 RTLFn = HasNowait ? OMPRTL__tgt_target_data_update_nowait 10459 : OMPRTL__tgt_target_data_update; 10460 break; 10461 case OMPD_parallel: 10462 case OMPD_for: 10463 case OMPD_parallel_for: 10464 case OMPD_parallel_master: 10465 case OMPD_parallel_sections: 10466 case OMPD_for_simd: 10467 case OMPD_parallel_for_simd: 10468 case OMPD_cancel: 10469 case OMPD_cancellation_point: 10470 case OMPD_ordered: 10471 case OMPD_threadprivate: 10472 case OMPD_allocate: 10473 case OMPD_task: 10474 case OMPD_simd: 10475 case OMPD_sections: 10476 case OMPD_section: 10477 case OMPD_single: 10478 case OMPD_master: 10479 case OMPD_critical: 10480 case OMPD_taskyield: 10481 case OMPD_barrier: 10482 case OMPD_taskwait: 10483 case OMPD_taskgroup: 10484 case OMPD_atomic: 10485 case OMPD_flush: 10486 case OMPD_depobj: 10487 case OMPD_scan: 10488 case OMPD_teams: 10489 case OMPD_target_data: 10490 case OMPD_distribute: 10491 case OMPD_distribute_simd: 10492 case OMPD_distribute_parallel_for: 10493 case OMPD_distribute_parallel_for_simd: 10494 case OMPD_teams_distribute: 10495 case OMPD_teams_distribute_simd: 10496 case OMPD_teams_distribute_parallel_for: 10497 case OMPD_teams_distribute_parallel_for_simd: 10498 case OMPD_declare_simd: 10499 case OMPD_declare_variant: 10500 case OMPD_begin_declare_variant: 10501 case OMPD_end_declare_variant: 10502 case OMPD_declare_target: 10503 case OMPD_end_declare_target: 10504 case OMPD_declare_reduction: 10505 case OMPD_declare_mapper: 10506 case OMPD_taskloop: 10507 case OMPD_taskloop_simd: 10508 case OMPD_master_taskloop: 10509 case OMPD_master_taskloop_simd: 10510 case OMPD_parallel_master_taskloop: 10511 case OMPD_parallel_master_taskloop_simd: 10512 case OMPD_target: 10513 case OMPD_target_simd: 10514 case OMPD_target_teams_distribute: 10515 case OMPD_target_teams_distribute_simd: 10516 case OMPD_target_teams_distribute_parallel_for: 10517 case OMPD_target_teams_distribute_parallel_for_simd: 10518 case OMPD_target_teams: 10519 case OMPD_target_parallel: 10520 case OMPD_target_parallel_for: 10521 case OMPD_target_parallel_for_simd: 10522 case OMPD_requires: 10523 case OMPD_unknown: 10524 llvm_unreachable("Unexpected standalone target data directive."); 10525 break; 10526 } 10527 CGF.EmitRuntimeCall(createRuntimeFunction(RTLFn), OffloadingArgs); 10528 }; 10529 10530 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray]( 10531 CodeGenFunction &CGF, PrePostActionTy &) { 10532 // Fill up the arrays with all the mapped variables. 10533 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 10534 MappableExprsHandler::MapValuesArrayTy Pointers; 10535 MappableExprsHandler::MapValuesArrayTy Sizes; 10536 MappableExprsHandler::MapFlagsArrayTy MapTypes; 10537 10538 // Get map clause information. 10539 MappableExprsHandler MEHandler(D, CGF); 10540 MEHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes); 10541 10542 TargetDataInfo Info; 10543 // Fill up the arrays and create the arguments. 10544 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 10545 emitOffloadingArraysArgument(CGF, Info.BasePointersArray, 10546 Info.PointersArray, Info.SizesArray, 10547 Info.MapTypesArray, Info); 10548 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 10549 InputInfo.BasePointersArray = 10550 Address(Info.BasePointersArray, CGM.getPointerAlign()); 10551 InputInfo.PointersArray = 10552 Address(Info.PointersArray, CGM.getPointerAlign()); 10553 InputInfo.SizesArray = 10554 Address(Info.SizesArray, CGM.getPointerAlign()); 10555 MapTypesArray = Info.MapTypesArray; 10556 if (D.hasClausesOfKind<OMPDependClause>()) 10557 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 10558 else 10559 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 10560 }; 10561 10562 if (IfCond) { 10563 emitIfClause(CGF, IfCond, TargetThenGen, 10564 [](CodeGenFunction &CGF, PrePostActionTy &) {}); 10565 } else { 10566 RegionCodeGenTy ThenRCG(TargetThenGen); 10567 ThenRCG(CGF); 10568 } 10569 } 10570 10571 namespace { 10572 /// Kind of parameter in a function with 'declare simd' directive. 10573 enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector }; 10574 /// Attribute set of the parameter. 10575 struct ParamAttrTy { 10576 ParamKindTy Kind = Vector; 10577 llvm::APSInt StrideOrArg; 10578 llvm::APSInt Alignment; 10579 }; 10580 } // namespace 10581 10582 static unsigned evaluateCDTSize(const FunctionDecl *FD, 10583 ArrayRef<ParamAttrTy> ParamAttrs) { 10584 // Every vector variant of a SIMD-enabled function has a vector length (VLEN). 10585 // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument 10586 // of that clause. The VLEN value must be power of 2. 10587 // In other case the notion of the function`s "characteristic data type" (CDT) 10588 // is used to compute the vector length. 10589 // CDT is defined in the following order: 10590 // a) For non-void function, the CDT is the return type. 10591 // b) If the function has any non-uniform, non-linear parameters, then the 10592 // CDT is the type of the first such parameter. 10593 // c) If the CDT determined by a) or b) above is struct, union, or class 10594 // type which is pass-by-value (except for the type that maps to the 10595 // built-in complex data type), the characteristic data type is int. 10596 // d) If none of the above three cases is applicable, the CDT is int. 10597 // The VLEN is then determined based on the CDT and the size of vector 10598 // register of that ISA for which current vector version is generated. The 10599 // VLEN is computed using the formula below: 10600 // VLEN = sizeof(vector_register) / sizeof(CDT), 10601 // where vector register size specified in section 3.2.1 Registers and the 10602 // Stack Frame of original AMD64 ABI document. 10603 QualType RetType = FD->getReturnType(); 10604 if (RetType.isNull()) 10605 return 0; 10606 ASTContext &C = FD->getASTContext(); 10607 QualType CDT; 10608 if (!RetType.isNull() && !RetType->isVoidType()) { 10609 CDT = RetType; 10610 } else { 10611 unsigned Offset = 0; 10612 if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) { 10613 if (ParamAttrs[Offset].Kind == Vector) 10614 CDT = C.getPointerType(C.getRecordType(MD->getParent())); 10615 ++Offset; 10616 } 10617 if (CDT.isNull()) { 10618 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 10619 if (ParamAttrs[I + Offset].Kind == Vector) { 10620 CDT = FD->getParamDecl(I)->getType(); 10621 break; 10622 } 10623 } 10624 } 10625 } 10626 if (CDT.isNull()) 10627 CDT = C.IntTy; 10628 CDT = CDT->getCanonicalTypeUnqualified(); 10629 if (CDT->isRecordType() || CDT->isUnionType()) 10630 CDT = C.IntTy; 10631 return C.getTypeSize(CDT); 10632 } 10633 10634 static void 10635 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn, 10636 const llvm::APSInt &VLENVal, 10637 ArrayRef<ParamAttrTy> ParamAttrs, 10638 OMPDeclareSimdDeclAttr::BranchStateTy State) { 10639 struct ISADataTy { 10640 char ISA; 10641 unsigned VecRegSize; 10642 }; 10643 ISADataTy ISAData[] = { 10644 { 10645 'b', 128 10646 }, // SSE 10647 { 10648 'c', 256 10649 }, // AVX 10650 { 10651 'd', 256 10652 }, // AVX2 10653 { 10654 'e', 512 10655 }, // AVX512 10656 }; 10657 llvm::SmallVector<char, 2> Masked; 10658 switch (State) { 10659 case OMPDeclareSimdDeclAttr::BS_Undefined: 10660 Masked.push_back('N'); 10661 Masked.push_back('M'); 10662 break; 10663 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10664 Masked.push_back('N'); 10665 break; 10666 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10667 Masked.push_back('M'); 10668 break; 10669 } 10670 for (char Mask : Masked) { 10671 for (const ISADataTy &Data : ISAData) { 10672 SmallString<256> Buffer; 10673 llvm::raw_svector_ostream Out(Buffer); 10674 Out << "_ZGV" << Data.ISA << Mask; 10675 if (!VLENVal) { 10676 unsigned NumElts = evaluateCDTSize(FD, ParamAttrs); 10677 assert(NumElts && "Non-zero simdlen/cdtsize expected"); 10678 Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts); 10679 } else { 10680 Out << VLENVal; 10681 } 10682 for (const ParamAttrTy &ParamAttr : ParamAttrs) { 10683 switch (ParamAttr.Kind){ 10684 case LinearWithVarStride: 10685 Out << 's' << ParamAttr.StrideOrArg; 10686 break; 10687 case Linear: 10688 Out << 'l'; 10689 if (!!ParamAttr.StrideOrArg) 10690 Out << ParamAttr.StrideOrArg; 10691 break; 10692 case Uniform: 10693 Out << 'u'; 10694 break; 10695 case Vector: 10696 Out << 'v'; 10697 break; 10698 } 10699 if (!!ParamAttr.Alignment) 10700 Out << 'a' << ParamAttr.Alignment; 10701 } 10702 Out << '_' << Fn->getName(); 10703 Fn->addFnAttr(Out.str()); 10704 } 10705 } 10706 } 10707 10708 // This are the Functions that are needed to mangle the name of the 10709 // vector functions generated by the compiler, according to the rules 10710 // defined in the "Vector Function ABI specifications for AArch64", 10711 // available at 10712 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi. 10713 10714 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI. 10715 /// 10716 /// TODO: Need to implement the behavior for reference marked with a 10717 /// var or no linear modifiers (1.b in the section). For this, we 10718 /// need to extend ParamKindTy to support the linear modifiers. 10719 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) { 10720 QT = QT.getCanonicalType(); 10721 10722 if (QT->isVoidType()) 10723 return false; 10724 10725 if (Kind == ParamKindTy::Uniform) 10726 return false; 10727 10728 if (Kind == ParamKindTy::Linear) 10729 return false; 10730 10731 // TODO: Handle linear references with modifiers 10732 10733 if (Kind == ParamKindTy::LinearWithVarStride) 10734 return false; 10735 10736 return true; 10737 } 10738 10739 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI. 10740 static bool getAArch64PBV(QualType QT, ASTContext &C) { 10741 QT = QT.getCanonicalType(); 10742 unsigned Size = C.getTypeSize(QT); 10743 10744 // Only scalars and complex within 16 bytes wide set PVB to true. 10745 if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128) 10746 return false; 10747 10748 if (QT->isFloatingType()) 10749 return true; 10750 10751 if (QT->isIntegerType()) 10752 return true; 10753 10754 if (QT->isPointerType()) 10755 return true; 10756 10757 // TODO: Add support for complex types (section 3.1.2, item 2). 10758 10759 return false; 10760 } 10761 10762 /// Computes the lane size (LS) of a return type or of an input parameter, 10763 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI. 10764 /// TODO: Add support for references, section 3.2.1, item 1. 10765 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) { 10766 if (getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) { 10767 QualType PTy = QT.getCanonicalType()->getPointeeType(); 10768 if (getAArch64PBV(PTy, C)) 10769 return C.getTypeSize(PTy); 10770 } 10771 if (getAArch64PBV(QT, C)) 10772 return C.getTypeSize(QT); 10773 10774 return C.getTypeSize(C.getUIntPtrType()); 10775 } 10776 10777 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the 10778 // signature of the scalar function, as defined in 3.2.2 of the 10779 // AAVFABI. 10780 static std::tuple<unsigned, unsigned, bool> 10781 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) { 10782 QualType RetType = FD->getReturnType().getCanonicalType(); 10783 10784 ASTContext &C = FD->getASTContext(); 10785 10786 bool OutputBecomesInput = false; 10787 10788 llvm::SmallVector<unsigned, 8> Sizes; 10789 if (!RetType->isVoidType()) { 10790 Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C)); 10791 if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {})) 10792 OutputBecomesInput = true; 10793 } 10794 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 10795 QualType QT = FD->getParamDecl(I)->getType().getCanonicalType(); 10796 Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C)); 10797 } 10798 10799 assert(!Sizes.empty() && "Unable to determine NDS and WDS."); 10800 // The LS of a function parameter / return value can only be a power 10801 // of 2, starting from 8 bits, up to 128. 10802 assert(std::all_of(Sizes.begin(), Sizes.end(), 10803 [](unsigned Size) { 10804 return Size == 8 || Size == 16 || Size == 32 || 10805 Size == 64 || Size == 128; 10806 }) && 10807 "Invalid size"); 10808 10809 return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)), 10810 *std::max_element(std::begin(Sizes), std::end(Sizes)), 10811 OutputBecomesInput); 10812 } 10813 10814 /// Mangle the parameter part of the vector function name according to 10815 /// their OpenMP classification. The mangling function is defined in 10816 /// section 3.5 of the AAVFABI. 10817 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) { 10818 SmallString<256> Buffer; 10819 llvm::raw_svector_ostream Out(Buffer); 10820 for (const auto &ParamAttr : ParamAttrs) { 10821 switch (ParamAttr.Kind) { 10822 case LinearWithVarStride: 10823 Out << "ls" << ParamAttr.StrideOrArg; 10824 break; 10825 case Linear: 10826 Out << 'l'; 10827 // Don't print the step value if it is not present or if it is 10828 // equal to 1. 10829 if (!!ParamAttr.StrideOrArg && ParamAttr.StrideOrArg != 1) 10830 Out << ParamAttr.StrideOrArg; 10831 break; 10832 case Uniform: 10833 Out << 'u'; 10834 break; 10835 case Vector: 10836 Out << 'v'; 10837 break; 10838 } 10839 10840 if (!!ParamAttr.Alignment) 10841 Out << 'a' << ParamAttr.Alignment; 10842 } 10843 10844 return std::string(Out.str()); 10845 } 10846 10847 // Function used to add the attribute. The parameter `VLEN` is 10848 // templated to allow the use of "x" when targeting scalable functions 10849 // for SVE. 10850 template <typename T> 10851 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix, 10852 char ISA, StringRef ParSeq, 10853 StringRef MangledName, bool OutputBecomesInput, 10854 llvm::Function *Fn) { 10855 SmallString<256> Buffer; 10856 llvm::raw_svector_ostream Out(Buffer); 10857 Out << Prefix << ISA << LMask << VLEN; 10858 if (OutputBecomesInput) 10859 Out << "v"; 10860 Out << ParSeq << "_" << MangledName; 10861 Fn->addFnAttr(Out.str()); 10862 } 10863 10864 // Helper function to generate the Advanced SIMD names depending on 10865 // the value of the NDS when simdlen is not present. 10866 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask, 10867 StringRef Prefix, char ISA, 10868 StringRef ParSeq, StringRef MangledName, 10869 bool OutputBecomesInput, 10870 llvm::Function *Fn) { 10871 switch (NDS) { 10872 case 8: 10873 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 10874 OutputBecomesInput, Fn); 10875 addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName, 10876 OutputBecomesInput, Fn); 10877 break; 10878 case 16: 10879 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 10880 OutputBecomesInput, Fn); 10881 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 10882 OutputBecomesInput, Fn); 10883 break; 10884 case 32: 10885 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 10886 OutputBecomesInput, Fn); 10887 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 10888 OutputBecomesInput, Fn); 10889 break; 10890 case 64: 10891 case 128: 10892 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 10893 OutputBecomesInput, Fn); 10894 break; 10895 default: 10896 llvm_unreachable("Scalar type is too wide."); 10897 } 10898 } 10899 10900 /// Emit vector function attributes for AArch64, as defined in the AAVFABI. 10901 static void emitAArch64DeclareSimdFunction( 10902 CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN, 10903 ArrayRef<ParamAttrTy> ParamAttrs, 10904 OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName, 10905 char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) { 10906 10907 // Get basic data for building the vector signature. 10908 const auto Data = getNDSWDS(FD, ParamAttrs); 10909 const unsigned NDS = std::get<0>(Data); 10910 const unsigned WDS = std::get<1>(Data); 10911 const bool OutputBecomesInput = std::get<2>(Data); 10912 10913 // Check the values provided via `simdlen` by the user. 10914 // 1. A `simdlen(1)` doesn't produce vector signatures, 10915 if (UserVLEN == 1) { 10916 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10917 DiagnosticsEngine::Warning, 10918 "The clause simdlen(1) has no effect when targeting aarch64."); 10919 CGM.getDiags().Report(SLoc, DiagID); 10920 return; 10921 } 10922 10923 // 2. Section 3.3.1, item 1: user input must be a power of 2 for 10924 // Advanced SIMD output. 10925 if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) { 10926 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10927 DiagnosticsEngine::Warning, "The value specified in simdlen must be a " 10928 "power of 2 when targeting Advanced SIMD."); 10929 CGM.getDiags().Report(SLoc, DiagID); 10930 return; 10931 } 10932 10933 // 3. Section 3.4.1. SVE fixed lengh must obey the architectural 10934 // limits. 10935 if (ISA == 's' && UserVLEN != 0) { 10936 if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) { 10937 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10938 DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit " 10939 "lanes in the architectural constraints " 10940 "for SVE (min is 128-bit, max is " 10941 "2048-bit, by steps of 128-bit)"); 10942 CGM.getDiags().Report(SLoc, DiagID) << WDS; 10943 return; 10944 } 10945 } 10946 10947 // Sort out parameter sequence. 10948 const std::string ParSeq = mangleVectorParameters(ParamAttrs); 10949 StringRef Prefix = "_ZGV"; 10950 // Generate simdlen from user input (if any). 10951 if (UserVLEN) { 10952 if (ISA == 's') { 10953 // SVE generates only a masked function. 10954 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10955 OutputBecomesInput, Fn); 10956 } else { 10957 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 10958 // Advanced SIMD generates one or two functions, depending on 10959 // the `[not]inbranch` clause. 10960 switch (State) { 10961 case OMPDeclareSimdDeclAttr::BS_Undefined: 10962 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 10963 OutputBecomesInput, Fn); 10964 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10965 OutputBecomesInput, Fn); 10966 break; 10967 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10968 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 10969 OutputBecomesInput, Fn); 10970 break; 10971 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10972 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10973 OutputBecomesInput, Fn); 10974 break; 10975 } 10976 } 10977 } else { 10978 // If no user simdlen is provided, follow the AAVFABI rules for 10979 // generating the vector length. 10980 if (ISA == 's') { 10981 // SVE, section 3.4.1, item 1. 10982 addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName, 10983 OutputBecomesInput, Fn); 10984 } else { 10985 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 10986 // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or 10987 // two vector names depending on the use of the clause 10988 // `[not]inbranch`. 10989 switch (State) { 10990 case OMPDeclareSimdDeclAttr::BS_Undefined: 10991 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 10992 OutputBecomesInput, Fn); 10993 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 10994 OutputBecomesInput, Fn); 10995 break; 10996 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10997 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 10998 OutputBecomesInput, Fn); 10999 break; 11000 case OMPDeclareSimdDeclAttr::BS_Inbranch: 11001 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 11002 OutputBecomesInput, Fn); 11003 break; 11004 } 11005 } 11006 } 11007 } 11008 11009 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD, 11010 llvm::Function *Fn) { 11011 ASTContext &C = CGM.getContext(); 11012 FD = FD->getMostRecentDecl(); 11013 // Map params to their positions in function decl. 11014 llvm::DenseMap<const Decl *, unsigned> ParamPositions; 11015 if (isa<CXXMethodDecl>(FD)) 11016 ParamPositions.try_emplace(FD, 0); 11017 unsigned ParamPos = ParamPositions.size(); 11018 for (const ParmVarDecl *P : FD->parameters()) { 11019 ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos); 11020 ++ParamPos; 11021 } 11022 while (FD) { 11023 for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) { 11024 llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size()); 11025 // Mark uniform parameters. 11026 for (const Expr *E : Attr->uniforms()) { 11027 E = E->IgnoreParenImpCasts(); 11028 unsigned Pos; 11029 if (isa<CXXThisExpr>(E)) { 11030 Pos = ParamPositions[FD]; 11031 } else { 11032 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 11033 ->getCanonicalDecl(); 11034 Pos = ParamPositions[PVD]; 11035 } 11036 ParamAttrs[Pos].Kind = Uniform; 11037 } 11038 // Get alignment info. 11039 auto NI = Attr->alignments_begin(); 11040 for (const Expr *E : Attr->aligneds()) { 11041 E = E->IgnoreParenImpCasts(); 11042 unsigned Pos; 11043 QualType ParmTy; 11044 if (isa<CXXThisExpr>(E)) { 11045 Pos = ParamPositions[FD]; 11046 ParmTy = E->getType(); 11047 } else { 11048 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 11049 ->getCanonicalDecl(); 11050 Pos = ParamPositions[PVD]; 11051 ParmTy = PVD->getType(); 11052 } 11053 ParamAttrs[Pos].Alignment = 11054 (*NI) 11055 ? (*NI)->EvaluateKnownConstInt(C) 11056 : llvm::APSInt::getUnsigned( 11057 C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy)) 11058 .getQuantity()); 11059 ++NI; 11060 } 11061 // Mark linear parameters. 11062 auto SI = Attr->steps_begin(); 11063 auto MI = Attr->modifiers_begin(); 11064 for (const Expr *E : Attr->linears()) { 11065 E = E->IgnoreParenImpCasts(); 11066 unsigned Pos; 11067 if (isa<CXXThisExpr>(E)) { 11068 Pos = ParamPositions[FD]; 11069 } else { 11070 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 11071 ->getCanonicalDecl(); 11072 Pos = ParamPositions[PVD]; 11073 } 11074 ParamAttrTy &ParamAttr = ParamAttrs[Pos]; 11075 ParamAttr.Kind = Linear; 11076 if (*SI) { 11077 Expr::EvalResult Result; 11078 if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) { 11079 if (const auto *DRE = 11080 cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) { 11081 if (const auto *StridePVD = cast<ParmVarDecl>(DRE->getDecl())) { 11082 ParamAttr.Kind = LinearWithVarStride; 11083 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned( 11084 ParamPositions[StridePVD->getCanonicalDecl()]); 11085 } 11086 } 11087 } else { 11088 ParamAttr.StrideOrArg = Result.Val.getInt(); 11089 } 11090 } 11091 ++SI; 11092 ++MI; 11093 } 11094 llvm::APSInt VLENVal; 11095 SourceLocation ExprLoc; 11096 const Expr *VLENExpr = Attr->getSimdlen(); 11097 if (VLENExpr) { 11098 VLENVal = VLENExpr->EvaluateKnownConstInt(C); 11099 ExprLoc = VLENExpr->getExprLoc(); 11100 } 11101 OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState(); 11102 if (CGM.getTriple().isX86()) { 11103 emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State); 11104 } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) { 11105 unsigned VLEN = VLENVal.getExtValue(); 11106 StringRef MangledName = Fn->getName(); 11107 if (CGM.getTarget().hasFeature("sve")) 11108 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 11109 MangledName, 's', 128, Fn, ExprLoc); 11110 if (CGM.getTarget().hasFeature("neon")) 11111 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 11112 MangledName, 'n', 128, Fn, ExprLoc); 11113 } 11114 } 11115 FD = FD->getPreviousDecl(); 11116 } 11117 } 11118 11119 namespace { 11120 /// Cleanup action for doacross support. 11121 class DoacrossCleanupTy final : public EHScopeStack::Cleanup { 11122 public: 11123 static const int DoacrossFinArgs = 2; 11124 11125 private: 11126 llvm::FunctionCallee RTLFn; 11127 llvm::Value *Args[DoacrossFinArgs]; 11128 11129 public: 11130 DoacrossCleanupTy(llvm::FunctionCallee RTLFn, 11131 ArrayRef<llvm::Value *> CallArgs) 11132 : RTLFn(RTLFn) { 11133 assert(CallArgs.size() == DoacrossFinArgs); 11134 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 11135 } 11136 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 11137 if (!CGF.HaveInsertPoint()) 11138 return; 11139 CGF.EmitRuntimeCall(RTLFn, Args); 11140 } 11141 }; 11142 } // namespace 11143 11144 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF, 11145 const OMPLoopDirective &D, 11146 ArrayRef<Expr *> NumIterations) { 11147 if (!CGF.HaveInsertPoint()) 11148 return; 11149 11150 ASTContext &C = CGM.getContext(); 11151 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 11152 RecordDecl *RD; 11153 if (KmpDimTy.isNull()) { 11154 // Build struct kmp_dim { // loop bounds info casted to kmp_int64 11155 // kmp_int64 lo; // lower 11156 // kmp_int64 up; // upper 11157 // kmp_int64 st; // stride 11158 // }; 11159 RD = C.buildImplicitRecord("kmp_dim"); 11160 RD->startDefinition(); 11161 addFieldToRecordDecl(C, RD, Int64Ty); 11162 addFieldToRecordDecl(C, RD, Int64Ty); 11163 addFieldToRecordDecl(C, RD, Int64Ty); 11164 RD->completeDefinition(); 11165 KmpDimTy = C.getRecordType(RD); 11166 } else { 11167 RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl()); 11168 } 11169 llvm::APInt Size(/*numBits=*/32, NumIterations.size()); 11170 QualType ArrayTy = 11171 C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0); 11172 11173 Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims"); 11174 CGF.EmitNullInitialization(DimsAddr, ArrayTy); 11175 enum { LowerFD = 0, UpperFD, StrideFD }; 11176 // Fill dims with data. 11177 for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) { 11178 LValue DimsLVal = CGF.MakeAddrLValue( 11179 CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy); 11180 // dims.upper = num_iterations; 11181 LValue UpperLVal = CGF.EmitLValueForField( 11182 DimsLVal, *std::next(RD->field_begin(), UpperFD)); 11183 llvm::Value *NumIterVal = 11184 CGF.EmitScalarConversion(CGF.EmitScalarExpr(NumIterations[I]), 11185 D.getNumIterations()->getType(), Int64Ty, 11186 D.getNumIterations()->getExprLoc()); 11187 CGF.EmitStoreOfScalar(NumIterVal, UpperLVal); 11188 // dims.stride = 1; 11189 LValue StrideLVal = CGF.EmitLValueForField( 11190 DimsLVal, *std::next(RD->field_begin(), StrideFD)); 11191 CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1), 11192 StrideLVal); 11193 } 11194 11195 // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, 11196 // kmp_int32 num_dims, struct kmp_dim * dims); 11197 llvm::Value *Args[] = { 11198 emitUpdateLocation(CGF, D.getBeginLoc()), 11199 getThreadID(CGF, D.getBeginLoc()), 11200 llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()), 11201 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11202 CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(), 11203 CGM.VoidPtrTy)}; 11204 11205 llvm::FunctionCallee RTLFn = 11206 createRuntimeFunction(OMPRTL__kmpc_doacross_init); 11207 CGF.EmitRuntimeCall(RTLFn, Args); 11208 llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = { 11209 emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())}; 11210 llvm::FunctionCallee FiniRTLFn = 11211 createRuntimeFunction(OMPRTL__kmpc_doacross_fini); 11212 CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 11213 llvm::makeArrayRef(FiniArgs)); 11214 } 11215 11216 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 11217 const OMPDependClause *C) { 11218 QualType Int64Ty = 11219 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 11220 llvm::APInt Size(/*numBits=*/32, C->getNumLoops()); 11221 QualType ArrayTy = CGM.getContext().getConstantArrayType( 11222 Int64Ty, Size, nullptr, ArrayType::Normal, 0); 11223 Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr"); 11224 for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) { 11225 const Expr *CounterVal = C->getLoopData(I); 11226 assert(CounterVal); 11227 llvm::Value *CntVal = CGF.EmitScalarConversion( 11228 CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty, 11229 CounterVal->getExprLoc()); 11230 CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I), 11231 /*Volatile=*/false, Int64Ty); 11232 } 11233 llvm::Value *Args[] = { 11234 emitUpdateLocation(CGF, C->getBeginLoc()), 11235 getThreadID(CGF, C->getBeginLoc()), 11236 CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()}; 11237 llvm::FunctionCallee RTLFn; 11238 if (C->getDependencyKind() == OMPC_DEPEND_source) { 11239 RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_post); 11240 } else { 11241 assert(C->getDependencyKind() == OMPC_DEPEND_sink); 11242 RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_wait); 11243 } 11244 CGF.EmitRuntimeCall(RTLFn, Args); 11245 } 11246 11247 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc, 11248 llvm::FunctionCallee Callee, 11249 ArrayRef<llvm::Value *> Args) const { 11250 assert(Loc.isValid() && "Outlined function call location must be valid."); 11251 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 11252 11253 if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) { 11254 if (Fn->doesNotThrow()) { 11255 CGF.EmitNounwindRuntimeCall(Fn, Args); 11256 return; 11257 } 11258 } 11259 CGF.EmitRuntimeCall(Callee, Args); 11260 } 11261 11262 void CGOpenMPRuntime::emitOutlinedFunctionCall( 11263 CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn, 11264 ArrayRef<llvm::Value *> Args) const { 11265 emitCall(CGF, Loc, OutlinedFn, Args); 11266 } 11267 11268 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) { 11269 if (const auto *FD = dyn_cast<FunctionDecl>(D)) 11270 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD)) 11271 HasEmittedDeclareTargetRegion = true; 11272 } 11273 11274 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF, 11275 const VarDecl *NativeParam, 11276 const VarDecl *TargetParam) const { 11277 return CGF.GetAddrOfLocalVar(NativeParam); 11278 } 11279 11280 namespace { 11281 /// Cleanup action for allocate support. 11282 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup { 11283 public: 11284 static const int CleanupArgs = 3; 11285 11286 private: 11287 llvm::FunctionCallee RTLFn; 11288 llvm::Value *Args[CleanupArgs]; 11289 11290 public: 11291 OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn, 11292 ArrayRef<llvm::Value *> CallArgs) 11293 : RTLFn(RTLFn) { 11294 assert(CallArgs.size() == CleanupArgs && 11295 "Size of arguments does not match."); 11296 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 11297 } 11298 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 11299 if (!CGF.HaveInsertPoint()) 11300 return; 11301 CGF.EmitRuntimeCall(RTLFn, Args); 11302 } 11303 }; 11304 } // namespace 11305 11306 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF, 11307 const VarDecl *VD) { 11308 if (!VD) 11309 return Address::invalid(); 11310 const VarDecl *CVD = VD->getCanonicalDecl(); 11311 if (!CVD->hasAttr<OMPAllocateDeclAttr>()) 11312 return Address::invalid(); 11313 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>(); 11314 // Use the default allocation. 11315 if (AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc && 11316 !AA->getAllocator()) 11317 return Address::invalid(); 11318 llvm::Value *Size; 11319 CharUnits Align = CGM.getContext().getDeclAlign(CVD); 11320 if (CVD->getType()->isVariablyModifiedType()) { 11321 Size = CGF.getTypeSize(CVD->getType()); 11322 // Align the size: ((size + align - 1) / align) * align 11323 Size = CGF.Builder.CreateNUWAdd( 11324 Size, CGM.getSize(Align - CharUnits::fromQuantity(1))); 11325 Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align)); 11326 Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align)); 11327 } else { 11328 CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType()); 11329 Size = CGM.getSize(Sz.alignTo(Align)); 11330 } 11331 llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc()); 11332 assert(AA->getAllocator() && 11333 "Expected allocator expression for non-default allocator."); 11334 llvm::Value *Allocator = CGF.EmitScalarExpr(AA->getAllocator()); 11335 // According to the standard, the original allocator type is a enum (integer). 11336 // Convert to pointer type, if required. 11337 if (Allocator->getType()->isIntegerTy()) 11338 Allocator = CGF.Builder.CreateIntToPtr(Allocator, CGM.VoidPtrTy); 11339 else if (Allocator->getType()->isPointerTy()) 11340 Allocator = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Allocator, 11341 CGM.VoidPtrTy); 11342 llvm::Value *Args[] = {ThreadID, Size, Allocator}; 11343 11344 llvm::Value *Addr = 11345 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_alloc), Args, 11346 getName({CVD->getName(), ".void.addr"})); 11347 llvm::Value *FiniArgs[OMPAllocateCleanupTy::CleanupArgs] = {ThreadID, Addr, 11348 Allocator}; 11349 llvm::FunctionCallee FiniRTLFn = createRuntimeFunction(OMPRTL__kmpc_free); 11350 11351 CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 11352 llvm::makeArrayRef(FiniArgs)); 11353 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11354 Addr, 11355 CGF.ConvertTypeForMem(CGM.getContext().getPointerType(CVD->getType())), 11356 getName({CVD->getName(), ".addr"})); 11357 return Address(Addr, Align); 11358 } 11359 11360 CGOpenMPRuntime::NontemporalDeclsRAII::NontemporalDeclsRAII( 11361 CodeGenModule &CGM, const OMPLoopDirective &S) 11362 : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) { 11363 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11364 if (!NeedToPush) 11365 return; 11366 NontemporalDeclsSet &DS = 11367 CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back(); 11368 for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) { 11369 for (const Stmt *Ref : C->private_refs()) { 11370 const auto *SimpleRefExpr = cast<Expr>(Ref)->IgnoreParenImpCasts(); 11371 const ValueDecl *VD; 11372 if (const auto *DRE = dyn_cast<DeclRefExpr>(SimpleRefExpr)) { 11373 VD = DRE->getDecl(); 11374 } else { 11375 const auto *ME = cast<MemberExpr>(SimpleRefExpr); 11376 assert((ME->isImplicitCXXThis() || 11377 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) && 11378 "Expected member of current class."); 11379 VD = ME->getMemberDecl(); 11380 } 11381 DS.insert(VD); 11382 } 11383 } 11384 } 11385 11386 CGOpenMPRuntime::NontemporalDeclsRAII::~NontemporalDeclsRAII() { 11387 if (!NeedToPush) 11388 return; 11389 CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back(); 11390 } 11391 11392 bool CGOpenMPRuntime::isNontemporalDecl(const ValueDecl *VD) const { 11393 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11394 11395 return llvm::any_of( 11396 CGM.getOpenMPRuntime().NontemporalDeclsStack, 11397 [VD](const NontemporalDeclsSet &Set) { return Set.count(VD) > 0; }); 11398 } 11399 11400 void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis( 11401 const OMPExecutableDirective &S, 11402 llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled) 11403 const { 11404 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs; 11405 // Vars in target/task regions must be excluded completely. 11406 if (isOpenMPTargetExecutionDirective(S.getDirectiveKind()) || 11407 isOpenMPTaskingDirective(S.getDirectiveKind())) { 11408 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions; 11409 getOpenMPCaptureRegions(CaptureRegions, S.getDirectiveKind()); 11410 const CapturedStmt *CS = S.getCapturedStmt(CaptureRegions.front()); 11411 for (const CapturedStmt::Capture &Cap : CS->captures()) { 11412 if (Cap.capturesVariable() || Cap.capturesVariableByCopy()) 11413 NeedToCheckForLPCs.insert(Cap.getCapturedVar()); 11414 } 11415 } 11416 // Exclude vars in private clauses. 11417 for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) { 11418 for (const Expr *Ref : C->varlists()) { 11419 if (!Ref->getType()->isScalarType()) 11420 continue; 11421 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11422 if (!DRE) 11423 continue; 11424 NeedToCheckForLPCs.insert(DRE->getDecl()); 11425 } 11426 } 11427 for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) { 11428 for (const Expr *Ref : C->varlists()) { 11429 if (!Ref->getType()->isScalarType()) 11430 continue; 11431 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11432 if (!DRE) 11433 continue; 11434 NeedToCheckForLPCs.insert(DRE->getDecl()); 11435 } 11436 } 11437 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) { 11438 for (const Expr *Ref : C->varlists()) { 11439 if (!Ref->getType()->isScalarType()) 11440 continue; 11441 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11442 if (!DRE) 11443 continue; 11444 NeedToCheckForLPCs.insert(DRE->getDecl()); 11445 } 11446 } 11447 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) { 11448 for (const Expr *Ref : C->varlists()) { 11449 if (!Ref->getType()->isScalarType()) 11450 continue; 11451 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11452 if (!DRE) 11453 continue; 11454 NeedToCheckForLPCs.insert(DRE->getDecl()); 11455 } 11456 } 11457 for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) { 11458 for (const Expr *Ref : C->varlists()) { 11459 if (!Ref->getType()->isScalarType()) 11460 continue; 11461 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11462 if (!DRE) 11463 continue; 11464 NeedToCheckForLPCs.insert(DRE->getDecl()); 11465 } 11466 } 11467 for (const Decl *VD : NeedToCheckForLPCs) { 11468 for (const LastprivateConditionalData &Data : 11469 llvm::reverse(CGM.getOpenMPRuntime().LastprivateConditionalStack)) { 11470 if (Data.DeclToUniqueName.count(VD) > 0) { 11471 if (!Data.Disabled) 11472 NeedToAddForLPCsAsDisabled.insert(VD); 11473 break; 11474 } 11475 } 11476 } 11477 } 11478 11479 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII( 11480 CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal) 11481 : CGM(CGF.CGM), 11482 Action((CGM.getLangOpts().OpenMP >= 50 && 11483 llvm::any_of(S.getClausesOfKind<OMPLastprivateClause>(), 11484 [](const OMPLastprivateClause *C) { 11485 return C->getKind() == 11486 OMPC_LASTPRIVATE_conditional; 11487 })) 11488 ? ActionToDo::PushAsLastprivateConditional 11489 : ActionToDo::DoNotPush) { 11490 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11491 if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush) 11492 return; 11493 assert(Action == ActionToDo::PushAsLastprivateConditional && 11494 "Expected a push action."); 11495 LastprivateConditionalData &Data = 11496 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back(); 11497 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) { 11498 if (C->getKind() != OMPC_LASTPRIVATE_conditional) 11499 continue; 11500 11501 for (const Expr *Ref : C->varlists()) { 11502 Data.DeclToUniqueName.insert(std::make_pair( 11503 cast<DeclRefExpr>(Ref->IgnoreParenImpCasts())->getDecl(), 11504 SmallString<16>(generateUniqueName(CGM, "pl_cond", Ref)))); 11505 } 11506 } 11507 Data.IVLVal = IVLVal; 11508 Data.Fn = CGF.CurFn; 11509 } 11510 11511 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII( 11512 CodeGenFunction &CGF, const OMPExecutableDirective &S) 11513 : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) { 11514 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11515 if (CGM.getLangOpts().OpenMP < 50) 11516 return; 11517 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled; 11518 tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled); 11519 if (!NeedToAddForLPCsAsDisabled.empty()) { 11520 Action = ActionToDo::DisableLastprivateConditional; 11521 LastprivateConditionalData &Data = 11522 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back(); 11523 for (const Decl *VD : NeedToAddForLPCsAsDisabled) 11524 Data.DeclToUniqueName.insert(std::make_pair(VD, SmallString<16>())); 11525 Data.Fn = CGF.CurFn; 11526 Data.Disabled = true; 11527 } 11528 } 11529 11530 CGOpenMPRuntime::LastprivateConditionalRAII 11531 CGOpenMPRuntime::LastprivateConditionalRAII::disable( 11532 CodeGenFunction &CGF, const OMPExecutableDirective &S) { 11533 return LastprivateConditionalRAII(CGF, S); 11534 } 11535 11536 CGOpenMPRuntime::LastprivateConditionalRAII::~LastprivateConditionalRAII() { 11537 if (CGM.getLangOpts().OpenMP < 50) 11538 return; 11539 if (Action == ActionToDo::DisableLastprivateConditional) { 11540 assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled && 11541 "Expected list of disabled private vars."); 11542 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back(); 11543 } 11544 if (Action == ActionToDo::PushAsLastprivateConditional) { 11545 assert( 11546 !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled && 11547 "Expected list of lastprivate conditional vars."); 11548 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back(); 11549 } 11550 } 11551 11552 Address CGOpenMPRuntime::emitLastprivateConditionalInit(CodeGenFunction &CGF, 11553 const VarDecl *VD) { 11554 ASTContext &C = CGM.getContext(); 11555 auto I = LastprivateConditionalToTypes.find(CGF.CurFn); 11556 if (I == LastprivateConditionalToTypes.end()) 11557 I = LastprivateConditionalToTypes.try_emplace(CGF.CurFn).first; 11558 QualType NewType; 11559 const FieldDecl *VDField; 11560 const FieldDecl *FiredField; 11561 LValue BaseLVal; 11562 auto VI = I->getSecond().find(VD); 11563 if (VI == I->getSecond().end()) { 11564 RecordDecl *RD = C.buildImplicitRecord("lasprivate.conditional"); 11565 RD->startDefinition(); 11566 VDField = addFieldToRecordDecl(C, RD, VD->getType().getNonReferenceType()); 11567 FiredField = addFieldToRecordDecl(C, RD, C.CharTy); 11568 RD->completeDefinition(); 11569 NewType = C.getRecordType(RD); 11570 Address Addr = CGF.CreateMemTemp(NewType, C.getDeclAlign(VD), VD->getName()); 11571 BaseLVal = CGF.MakeAddrLValue(Addr, NewType, AlignmentSource::Decl); 11572 I->getSecond().try_emplace(VD, NewType, VDField, FiredField, BaseLVal); 11573 } else { 11574 NewType = std::get<0>(VI->getSecond()); 11575 VDField = std::get<1>(VI->getSecond()); 11576 FiredField = std::get<2>(VI->getSecond()); 11577 BaseLVal = std::get<3>(VI->getSecond()); 11578 } 11579 LValue FiredLVal = 11580 CGF.EmitLValueForField(BaseLVal, FiredField); 11581 CGF.EmitStoreOfScalar( 11582 llvm::ConstantInt::getNullValue(CGF.ConvertTypeForMem(C.CharTy)), 11583 FiredLVal); 11584 return CGF.EmitLValueForField(BaseLVal, VDField).getAddress(CGF); 11585 } 11586 11587 namespace { 11588 /// Checks if the lastprivate conditional variable is referenced in LHS. 11589 class LastprivateConditionalRefChecker final 11590 : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> { 11591 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM; 11592 const Expr *FoundE = nullptr; 11593 const Decl *FoundD = nullptr; 11594 StringRef UniqueDeclName; 11595 LValue IVLVal; 11596 llvm::Function *FoundFn = nullptr; 11597 SourceLocation Loc; 11598 11599 public: 11600 bool VisitDeclRefExpr(const DeclRefExpr *E) { 11601 for (const CGOpenMPRuntime::LastprivateConditionalData &D : 11602 llvm::reverse(LPM)) { 11603 auto It = D.DeclToUniqueName.find(E->getDecl()); 11604 if (It == D.DeclToUniqueName.end()) 11605 continue; 11606 if (D.Disabled) 11607 return false; 11608 FoundE = E; 11609 FoundD = E->getDecl()->getCanonicalDecl(); 11610 UniqueDeclName = It->second; 11611 IVLVal = D.IVLVal; 11612 FoundFn = D.Fn; 11613 break; 11614 } 11615 return FoundE == E; 11616 } 11617 bool VisitMemberExpr(const MemberExpr *E) { 11618 if (!CodeGenFunction::IsWrappedCXXThis(E->getBase())) 11619 return false; 11620 for (const CGOpenMPRuntime::LastprivateConditionalData &D : 11621 llvm::reverse(LPM)) { 11622 auto It = D.DeclToUniqueName.find(E->getMemberDecl()); 11623 if (It == D.DeclToUniqueName.end()) 11624 continue; 11625 if (D.Disabled) 11626 return false; 11627 FoundE = E; 11628 FoundD = E->getMemberDecl()->getCanonicalDecl(); 11629 UniqueDeclName = It->second; 11630 IVLVal = D.IVLVal; 11631 FoundFn = D.Fn; 11632 break; 11633 } 11634 return FoundE == E; 11635 } 11636 bool VisitStmt(const Stmt *S) { 11637 for (const Stmt *Child : S->children()) { 11638 if (!Child) 11639 continue; 11640 if (const auto *E = dyn_cast<Expr>(Child)) 11641 if (!E->isGLValue()) 11642 continue; 11643 if (Visit(Child)) 11644 return true; 11645 } 11646 return false; 11647 } 11648 explicit LastprivateConditionalRefChecker( 11649 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM) 11650 : LPM(LPM) {} 11651 std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *> 11652 getFoundData() const { 11653 return std::make_tuple(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn); 11654 } 11655 }; 11656 } // namespace 11657 11658 void CGOpenMPRuntime::emitLastprivateConditionalUpdate(CodeGenFunction &CGF, 11659 LValue IVLVal, 11660 StringRef UniqueDeclName, 11661 LValue LVal, 11662 SourceLocation Loc) { 11663 // Last updated loop counter for the lastprivate conditional var. 11664 // int<xx> last_iv = 0; 11665 llvm::Type *LLIVTy = CGF.ConvertTypeForMem(IVLVal.getType()); 11666 llvm::Constant *LastIV = 11667 getOrCreateInternalVariable(LLIVTy, getName({UniqueDeclName, "iv"})); 11668 cast<llvm::GlobalVariable>(LastIV)->setAlignment( 11669 IVLVal.getAlignment().getAsAlign()); 11670 LValue LastIVLVal = CGF.MakeNaturalAlignAddrLValue(LastIV, IVLVal.getType()); 11671 11672 // Last value of the lastprivate conditional. 11673 // decltype(priv_a) last_a; 11674 llvm::Constant *Last = getOrCreateInternalVariable( 11675 CGF.ConvertTypeForMem(LVal.getType()), UniqueDeclName); 11676 cast<llvm::GlobalVariable>(Last)->setAlignment( 11677 LVal.getAlignment().getAsAlign()); 11678 LValue LastLVal = 11679 CGF.MakeAddrLValue(Last, LVal.getType(), LVal.getAlignment()); 11680 11681 // Global loop counter. Required to handle inner parallel-for regions. 11682 // iv 11683 llvm::Value *IVVal = CGF.EmitLoadOfScalar(IVLVal, Loc); 11684 11685 // #pragma omp critical(a) 11686 // if (last_iv <= iv) { 11687 // last_iv = iv; 11688 // last_a = priv_a; 11689 // } 11690 auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal, 11691 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 11692 Action.Enter(CGF); 11693 llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(LastIVLVal, Loc); 11694 // (last_iv <= iv) ? Check if the variable is updated and store new 11695 // value in global var. 11696 llvm::Value *CmpRes; 11697 if (IVLVal.getType()->isSignedIntegerType()) { 11698 CmpRes = CGF.Builder.CreateICmpSLE(LastIVVal, IVVal); 11699 } else { 11700 assert(IVLVal.getType()->isUnsignedIntegerType() && 11701 "Loop iteration variable must be integer."); 11702 CmpRes = CGF.Builder.CreateICmpULE(LastIVVal, IVVal); 11703 } 11704 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lp_cond_then"); 11705 llvm::BasicBlock *ExitBB = CGF.createBasicBlock("lp_cond_exit"); 11706 CGF.Builder.CreateCondBr(CmpRes, ThenBB, ExitBB); 11707 // { 11708 CGF.EmitBlock(ThenBB); 11709 11710 // last_iv = iv; 11711 CGF.EmitStoreOfScalar(IVVal, LastIVLVal); 11712 11713 // last_a = priv_a; 11714 switch (CGF.getEvaluationKind(LVal.getType())) { 11715 case TEK_Scalar: { 11716 llvm::Value *PrivVal = CGF.EmitLoadOfScalar(LVal, Loc); 11717 CGF.EmitStoreOfScalar(PrivVal, LastLVal); 11718 break; 11719 } 11720 case TEK_Complex: { 11721 CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(LVal, Loc); 11722 CGF.EmitStoreOfComplex(PrivVal, LastLVal, /*isInit=*/false); 11723 break; 11724 } 11725 case TEK_Aggregate: 11726 llvm_unreachable( 11727 "Aggregates are not supported in lastprivate conditional."); 11728 } 11729 // } 11730 CGF.EmitBranch(ExitBB); 11731 // There is no need to emit line number for unconditional branch. 11732 (void)ApplyDebugLocation::CreateEmpty(CGF); 11733 CGF.EmitBlock(ExitBB, /*IsFinished=*/true); 11734 }; 11735 11736 if (CGM.getLangOpts().OpenMPSimd) { 11737 // Do not emit as a critical region as no parallel region could be emitted. 11738 RegionCodeGenTy ThenRCG(CodeGen); 11739 ThenRCG(CGF); 11740 } else { 11741 emitCriticalRegion(CGF, UniqueDeclName, CodeGen, Loc); 11742 } 11743 } 11744 11745 void CGOpenMPRuntime::checkAndEmitLastprivateConditional(CodeGenFunction &CGF, 11746 const Expr *LHS) { 11747 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty()) 11748 return; 11749 LastprivateConditionalRefChecker Checker(LastprivateConditionalStack); 11750 if (!Checker.Visit(LHS)) 11751 return; 11752 const Expr *FoundE; 11753 const Decl *FoundD; 11754 StringRef UniqueDeclName; 11755 LValue IVLVal; 11756 llvm::Function *FoundFn; 11757 std::tie(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn) = 11758 Checker.getFoundData(); 11759 if (FoundFn != CGF.CurFn) { 11760 // Special codegen for inner parallel regions. 11761 // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1; 11762 auto It = LastprivateConditionalToTypes[FoundFn].find(FoundD); 11763 assert(It != LastprivateConditionalToTypes[FoundFn].end() && 11764 "Lastprivate conditional is not found in outer region."); 11765 QualType StructTy = std::get<0>(It->getSecond()); 11766 const FieldDecl* FiredDecl = std::get<2>(It->getSecond()); 11767 LValue PrivLVal = CGF.EmitLValue(FoundE); 11768 Address StructAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11769 PrivLVal.getAddress(CGF), 11770 CGF.ConvertTypeForMem(CGF.getContext().getPointerType(StructTy))); 11771 LValue BaseLVal = 11772 CGF.MakeAddrLValue(StructAddr, StructTy, AlignmentSource::Decl); 11773 LValue FiredLVal = CGF.EmitLValueForField(BaseLVal, FiredDecl); 11774 CGF.EmitAtomicStore(RValue::get(llvm::ConstantInt::get( 11775 CGF.ConvertTypeForMem(FiredDecl->getType()), 1)), 11776 FiredLVal, llvm::AtomicOrdering::Unordered, 11777 /*IsVolatile=*/true, /*isInit=*/false); 11778 return; 11779 } 11780 11781 // Private address of the lastprivate conditional in the current context. 11782 // priv_a 11783 LValue LVal = CGF.EmitLValue(FoundE); 11784 emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal, 11785 FoundE->getExprLoc()); 11786 } 11787 11788 void CGOpenMPRuntime::checkAndEmitSharedLastprivateConditional( 11789 CodeGenFunction &CGF, const OMPExecutableDirective &D, 11790 const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) { 11791 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty()) 11792 return; 11793 auto Range = llvm::reverse(LastprivateConditionalStack); 11794 auto It = llvm::find_if( 11795 Range, [](const LastprivateConditionalData &D) { return !D.Disabled; }); 11796 if (It == Range.end() || It->Fn != CGF.CurFn) 11797 return; 11798 auto LPCI = LastprivateConditionalToTypes.find(It->Fn); 11799 assert(LPCI != LastprivateConditionalToTypes.end() && 11800 "Lastprivates must be registered already."); 11801 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions; 11802 getOpenMPCaptureRegions(CaptureRegions, D.getDirectiveKind()); 11803 const CapturedStmt *CS = D.getCapturedStmt(CaptureRegions.back()); 11804 for (const auto &Pair : It->DeclToUniqueName) { 11805 const auto *VD = cast<VarDecl>(Pair.first->getCanonicalDecl()); 11806 if (!CS->capturesVariable(VD) || IgnoredDecls.count(VD) > 0) 11807 continue; 11808 auto I = LPCI->getSecond().find(Pair.first); 11809 assert(I != LPCI->getSecond().end() && 11810 "Lastprivate must be rehistered already."); 11811 // bool Cmp = priv_a.Fired != 0; 11812 LValue BaseLVal = std::get<3>(I->getSecond()); 11813 LValue FiredLVal = 11814 CGF.EmitLValueForField(BaseLVal, std::get<2>(I->getSecond())); 11815 llvm::Value *Res = CGF.EmitLoadOfScalar(FiredLVal, D.getBeginLoc()); 11816 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Res); 11817 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lpc.then"); 11818 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("lpc.done"); 11819 // if (Cmp) { 11820 CGF.Builder.CreateCondBr(Cmp, ThenBB, DoneBB); 11821 CGF.EmitBlock(ThenBB); 11822 Address Addr = CGF.GetAddrOfLocalVar(VD); 11823 LValue LVal; 11824 if (VD->getType()->isReferenceType()) 11825 LVal = CGF.EmitLoadOfReferenceLValue(Addr, VD->getType(), 11826 AlignmentSource::Decl); 11827 else 11828 LVal = CGF.MakeAddrLValue(Addr, VD->getType().getNonReferenceType(), 11829 AlignmentSource::Decl); 11830 emitLastprivateConditionalUpdate(CGF, It->IVLVal, Pair.second, LVal, 11831 D.getBeginLoc()); 11832 auto AL = ApplyDebugLocation::CreateArtificial(CGF); 11833 CGF.EmitBlock(DoneBB, /*IsFinal=*/true); 11834 // } 11835 } 11836 } 11837 11838 void CGOpenMPRuntime::emitLastprivateConditionalFinalUpdate( 11839 CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD, 11840 SourceLocation Loc) { 11841 if (CGF.getLangOpts().OpenMP < 50) 11842 return; 11843 auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(VD); 11844 assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() && 11845 "Unknown lastprivate conditional variable."); 11846 StringRef UniqueName = It->second; 11847 llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(UniqueName); 11848 // The variable was not updated in the region - exit. 11849 if (!GV) 11850 return; 11851 LValue LPLVal = CGF.MakeAddrLValue( 11852 GV, PrivLVal.getType().getNonReferenceType(), PrivLVal.getAlignment()); 11853 llvm::Value *Res = CGF.EmitLoadOfScalar(LPLVal, Loc); 11854 CGF.EmitStoreOfScalar(Res, PrivLVal); 11855 } 11856 11857 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction( 11858 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11859 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 11860 llvm_unreachable("Not supported in SIMD-only mode"); 11861 } 11862 11863 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction( 11864 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11865 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 11866 llvm_unreachable("Not supported in SIMD-only mode"); 11867 } 11868 11869 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction( 11870 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11871 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 11872 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 11873 bool Tied, unsigned &NumberOfParts) { 11874 llvm_unreachable("Not supported in SIMD-only mode"); 11875 } 11876 11877 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF, 11878 SourceLocation Loc, 11879 llvm::Function *OutlinedFn, 11880 ArrayRef<llvm::Value *> CapturedVars, 11881 const Expr *IfCond) { 11882 llvm_unreachable("Not supported in SIMD-only mode"); 11883 } 11884 11885 void CGOpenMPSIMDRuntime::emitCriticalRegion( 11886 CodeGenFunction &CGF, StringRef CriticalName, 11887 const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc, 11888 const Expr *Hint) { 11889 llvm_unreachable("Not supported in SIMD-only mode"); 11890 } 11891 11892 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF, 11893 const RegionCodeGenTy &MasterOpGen, 11894 SourceLocation Loc) { 11895 llvm_unreachable("Not supported in SIMD-only mode"); 11896 } 11897 11898 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 11899 SourceLocation Loc) { 11900 llvm_unreachable("Not supported in SIMD-only mode"); 11901 } 11902 11903 void CGOpenMPSIMDRuntime::emitTaskgroupRegion( 11904 CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen, 11905 SourceLocation Loc) { 11906 llvm_unreachable("Not supported in SIMD-only mode"); 11907 } 11908 11909 void CGOpenMPSIMDRuntime::emitSingleRegion( 11910 CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen, 11911 SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars, 11912 ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs, 11913 ArrayRef<const Expr *> AssignmentOps) { 11914 llvm_unreachable("Not supported in SIMD-only mode"); 11915 } 11916 11917 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF, 11918 const RegionCodeGenTy &OrderedOpGen, 11919 SourceLocation Loc, 11920 bool IsThreads) { 11921 llvm_unreachable("Not supported in SIMD-only mode"); 11922 } 11923 11924 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF, 11925 SourceLocation Loc, 11926 OpenMPDirectiveKind Kind, 11927 bool EmitChecks, 11928 bool ForceSimpleCall) { 11929 llvm_unreachable("Not supported in SIMD-only mode"); 11930 } 11931 11932 void CGOpenMPSIMDRuntime::emitForDispatchInit( 11933 CodeGenFunction &CGF, SourceLocation Loc, 11934 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 11935 bool Ordered, const DispatchRTInput &DispatchValues) { 11936 llvm_unreachable("Not supported in SIMD-only mode"); 11937 } 11938 11939 void CGOpenMPSIMDRuntime::emitForStaticInit( 11940 CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind, 11941 const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) { 11942 llvm_unreachable("Not supported in SIMD-only mode"); 11943 } 11944 11945 void CGOpenMPSIMDRuntime::emitDistributeStaticInit( 11946 CodeGenFunction &CGF, SourceLocation Loc, 11947 OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) { 11948 llvm_unreachable("Not supported in SIMD-only mode"); 11949 } 11950 11951 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 11952 SourceLocation Loc, 11953 unsigned IVSize, 11954 bool IVSigned) { 11955 llvm_unreachable("Not supported in SIMD-only mode"); 11956 } 11957 11958 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF, 11959 SourceLocation Loc, 11960 OpenMPDirectiveKind DKind) { 11961 llvm_unreachable("Not supported in SIMD-only mode"); 11962 } 11963 11964 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF, 11965 SourceLocation Loc, 11966 unsigned IVSize, bool IVSigned, 11967 Address IL, Address LB, 11968 Address UB, Address ST) { 11969 llvm_unreachable("Not supported in SIMD-only mode"); 11970 } 11971 11972 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 11973 llvm::Value *NumThreads, 11974 SourceLocation Loc) { 11975 llvm_unreachable("Not supported in SIMD-only mode"); 11976 } 11977 11978 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF, 11979 ProcBindKind ProcBind, 11980 SourceLocation Loc) { 11981 llvm_unreachable("Not supported in SIMD-only mode"); 11982 } 11983 11984 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 11985 const VarDecl *VD, 11986 Address VDAddr, 11987 SourceLocation Loc) { 11988 llvm_unreachable("Not supported in SIMD-only mode"); 11989 } 11990 11991 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition( 11992 const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit, 11993 CodeGenFunction *CGF) { 11994 llvm_unreachable("Not supported in SIMD-only mode"); 11995 } 11996 11997 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate( 11998 CodeGenFunction &CGF, QualType VarType, StringRef Name) { 11999 llvm_unreachable("Not supported in SIMD-only mode"); 12000 } 12001 12002 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF, 12003 ArrayRef<const Expr *> Vars, 12004 SourceLocation Loc, 12005 llvm::AtomicOrdering AO) { 12006 llvm_unreachable("Not supported in SIMD-only mode"); 12007 } 12008 12009 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 12010 const OMPExecutableDirective &D, 12011 llvm::Function *TaskFunction, 12012 QualType SharedsTy, Address Shareds, 12013 const Expr *IfCond, 12014 const OMPTaskDataTy &Data) { 12015 llvm_unreachable("Not supported in SIMD-only mode"); 12016 } 12017 12018 void CGOpenMPSIMDRuntime::emitTaskLoopCall( 12019 CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D, 12020 llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, 12021 const Expr *IfCond, const OMPTaskDataTy &Data) { 12022 llvm_unreachable("Not supported in SIMD-only mode"); 12023 } 12024 12025 void CGOpenMPSIMDRuntime::emitReduction( 12026 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates, 12027 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 12028 ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) { 12029 assert(Options.SimpleReduction && "Only simple reduction is expected."); 12030 CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs, 12031 ReductionOps, Options); 12032 } 12033 12034 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit( 12035 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 12036 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 12037 llvm_unreachable("Not supported in SIMD-only mode"); 12038 } 12039 12040 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 12041 SourceLocation Loc, 12042 ReductionCodeGen &RCG, 12043 unsigned N) { 12044 llvm_unreachable("Not supported in SIMD-only mode"); 12045 } 12046 12047 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF, 12048 SourceLocation Loc, 12049 llvm::Value *ReductionsPtr, 12050 LValue SharedLVal) { 12051 llvm_unreachable("Not supported in SIMD-only mode"); 12052 } 12053 12054 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 12055 SourceLocation Loc) { 12056 llvm_unreachable("Not supported in SIMD-only mode"); 12057 } 12058 12059 void CGOpenMPSIMDRuntime::emitCancellationPointCall( 12060 CodeGenFunction &CGF, SourceLocation Loc, 12061 OpenMPDirectiveKind CancelRegion) { 12062 llvm_unreachable("Not supported in SIMD-only mode"); 12063 } 12064 12065 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF, 12066 SourceLocation Loc, const Expr *IfCond, 12067 OpenMPDirectiveKind CancelRegion) { 12068 llvm_unreachable("Not supported in SIMD-only mode"); 12069 } 12070 12071 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction( 12072 const OMPExecutableDirective &D, StringRef ParentName, 12073 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 12074 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 12075 llvm_unreachable("Not supported in SIMD-only mode"); 12076 } 12077 12078 void CGOpenMPSIMDRuntime::emitTargetCall( 12079 CodeGenFunction &CGF, const OMPExecutableDirective &D, 12080 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 12081 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device, 12082 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 12083 const OMPLoopDirective &D)> 12084 SizeEmitter) { 12085 llvm_unreachable("Not supported in SIMD-only mode"); 12086 } 12087 12088 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) { 12089 llvm_unreachable("Not supported in SIMD-only mode"); 12090 } 12091 12092 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 12093 llvm_unreachable("Not supported in SIMD-only mode"); 12094 } 12095 12096 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) { 12097 return false; 12098 } 12099 12100 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF, 12101 const OMPExecutableDirective &D, 12102 SourceLocation Loc, 12103 llvm::Function *OutlinedFn, 12104 ArrayRef<llvm::Value *> CapturedVars) { 12105 llvm_unreachable("Not supported in SIMD-only mode"); 12106 } 12107 12108 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 12109 const Expr *NumTeams, 12110 const Expr *ThreadLimit, 12111 SourceLocation Loc) { 12112 llvm_unreachable("Not supported in SIMD-only mode"); 12113 } 12114 12115 void CGOpenMPSIMDRuntime::emitTargetDataCalls( 12116 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 12117 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 12118 llvm_unreachable("Not supported in SIMD-only mode"); 12119 } 12120 12121 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall( 12122 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 12123 const Expr *Device) { 12124 llvm_unreachable("Not supported in SIMD-only mode"); 12125 } 12126 12127 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF, 12128 const OMPLoopDirective &D, 12129 ArrayRef<Expr *> NumIterations) { 12130 llvm_unreachable("Not supported in SIMD-only mode"); 12131 } 12132 12133 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 12134 const OMPDependClause *C) { 12135 llvm_unreachable("Not supported in SIMD-only mode"); 12136 } 12137 12138 const VarDecl * 12139 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD, 12140 const VarDecl *NativeParam) const { 12141 llvm_unreachable("Not supported in SIMD-only mode"); 12142 } 12143 12144 Address 12145 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF, 12146 const VarDecl *NativeParam, 12147 const VarDecl *TargetParam) const { 12148 llvm_unreachable("Not supported in SIMD-only mode"); 12149 } 12150