1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This provides a class for OpenMP runtime code generation. 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "CGOpenMPRuntime.h" 14 #include "CGCXXABI.h" 15 #include "CGCleanup.h" 16 #include "CGRecordLayout.h" 17 #include "CodeGenFunction.h" 18 #include "clang/AST/Attr.h" 19 #include "clang/AST/Decl.h" 20 #include "clang/AST/OpenMPClause.h" 21 #include "clang/AST/StmtOpenMP.h" 22 #include "clang/AST/StmtVisitor.h" 23 #include "clang/Basic/BitmaskEnum.h" 24 #include "clang/Basic/OpenMPKinds.h" 25 #include "clang/CodeGen/ConstantInitBuilder.h" 26 #include "llvm/ADT/ArrayRef.h" 27 #include "llvm/ADT/SetOperations.h" 28 #include "llvm/ADT/StringExtras.h" 29 #include "llvm/Bitcode/BitcodeReader.h" 30 #include "llvm/Frontend/OpenMP/OMPIRBuilder.h" 31 #include "llvm/IR/DerivedTypes.h" 32 #include "llvm/IR/GlobalValue.h" 33 #include "llvm/IR/Value.h" 34 #include "llvm/Support/AtomicOrdering.h" 35 #include "llvm/Support/Format.h" 36 #include "llvm/Support/raw_ostream.h" 37 #include <cassert> 38 39 using namespace clang; 40 using namespace CodeGen; 41 using namespace llvm::omp; 42 43 namespace { 44 /// Base class for handling code generation inside OpenMP regions. 45 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo { 46 public: 47 /// Kinds of OpenMP regions used in codegen. 48 enum CGOpenMPRegionKind { 49 /// Region with outlined function for standalone 'parallel' 50 /// directive. 51 ParallelOutlinedRegion, 52 /// Region with outlined function for standalone 'task' directive. 53 TaskOutlinedRegion, 54 /// Region for constructs that do not require function outlining, 55 /// like 'for', 'sections', 'atomic' etc. directives. 56 InlinedRegion, 57 /// Region with outlined function for standalone 'target' directive. 58 TargetRegion, 59 }; 60 61 CGOpenMPRegionInfo(const CapturedStmt &CS, 62 const CGOpenMPRegionKind RegionKind, 63 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 64 bool HasCancel) 65 : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind), 66 CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {} 67 68 CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind, 69 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 70 bool HasCancel) 71 : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen), 72 Kind(Kind), HasCancel(HasCancel) {} 73 74 /// Get a variable or parameter for storing global thread id 75 /// inside OpenMP construct. 76 virtual const VarDecl *getThreadIDVariable() const = 0; 77 78 /// Emit the captured statement body. 79 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override; 80 81 /// Get an LValue for the current ThreadID variable. 82 /// \return LValue for thread id variable. This LValue always has type int32*. 83 virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF); 84 85 virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {} 86 87 CGOpenMPRegionKind getRegionKind() const { return RegionKind; } 88 89 OpenMPDirectiveKind getDirectiveKind() const { return Kind; } 90 91 bool hasCancel() const { return HasCancel; } 92 93 static bool classof(const CGCapturedStmtInfo *Info) { 94 return Info->getKind() == CR_OpenMP; 95 } 96 97 ~CGOpenMPRegionInfo() override = default; 98 99 protected: 100 CGOpenMPRegionKind RegionKind; 101 RegionCodeGenTy CodeGen; 102 OpenMPDirectiveKind Kind; 103 bool HasCancel; 104 }; 105 106 /// API for captured statement code generation in OpenMP constructs. 107 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo { 108 public: 109 CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar, 110 const RegionCodeGenTy &CodeGen, 111 OpenMPDirectiveKind Kind, bool HasCancel, 112 StringRef HelperName) 113 : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind, 114 HasCancel), 115 ThreadIDVar(ThreadIDVar), HelperName(HelperName) { 116 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 117 } 118 119 /// Get a variable or parameter for storing global thread id 120 /// inside OpenMP construct. 121 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 122 123 /// Get the name of the capture helper. 124 StringRef getHelperName() const override { return HelperName; } 125 126 static bool classof(const CGCapturedStmtInfo *Info) { 127 return CGOpenMPRegionInfo::classof(Info) && 128 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 129 ParallelOutlinedRegion; 130 } 131 132 private: 133 /// A variable or parameter storing global thread id for OpenMP 134 /// constructs. 135 const VarDecl *ThreadIDVar; 136 StringRef HelperName; 137 }; 138 139 /// API for captured statement code generation in OpenMP constructs. 140 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo { 141 public: 142 class UntiedTaskActionTy final : public PrePostActionTy { 143 bool Untied; 144 const VarDecl *PartIDVar; 145 const RegionCodeGenTy UntiedCodeGen; 146 llvm::SwitchInst *UntiedSwitch = nullptr; 147 148 public: 149 UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar, 150 const RegionCodeGenTy &UntiedCodeGen) 151 : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {} 152 void Enter(CodeGenFunction &CGF) override { 153 if (Untied) { 154 // Emit task switching point. 155 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 156 CGF.GetAddrOfLocalVar(PartIDVar), 157 PartIDVar->getType()->castAs<PointerType>()); 158 llvm::Value *Res = 159 CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation()); 160 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done."); 161 UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB); 162 CGF.EmitBlock(DoneBB); 163 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 164 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 165 UntiedSwitch->addCase(CGF.Builder.getInt32(0), 166 CGF.Builder.GetInsertBlock()); 167 emitUntiedSwitch(CGF); 168 } 169 } 170 void emitUntiedSwitch(CodeGenFunction &CGF) const { 171 if (Untied) { 172 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 173 CGF.GetAddrOfLocalVar(PartIDVar), 174 PartIDVar->getType()->castAs<PointerType>()); 175 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 176 PartIdLVal); 177 UntiedCodeGen(CGF); 178 CodeGenFunction::JumpDest CurPoint = 179 CGF.getJumpDestInCurrentScope(".untied.next."); 180 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 181 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 182 UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 183 CGF.Builder.GetInsertBlock()); 184 CGF.EmitBranchThroughCleanup(CurPoint); 185 CGF.EmitBlock(CurPoint.getBlock()); 186 } 187 } 188 unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); } 189 }; 190 CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS, 191 const VarDecl *ThreadIDVar, 192 const RegionCodeGenTy &CodeGen, 193 OpenMPDirectiveKind Kind, bool HasCancel, 194 const UntiedTaskActionTy &Action) 195 : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel), 196 ThreadIDVar(ThreadIDVar), Action(Action) { 197 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 198 } 199 200 /// Get a variable or parameter for storing global thread id 201 /// inside OpenMP construct. 202 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 203 204 /// Get an LValue for the current ThreadID variable. 205 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override; 206 207 /// Get the name of the capture helper. 208 StringRef getHelperName() const override { return ".omp_outlined."; } 209 210 void emitUntiedSwitch(CodeGenFunction &CGF) override { 211 Action.emitUntiedSwitch(CGF); 212 } 213 214 static bool classof(const CGCapturedStmtInfo *Info) { 215 return CGOpenMPRegionInfo::classof(Info) && 216 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 217 TaskOutlinedRegion; 218 } 219 220 private: 221 /// A variable or parameter storing global thread id for OpenMP 222 /// constructs. 223 const VarDecl *ThreadIDVar; 224 /// Action for emitting code for untied tasks. 225 const UntiedTaskActionTy &Action; 226 }; 227 228 /// API for inlined captured statement code generation in OpenMP 229 /// constructs. 230 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo { 231 public: 232 CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI, 233 const RegionCodeGenTy &CodeGen, 234 OpenMPDirectiveKind Kind, bool HasCancel) 235 : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel), 236 OldCSI(OldCSI), 237 OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {} 238 239 // Retrieve the value of the context parameter. 240 llvm::Value *getContextValue() const override { 241 if (OuterRegionInfo) 242 return OuterRegionInfo->getContextValue(); 243 llvm_unreachable("No context value for inlined OpenMP region"); 244 } 245 246 void setContextValue(llvm::Value *V) override { 247 if (OuterRegionInfo) { 248 OuterRegionInfo->setContextValue(V); 249 return; 250 } 251 llvm_unreachable("No context value for inlined OpenMP region"); 252 } 253 254 /// Lookup the captured field decl for a variable. 255 const FieldDecl *lookup(const VarDecl *VD) const override { 256 if (OuterRegionInfo) 257 return OuterRegionInfo->lookup(VD); 258 // If there is no outer outlined region,no need to lookup in a list of 259 // captured variables, we can use the original one. 260 return nullptr; 261 } 262 263 FieldDecl *getThisFieldDecl() const override { 264 if (OuterRegionInfo) 265 return OuterRegionInfo->getThisFieldDecl(); 266 return nullptr; 267 } 268 269 /// Get a variable or parameter for storing global thread id 270 /// inside OpenMP construct. 271 const VarDecl *getThreadIDVariable() const override { 272 if (OuterRegionInfo) 273 return OuterRegionInfo->getThreadIDVariable(); 274 return nullptr; 275 } 276 277 /// Get an LValue for the current ThreadID variable. 278 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override { 279 if (OuterRegionInfo) 280 return OuterRegionInfo->getThreadIDVariableLValue(CGF); 281 llvm_unreachable("No LValue for inlined OpenMP construct"); 282 } 283 284 /// Get the name of the capture helper. 285 StringRef getHelperName() const override { 286 if (auto *OuterRegionInfo = getOldCSI()) 287 return OuterRegionInfo->getHelperName(); 288 llvm_unreachable("No helper name for inlined OpenMP construct"); 289 } 290 291 void emitUntiedSwitch(CodeGenFunction &CGF) override { 292 if (OuterRegionInfo) 293 OuterRegionInfo->emitUntiedSwitch(CGF); 294 } 295 296 CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; } 297 298 static bool classof(const CGCapturedStmtInfo *Info) { 299 return CGOpenMPRegionInfo::classof(Info) && 300 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion; 301 } 302 303 ~CGOpenMPInlinedRegionInfo() override = default; 304 305 private: 306 /// CodeGen info about outer OpenMP region. 307 CodeGenFunction::CGCapturedStmtInfo *OldCSI; 308 CGOpenMPRegionInfo *OuterRegionInfo; 309 }; 310 311 /// API for captured statement code generation in OpenMP target 312 /// constructs. For this captures, implicit parameters are used instead of the 313 /// captured fields. The name of the target region has to be unique in a given 314 /// application so it is provided by the client, because only the client has 315 /// the information to generate that. 316 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo { 317 public: 318 CGOpenMPTargetRegionInfo(const CapturedStmt &CS, 319 const RegionCodeGenTy &CodeGen, StringRef HelperName) 320 : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target, 321 /*HasCancel=*/false), 322 HelperName(HelperName) {} 323 324 /// This is unused for target regions because each starts executing 325 /// with a single thread. 326 const VarDecl *getThreadIDVariable() const override { return nullptr; } 327 328 /// Get the name of the capture helper. 329 StringRef getHelperName() const override { return HelperName; } 330 331 static bool classof(const CGCapturedStmtInfo *Info) { 332 return CGOpenMPRegionInfo::classof(Info) && 333 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion; 334 } 335 336 private: 337 StringRef HelperName; 338 }; 339 340 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) { 341 llvm_unreachable("No codegen for expressions"); 342 } 343 /// API for generation of expressions captured in a innermost OpenMP 344 /// region. 345 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo { 346 public: 347 CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS) 348 : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen, 349 OMPD_unknown, 350 /*HasCancel=*/false), 351 PrivScope(CGF) { 352 // Make sure the globals captured in the provided statement are local by 353 // using the privatization logic. We assume the same variable is not 354 // captured more than once. 355 for (const auto &C : CS.captures()) { 356 if (!C.capturesVariable() && !C.capturesVariableByCopy()) 357 continue; 358 359 const VarDecl *VD = C.getCapturedVar(); 360 if (VD->isLocalVarDeclOrParm()) 361 continue; 362 363 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD), 364 /*RefersToEnclosingVariableOrCapture=*/false, 365 VD->getType().getNonReferenceType(), VK_LValue, 366 C.getLocation()); 367 PrivScope.addPrivate( 368 VD, [&CGF, &DRE]() { return CGF.EmitLValue(&DRE).getAddress(CGF); }); 369 } 370 (void)PrivScope.Privatize(); 371 } 372 373 /// Lookup the captured field decl for a variable. 374 const FieldDecl *lookup(const VarDecl *VD) const override { 375 if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD)) 376 return FD; 377 return nullptr; 378 } 379 380 /// Emit the captured statement body. 381 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override { 382 llvm_unreachable("No body for expressions"); 383 } 384 385 /// Get a variable or parameter for storing global thread id 386 /// inside OpenMP construct. 387 const VarDecl *getThreadIDVariable() const override { 388 llvm_unreachable("No thread id for expressions"); 389 } 390 391 /// Get the name of the capture helper. 392 StringRef getHelperName() const override { 393 llvm_unreachable("No helper name for expressions"); 394 } 395 396 static bool classof(const CGCapturedStmtInfo *Info) { return false; } 397 398 private: 399 /// Private scope to capture global variables. 400 CodeGenFunction::OMPPrivateScope PrivScope; 401 }; 402 403 /// RAII for emitting code of OpenMP constructs. 404 class InlinedOpenMPRegionRAII { 405 CodeGenFunction &CGF; 406 llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields; 407 FieldDecl *LambdaThisCaptureField = nullptr; 408 const CodeGen::CGBlockInfo *BlockInfo = nullptr; 409 410 public: 411 /// Constructs region for combined constructs. 412 /// \param CodeGen Code generation sequence for combined directives. Includes 413 /// a list of functions used for code generation of implicitly inlined 414 /// regions. 415 InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen, 416 OpenMPDirectiveKind Kind, bool HasCancel) 417 : CGF(CGF) { 418 // Start emission for the construct. 419 CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo( 420 CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel); 421 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 422 LambdaThisCaptureField = CGF.LambdaThisCaptureField; 423 CGF.LambdaThisCaptureField = nullptr; 424 BlockInfo = CGF.BlockInfo; 425 CGF.BlockInfo = nullptr; 426 } 427 428 ~InlinedOpenMPRegionRAII() { 429 // Restore original CapturedStmtInfo only if we're done with code emission. 430 auto *OldCSI = 431 cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI(); 432 delete CGF.CapturedStmtInfo; 433 CGF.CapturedStmtInfo = OldCSI; 434 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 435 CGF.LambdaThisCaptureField = LambdaThisCaptureField; 436 CGF.BlockInfo = BlockInfo; 437 } 438 }; 439 440 /// Values for bit flags used in the ident_t to describe the fields. 441 /// All enumeric elements are named and described in accordance with the code 442 /// from https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 443 enum OpenMPLocationFlags : unsigned { 444 /// Use trampoline for internal microtask. 445 OMP_IDENT_IMD = 0x01, 446 /// Use c-style ident structure. 447 OMP_IDENT_KMPC = 0x02, 448 /// Atomic reduction option for kmpc_reduce. 449 OMP_ATOMIC_REDUCE = 0x10, 450 /// Explicit 'barrier' directive. 451 OMP_IDENT_BARRIER_EXPL = 0x20, 452 /// Implicit barrier in code. 453 OMP_IDENT_BARRIER_IMPL = 0x40, 454 /// Implicit barrier in 'for' directive. 455 OMP_IDENT_BARRIER_IMPL_FOR = 0x40, 456 /// Implicit barrier in 'sections' directive. 457 OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0, 458 /// Implicit barrier in 'single' directive. 459 OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140, 460 /// Call of __kmp_for_static_init for static loop. 461 OMP_IDENT_WORK_LOOP = 0x200, 462 /// Call of __kmp_for_static_init for sections. 463 OMP_IDENT_WORK_SECTIONS = 0x400, 464 /// Call of __kmp_for_static_init for distribute. 465 OMP_IDENT_WORK_DISTRIBUTE = 0x800, 466 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE) 467 }; 468 469 namespace { 470 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 471 /// Values for bit flags for marking which requires clauses have been used. 472 enum OpenMPOffloadingRequiresDirFlags : int64_t { 473 /// flag undefined. 474 OMP_REQ_UNDEFINED = 0x000, 475 /// no requires clause present. 476 OMP_REQ_NONE = 0x001, 477 /// reverse_offload clause. 478 OMP_REQ_REVERSE_OFFLOAD = 0x002, 479 /// unified_address clause. 480 OMP_REQ_UNIFIED_ADDRESS = 0x004, 481 /// unified_shared_memory clause. 482 OMP_REQ_UNIFIED_SHARED_MEMORY = 0x008, 483 /// dynamic_allocators clause. 484 OMP_REQ_DYNAMIC_ALLOCATORS = 0x010, 485 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS) 486 }; 487 488 enum OpenMPOffloadingReservedDeviceIDs { 489 /// Device ID if the device was not defined, runtime should get it 490 /// from environment variables in the spec. 491 OMP_DEVICEID_UNDEF = -1, 492 }; 493 } // anonymous namespace 494 495 /// Describes ident structure that describes a source location. 496 /// All descriptions are taken from 497 /// https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 498 /// Original structure: 499 /// typedef struct ident { 500 /// kmp_int32 reserved_1; /**< might be used in Fortran; 501 /// see above */ 502 /// kmp_int32 flags; /**< also f.flags; KMP_IDENT_xxx flags; 503 /// KMP_IDENT_KMPC identifies this union 504 /// member */ 505 /// kmp_int32 reserved_2; /**< not really used in Fortran any more; 506 /// see above */ 507 ///#if USE_ITT_BUILD 508 /// /* but currently used for storing 509 /// region-specific ITT */ 510 /// /* contextual information. */ 511 ///#endif /* USE_ITT_BUILD */ 512 /// kmp_int32 reserved_3; /**< source[4] in Fortran, do not use for 513 /// C++ */ 514 /// char const *psource; /**< String describing the source location. 515 /// The string is composed of semi-colon separated 516 // fields which describe the source file, 517 /// the function and a pair of line numbers that 518 /// delimit the construct. 519 /// */ 520 /// } ident_t; 521 enum IdentFieldIndex { 522 /// might be used in Fortran 523 IdentField_Reserved_1, 524 /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member. 525 IdentField_Flags, 526 /// Not really used in Fortran any more 527 IdentField_Reserved_2, 528 /// Source[4] in Fortran, do not use for C++ 529 IdentField_Reserved_3, 530 /// String describing the source location. The string is composed of 531 /// semi-colon separated fields which describe the source file, the function 532 /// and a pair of line numbers that delimit the construct. 533 IdentField_PSource 534 }; 535 536 /// Schedule types for 'omp for' loops (these enumerators are taken from 537 /// the enum sched_type in kmp.h). 538 enum OpenMPSchedType { 539 /// Lower bound for default (unordered) versions. 540 OMP_sch_lower = 32, 541 OMP_sch_static_chunked = 33, 542 OMP_sch_static = 34, 543 OMP_sch_dynamic_chunked = 35, 544 OMP_sch_guided_chunked = 36, 545 OMP_sch_runtime = 37, 546 OMP_sch_auto = 38, 547 /// static with chunk adjustment (e.g., simd) 548 OMP_sch_static_balanced_chunked = 45, 549 /// Lower bound for 'ordered' versions. 550 OMP_ord_lower = 64, 551 OMP_ord_static_chunked = 65, 552 OMP_ord_static = 66, 553 OMP_ord_dynamic_chunked = 67, 554 OMP_ord_guided_chunked = 68, 555 OMP_ord_runtime = 69, 556 OMP_ord_auto = 70, 557 OMP_sch_default = OMP_sch_static, 558 /// dist_schedule types 559 OMP_dist_sch_static_chunked = 91, 560 OMP_dist_sch_static = 92, 561 /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers. 562 /// Set if the monotonic schedule modifier was present. 563 OMP_sch_modifier_monotonic = (1 << 29), 564 /// Set if the nonmonotonic schedule modifier was present. 565 OMP_sch_modifier_nonmonotonic = (1 << 30), 566 }; 567 568 enum OpenMPRTLFunction { 569 /// Call to void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, 570 /// kmpc_micro microtask, ...); 571 OMPRTL__kmpc_fork_call, 572 /// Call to void *__kmpc_threadprivate_cached(ident_t *loc, 573 /// kmp_int32 global_tid, void *data, size_t size, void ***cache); 574 OMPRTL__kmpc_threadprivate_cached, 575 /// Call to void __kmpc_threadprivate_register( ident_t *, 576 /// void *data, kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor); 577 OMPRTL__kmpc_threadprivate_register, 578 // Call to __kmpc_int32 kmpc_global_thread_num(ident_t *loc); 579 OMPRTL__kmpc_global_thread_num, 580 // Call to void __kmpc_critical(ident_t *loc, kmp_int32 global_tid, 581 // kmp_critical_name *crit); 582 OMPRTL__kmpc_critical, 583 // Call to void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 584 // global_tid, kmp_critical_name *crit, uintptr_t hint); 585 OMPRTL__kmpc_critical_with_hint, 586 // Call to void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid, 587 // kmp_critical_name *crit); 588 OMPRTL__kmpc_end_critical, 589 // Call to kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32 590 // global_tid); 591 OMPRTL__kmpc_cancel_barrier, 592 // Call to void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid); 593 OMPRTL__kmpc_barrier, 594 // Call to void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid); 595 OMPRTL__kmpc_for_static_fini, 596 // Call to void __kmpc_serialized_parallel(ident_t *loc, kmp_int32 597 // global_tid); 598 OMPRTL__kmpc_serialized_parallel, 599 // Call to void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32 600 // global_tid); 601 OMPRTL__kmpc_end_serialized_parallel, 602 // Call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid, 603 // kmp_int32 num_threads); 604 OMPRTL__kmpc_push_num_threads, 605 // Call to void __kmpc_flush(ident_t *loc); 606 OMPRTL__kmpc_flush, 607 // Call to kmp_int32 __kmpc_master(ident_t *, kmp_int32 global_tid); 608 OMPRTL__kmpc_master, 609 // Call to void __kmpc_end_master(ident_t *, kmp_int32 global_tid); 610 OMPRTL__kmpc_end_master, 611 // Call to kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid, 612 // int end_part); 613 OMPRTL__kmpc_omp_taskyield, 614 // Call to kmp_int32 __kmpc_single(ident_t *, kmp_int32 global_tid); 615 OMPRTL__kmpc_single, 616 // Call to void __kmpc_end_single(ident_t *, kmp_int32 global_tid); 617 OMPRTL__kmpc_end_single, 618 // Call to kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 619 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 620 // kmp_routine_entry_t *task_entry); 621 OMPRTL__kmpc_omp_task_alloc, 622 // Call to kmp_task_t * __kmpc_omp_target_task_alloc(ident_t *, 623 // kmp_int32 gtid, kmp_int32 flags, size_t sizeof_kmp_task_t, 624 // size_t sizeof_shareds, kmp_routine_entry_t *task_entry, 625 // kmp_int64 device_id); 626 OMPRTL__kmpc_omp_target_task_alloc, 627 // Call to kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t * 628 // new_task); 629 OMPRTL__kmpc_omp_task, 630 // Call to void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid, 631 // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *), 632 // kmp_int32 didit); 633 OMPRTL__kmpc_copyprivate, 634 // Call to kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid, 635 // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void 636 // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck); 637 OMPRTL__kmpc_reduce, 638 // Call to kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32 639 // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, 640 // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name 641 // *lck); 642 OMPRTL__kmpc_reduce_nowait, 643 // Call to void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid, 644 // kmp_critical_name *lck); 645 OMPRTL__kmpc_end_reduce, 646 // Call to void __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid, 647 // kmp_critical_name *lck); 648 OMPRTL__kmpc_end_reduce_nowait, 649 // Call to void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 650 // kmp_task_t * new_task); 651 OMPRTL__kmpc_omp_task_begin_if0, 652 // Call to void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 653 // kmp_task_t * new_task); 654 OMPRTL__kmpc_omp_task_complete_if0, 655 // Call to void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid); 656 OMPRTL__kmpc_ordered, 657 // Call to void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid); 658 OMPRTL__kmpc_end_ordered, 659 // Call to kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 660 // global_tid); 661 OMPRTL__kmpc_omp_taskwait, 662 // Call to void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid); 663 OMPRTL__kmpc_taskgroup, 664 // Call to void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid); 665 OMPRTL__kmpc_end_taskgroup, 666 // Call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid, 667 // int proc_bind); 668 OMPRTL__kmpc_push_proc_bind, 669 // Call to kmp_int32 __kmpc_omp_task_with_deps(ident_t *loc_ref, kmp_int32 670 // gtid, kmp_task_t * new_task, kmp_int32 ndeps, kmp_depend_info_t 671 // *dep_list, kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 672 OMPRTL__kmpc_omp_task_with_deps, 673 // Call to void __kmpc_omp_wait_deps(ident_t *loc_ref, kmp_int32 674 // gtid, kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 675 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 676 OMPRTL__kmpc_omp_wait_deps, 677 // Call to kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 678 // global_tid, kmp_int32 cncl_kind); 679 OMPRTL__kmpc_cancellationpoint, 680 // Call to kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 681 // kmp_int32 cncl_kind); 682 OMPRTL__kmpc_cancel, 683 // Call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32 global_tid, 684 // kmp_int32 num_teams, kmp_int32 thread_limit); 685 OMPRTL__kmpc_push_num_teams, 686 // Call to void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro 687 // microtask, ...); 688 OMPRTL__kmpc_fork_teams, 689 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 690 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 691 // sched, kmp_uint64 grainsize, void *task_dup); 692 OMPRTL__kmpc_taskloop, 693 // Call to void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32 694 // num_dims, struct kmp_dim *dims); 695 OMPRTL__kmpc_doacross_init, 696 // Call to void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid); 697 OMPRTL__kmpc_doacross_fini, 698 // Call to void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64 699 // *vec); 700 OMPRTL__kmpc_doacross_post, 701 // Call to void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64 702 // *vec); 703 OMPRTL__kmpc_doacross_wait, 704 // Call to void *__kmpc_task_reduction_init(int gtid, int num_data, void 705 // *data); 706 OMPRTL__kmpc_task_reduction_init, 707 // Call to void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 708 // *d); 709 OMPRTL__kmpc_task_reduction_get_th_data, 710 // Call to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t al); 711 OMPRTL__kmpc_alloc, 712 // Call to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t al); 713 OMPRTL__kmpc_free, 714 715 // 716 // Offloading related calls 717 // 718 // Call to void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64 719 // size); 720 OMPRTL__kmpc_push_target_tripcount, 721 // Call to int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t 722 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 723 // *arg_types); 724 OMPRTL__tgt_target, 725 // Call to int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr, 726 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 727 // *arg_types); 728 OMPRTL__tgt_target_nowait, 729 // Call to int32_t __tgt_target_teams(int64_t device_id, void *host_ptr, 730 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 731 // *arg_types, int32_t num_teams, int32_t thread_limit); 732 OMPRTL__tgt_target_teams, 733 // Call to int32_t __tgt_target_teams_nowait(int64_t device_id, void 734 // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t 735 // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 736 OMPRTL__tgt_target_teams_nowait, 737 // Call to void __tgt_register_requires(int64_t flags); 738 OMPRTL__tgt_register_requires, 739 // Call to void __tgt_target_data_begin(int64_t device_id, int32_t arg_num, 740 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 741 OMPRTL__tgt_target_data_begin, 742 // Call to void __tgt_target_data_begin_nowait(int64_t device_id, int32_t 743 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 744 // *arg_types); 745 OMPRTL__tgt_target_data_begin_nowait, 746 // Call to void __tgt_target_data_end(int64_t device_id, int32_t arg_num, 747 // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types); 748 OMPRTL__tgt_target_data_end, 749 // Call to void __tgt_target_data_end_nowait(int64_t device_id, int32_t 750 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 751 // *arg_types); 752 OMPRTL__tgt_target_data_end_nowait, 753 // Call to void __tgt_target_data_update(int64_t device_id, int32_t arg_num, 754 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 755 OMPRTL__tgt_target_data_update, 756 // Call to void __tgt_target_data_update_nowait(int64_t device_id, int32_t 757 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 758 // *arg_types); 759 OMPRTL__tgt_target_data_update_nowait, 760 // Call to int64_t __tgt_mapper_num_components(void *rt_mapper_handle); 761 OMPRTL__tgt_mapper_num_components, 762 // Call to void __tgt_push_mapper_component(void *rt_mapper_handle, void 763 // *base, void *begin, int64_t size, int64_t type); 764 OMPRTL__tgt_push_mapper_component, 765 }; 766 767 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP 768 /// region. 769 class CleanupTy final : public EHScopeStack::Cleanup { 770 PrePostActionTy *Action; 771 772 public: 773 explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {} 774 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 775 if (!CGF.HaveInsertPoint()) 776 return; 777 Action->Exit(CGF); 778 } 779 }; 780 781 } // anonymous namespace 782 783 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const { 784 CodeGenFunction::RunCleanupsScope Scope(CGF); 785 if (PrePostAction) { 786 CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction); 787 Callback(CodeGen, CGF, *PrePostAction); 788 } else { 789 PrePostActionTy Action; 790 Callback(CodeGen, CGF, Action); 791 } 792 } 793 794 /// Check if the combiner is a call to UDR combiner and if it is so return the 795 /// UDR decl used for reduction. 796 static const OMPDeclareReductionDecl * 797 getReductionInit(const Expr *ReductionOp) { 798 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 799 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 800 if (const auto *DRE = 801 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 802 if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) 803 return DRD; 804 return nullptr; 805 } 806 807 static void emitInitWithReductionInitializer(CodeGenFunction &CGF, 808 const OMPDeclareReductionDecl *DRD, 809 const Expr *InitOp, 810 Address Private, Address Original, 811 QualType Ty) { 812 if (DRD->getInitializer()) { 813 std::pair<llvm::Function *, llvm::Function *> Reduction = 814 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 815 const auto *CE = cast<CallExpr>(InitOp); 816 const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee()); 817 const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts(); 818 const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts(); 819 const auto *LHSDRE = 820 cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr()); 821 const auto *RHSDRE = 822 cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr()); 823 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 824 PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()), 825 [=]() { return Private; }); 826 PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()), 827 [=]() { return Original; }); 828 (void)PrivateScope.Privatize(); 829 RValue Func = RValue::get(Reduction.second); 830 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 831 CGF.EmitIgnoredExpr(InitOp); 832 } else { 833 llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty); 834 std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"}); 835 auto *GV = new llvm::GlobalVariable( 836 CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true, 837 llvm::GlobalValue::PrivateLinkage, Init, Name); 838 LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty); 839 RValue InitRVal; 840 switch (CGF.getEvaluationKind(Ty)) { 841 case TEK_Scalar: 842 InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation()); 843 break; 844 case TEK_Complex: 845 InitRVal = 846 RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation())); 847 break; 848 case TEK_Aggregate: 849 InitRVal = RValue::getAggregate(LV.getAddress(CGF)); 850 break; 851 } 852 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_RValue); 853 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal); 854 CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(), 855 /*IsInitializer=*/false); 856 } 857 } 858 859 /// Emit initialization of arrays of complex types. 860 /// \param DestAddr Address of the array. 861 /// \param Type Type of array. 862 /// \param Init Initial expression of array. 863 /// \param SrcAddr Address of the original array. 864 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr, 865 QualType Type, bool EmitDeclareReductionInit, 866 const Expr *Init, 867 const OMPDeclareReductionDecl *DRD, 868 Address SrcAddr = Address::invalid()) { 869 // Perform element-by-element initialization. 870 QualType ElementTy; 871 872 // Drill down to the base element type on both arrays. 873 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 874 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr); 875 DestAddr = 876 CGF.Builder.CreateElementBitCast(DestAddr, DestAddr.getElementType()); 877 if (DRD) 878 SrcAddr = 879 CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType()); 880 881 llvm::Value *SrcBegin = nullptr; 882 if (DRD) 883 SrcBegin = SrcAddr.getPointer(); 884 llvm::Value *DestBegin = DestAddr.getPointer(); 885 // Cast from pointer to array type to pointer to single element. 886 llvm::Value *DestEnd = CGF.Builder.CreateGEP(DestBegin, NumElements); 887 // The basic structure here is a while-do loop. 888 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body"); 889 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done"); 890 llvm::Value *IsEmpty = 891 CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty"); 892 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 893 894 // Enter the loop body, making that address the current address. 895 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 896 CGF.EmitBlock(BodyBB); 897 898 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 899 900 llvm::PHINode *SrcElementPHI = nullptr; 901 Address SrcElementCurrent = Address::invalid(); 902 if (DRD) { 903 SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2, 904 "omp.arraycpy.srcElementPast"); 905 SrcElementPHI->addIncoming(SrcBegin, EntryBB); 906 SrcElementCurrent = 907 Address(SrcElementPHI, 908 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 909 } 910 llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI( 911 DestBegin->getType(), 2, "omp.arraycpy.destElementPast"); 912 DestElementPHI->addIncoming(DestBegin, EntryBB); 913 Address DestElementCurrent = 914 Address(DestElementPHI, 915 DestAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 916 917 // Emit copy. 918 { 919 CodeGenFunction::RunCleanupsScope InitScope(CGF); 920 if (EmitDeclareReductionInit) { 921 emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent, 922 SrcElementCurrent, ElementTy); 923 } else 924 CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(), 925 /*IsInitializer=*/false); 926 } 927 928 if (DRD) { 929 // Shift the address forward by one element. 930 llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32( 931 SrcElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 932 SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock()); 933 } 934 935 // Shift the address forward by one element. 936 llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32( 937 DestElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 938 // Check whether we've reached the end. 939 llvm::Value *Done = 940 CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done"); 941 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 942 DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock()); 943 944 // Done. 945 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 946 } 947 948 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) { 949 return CGF.EmitOMPSharedLValue(E); 950 } 951 952 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF, 953 const Expr *E) { 954 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E)) 955 return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false); 956 return LValue(); 957 } 958 959 void ReductionCodeGen::emitAggregateInitialization( 960 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 961 const OMPDeclareReductionDecl *DRD) { 962 // Emit VarDecl with copy init for arrays. 963 // Get the address of the original variable captured in current 964 // captured region. 965 const auto *PrivateVD = 966 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 967 bool EmitDeclareReductionInit = 968 DRD && (DRD->getInitializer() || !PrivateVD->hasInit()); 969 EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(), 970 EmitDeclareReductionInit, 971 EmitDeclareReductionInit ? ClausesData[N].ReductionOp 972 : PrivateVD->getInit(), 973 DRD, SharedLVal.getAddress(CGF)); 974 } 975 976 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds, 977 ArrayRef<const Expr *> Privates, 978 ArrayRef<const Expr *> ReductionOps) { 979 ClausesData.reserve(Shareds.size()); 980 SharedAddresses.reserve(Shareds.size()); 981 Sizes.reserve(Shareds.size()); 982 BaseDecls.reserve(Shareds.size()); 983 auto IPriv = Privates.begin(); 984 auto IRed = ReductionOps.begin(); 985 for (const Expr *Ref : Shareds) { 986 ClausesData.emplace_back(Ref, *IPriv, *IRed); 987 std::advance(IPriv, 1); 988 std::advance(IRed, 1); 989 } 990 } 991 992 void ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, unsigned N) { 993 assert(SharedAddresses.size() == N && 994 "Number of generated lvalues must be exactly N."); 995 LValue First = emitSharedLValue(CGF, ClausesData[N].Ref); 996 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref); 997 SharedAddresses.emplace_back(First, Second); 998 } 999 1000 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) { 1001 const auto *PrivateVD = 1002 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1003 QualType PrivateType = PrivateVD->getType(); 1004 bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref); 1005 if (!PrivateType->isVariablyModifiedType()) { 1006 Sizes.emplace_back( 1007 CGF.getTypeSize( 1008 SharedAddresses[N].first.getType().getNonReferenceType()), 1009 nullptr); 1010 return; 1011 } 1012 llvm::Value *Size; 1013 llvm::Value *SizeInChars; 1014 auto *ElemType = cast<llvm::PointerType>( 1015 SharedAddresses[N].first.getPointer(CGF)->getType()) 1016 ->getElementType(); 1017 auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType); 1018 if (AsArraySection) { 1019 Size = CGF.Builder.CreatePtrDiff(SharedAddresses[N].second.getPointer(CGF), 1020 SharedAddresses[N].first.getPointer(CGF)); 1021 Size = CGF.Builder.CreateNUWAdd( 1022 Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1)); 1023 SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf); 1024 } else { 1025 SizeInChars = CGF.getTypeSize( 1026 SharedAddresses[N].first.getType().getNonReferenceType()); 1027 Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf); 1028 } 1029 Sizes.emplace_back(SizeInChars, Size); 1030 CodeGenFunction::OpaqueValueMapping OpaqueMap( 1031 CGF, 1032 cast<OpaqueValueExpr>( 1033 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 1034 RValue::get(Size)); 1035 CGF.EmitVariablyModifiedType(PrivateType); 1036 } 1037 1038 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N, 1039 llvm::Value *Size) { 1040 const auto *PrivateVD = 1041 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1042 QualType PrivateType = PrivateVD->getType(); 1043 if (!PrivateType->isVariablyModifiedType()) { 1044 assert(!Size && !Sizes[N].second && 1045 "Size should be nullptr for non-variably modified reduction " 1046 "items."); 1047 return; 1048 } 1049 CodeGenFunction::OpaqueValueMapping OpaqueMap( 1050 CGF, 1051 cast<OpaqueValueExpr>( 1052 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 1053 RValue::get(Size)); 1054 CGF.EmitVariablyModifiedType(PrivateType); 1055 } 1056 1057 void ReductionCodeGen::emitInitialization( 1058 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 1059 llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) { 1060 assert(SharedAddresses.size() > N && "No variable was generated"); 1061 const auto *PrivateVD = 1062 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1063 const OMPDeclareReductionDecl *DRD = 1064 getReductionInit(ClausesData[N].ReductionOp); 1065 QualType PrivateType = PrivateVD->getType(); 1066 PrivateAddr = CGF.Builder.CreateElementBitCast( 1067 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 1068 QualType SharedType = SharedAddresses[N].first.getType(); 1069 SharedLVal = CGF.MakeAddrLValue( 1070 CGF.Builder.CreateElementBitCast(SharedLVal.getAddress(CGF), 1071 CGF.ConvertTypeForMem(SharedType)), 1072 SharedType, SharedAddresses[N].first.getBaseInfo(), 1073 CGF.CGM.getTBAAInfoForSubobject(SharedAddresses[N].first, SharedType)); 1074 if (CGF.getContext().getAsArrayType(PrivateVD->getType())) { 1075 emitAggregateInitialization(CGF, N, PrivateAddr, SharedLVal, DRD); 1076 } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) { 1077 emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp, 1078 PrivateAddr, SharedLVal.getAddress(CGF), 1079 SharedLVal.getType()); 1080 } else if (!DefaultInit(CGF) && PrivateVD->hasInit() && 1081 !CGF.isTrivialInitializer(PrivateVD->getInit())) { 1082 CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr, 1083 PrivateVD->getType().getQualifiers(), 1084 /*IsInitializer=*/false); 1085 } 1086 } 1087 1088 bool ReductionCodeGen::needCleanups(unsigned N) { 1089 const auto *PrivateVD = 1090 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1091 QualType PrivateType = PrivateVD->getType(); 1092 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 1093 return DTorKind != QualType::DK_none; 1094 } 1095 1096 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N, 1097 Address PrivateAddr) { 1098 const auto *PrivateVD = 1099 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1100 QualType PrivateType = PrivateVD->getType(); 1101 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 1102 if (needCleanups(N)) { 1103 PrivateAddr = CGF.Builder.CreateElementBitCast( 1104 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 1105 CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType); 1106 } 1107 } 1108 1109 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 1110 LValue BaseLV) { 1111 BaseTy = BaseTy.getNonReferenceType(); 1112 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 1113 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 1114 if (const auto *PtrTy = BaseTy->getAs<PointerType>()) { 1115 BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(CGF), PtrTy); 1116 } else { 1117 LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(CGF), BaseTy); 1118 BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal); 1119 } 1120 BaseTy = BaseTy->getPointeeType(); 1121 } 1122 return CGF.MakeAddrLValue( 1123 CGF.Builder.CreateElementBitCast(BaseLV.getAddress(CGF), 1124 CGF.ConvertTypeForMem(ElTy)), 1125 BaseLV.getType(), BaseLV.getBaseInfo(), 1126 CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType())); 1127 } 1128 1129 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 1130 llvm::Type *BaseLVType, CharUnits BaseLVAlignment, 1131 llvm::Value *Addr) { 1132 Address Tmp = Address::invalid(); 1133 Address TopTmp = Address::invalid(); 1134 Address MostTopTmp = Address::invalid(); 1135 BaseTy = BaseTy.getNonReferenceType(); 1136 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 1137 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 1138 Tmp = CGF.CreateMemTemp(BaseTy); 1139 if (TopTmp.isValid()) 1140 CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp); 1141 else 1142 MostTopTmp = Tmp; 1143 TopTmp = Tmp; 1144 BaseTy = BaseTy->getPointeeType(); 1145 } 1146 llvm::Type *Ty = BaseLVType; 1147 if (Tmp.isValid()) 1148 Ty = Tmp.getElementType(); 1149 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty); 1150 if (Tmp.isValid()) { 1151 CGF.Builder.CreateStore(Addr, Tmp); 1152 return MostTopTmp; 1153 } 1154 return Address(Addr, BaseLVAlignment); 1155 } 1156 1157 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) { 1158 const VarDecl *OrigVD = nullptr; 1159 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) { 1160 const Expr *Base = OASE->getBase()->IgnoreParenImpCasts(); 1161 while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base)) 1162 Base = TempOASE->getBase()->IgnoreParenImpCasts(); 1163 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 1164 Base = TempASE->getBase()->IgnoreParenImpCasts(); 1165 DE = cast<DeclRefExpr>(Base); 1166 OrigVD = cast<VarDecl>(DE->getDecl()); 1167 } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) { 1168 const Expr *Base = ASE->getBase()->IgnoreParenImpCasts(); 1169 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 1170 Base = TempASE->getBase()->IgnoreParenImpCasts(); 1171 DE = cast<DeclRefExpr>(Base); 1172 OrigVD = cast<VarDecl>(DE->getDecl()); 1173 } 1174 return OrigVD; 1175 } 1176 1177 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N, 1178 Address PrivateAddr) { 1179 const DeclRefExpr *DE; 1180 if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) { 1181 BaseDecls.emplace_back(OrigVD); 1182 LValue OriginalBaseLValue = CGF.EmitLValue(DE); 1183 LValue BaseLValue = 1184 loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(), 1185 OriginalBaseLValue); 1186 llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff( 1187 BaseLValue.getPointer(CGF), SharedAddresses[N].first.getPointer(CGF)); 1188 llvm::Value *PrivatePointer = 1189 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 1190 PrivateAddr.getPointer(), 1191 SharedAddresses[N].first.getAddress(CGF).getType()); 1192 llvm::Value *Ptr = CGF.Builder.CreateGEP(PrivatePointer, Adjustment); 1193 return castToBase(CGF, OrigVD->getType(), 1194 SharedAddresses[N].first.getType(), 1195 OriginalBaseLValue.getAddress(CGF).getType(), 1196 OriginalBaseLValue.getAlignment(), Ptr); 1197 } 1198 BaseDecls.emplace_back( 1199 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl())); 1200 return PrivateAddr; 1201 } 1202 1203 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const { 1204 const OMPDeclareReductionDecl *DRD = 1205 getReductionInit(ClausesData[N].ReductionOp); 1206 return DRD && DRD->getInitializer(); 1207 } 1208 1209 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) { 1210 return CGF.EmitLoadOfPointerLValue( 1211 CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1212 getThreadIDVariable()->getType()->castAs<PointerType>()); 1213 } 1214 1215 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) { 1216 if (!CGF.HaveInsertPoint()) 1217 return; 1218 // 1.2.2 OpenMP Language Terminology 1219 // Structured block - An executable statement with a single entry at the 1220 // top and a single exit at the bottom. 1221 // The point of exit cannot be a branch out of the structured block. 1222 // longjmp() and throw() must not violate the entry/exit criteria. 1223 CGF.EHStack.pushTerminate(); 1224 CodeGen(CGF); 1225 CGF.EHStack.popTerminate(); 1226 } 1227 1228 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue( 1229 CodeGenFunction &CGF) { 1230 return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1231 getThreadIDVariable()->getType(), 1232 AlignmentSource::Decl); 1233 } 1234 1235 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC, 1236 QualType FieldTy) { 1237 auto *Field = FieldDecl::Create( 1238 C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy, 1239 C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()), 1240 /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit); 1241 Field->setAccess(AS_public); 1242 DC->addDecl(Field); 1243 return Field; 1244 } 1245 1246 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator, 1247 StringRef Separator) 1248 : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator), 1249 OffloadEntriesInfoManager(CGM) { 1250 ASTContext &C = CGM.getContext(); 1251 RecordDecl *RD = C.buildImplicitRecord("ident_t"); 1252 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 1253 RD->startDefinition(); 1254 // reserved_1 1255 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1256 // flags 1257 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1258 // reserved_2 1259 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1260 // reserved_3 1261 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1262 // psource 1263 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 1264 RD->completeDefinition(); 1265 IdentQTy = C.getRecordType(RD); 1266 IdentTy = CGM.getTypes().ConvertRecordDeclType(RD); 1267 KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8); 1268 1269 loadOffloadInfoMetadata(); 1270 } 1271 1272 bool CGOpenMPRuntime::tryEmitDeclareVariant(const GlobalDecl &NewGD, 1273 const GlobalDecl &OldGD, 1274 llvm::GlobalValue *OrigAddr, 1275 bool IsForDefinition) { 1276 // Emit at least a definition for the aliasee if the the address of the 1277 // original function is requested. 1278 if (IsForDefinition || OrigAddr) 1279 (void)CGM.GetAddrOfGlobal(NewGD); 1280 StringRef NewMangledName = CGM.getMangledName(NewGD); 1281 llvm::GlobalValue *Addr = CGM.GetGlobalValue(NewMangledName); 1282 if (Addr && !Addr->isDeclaration()) { 1283 const auto *D = cast<FunctionDecl>(OldGD.getDecl()); 1284 const CGFunctionInfo &FI = CGM.getTypes().arrangeGlobalDeclaration(NewGD); 1285 llvm::Type *DeclTy = CGM.getTypes().GetFunctionType(FI); 1286 1287 // Create a reference to the named value. This ensures that it is emitted 1288 // if a deferred decl. 1289 llvm::GlobalValue::LinkageTypes LT = CGM.getFunctionLinkage(OldGD); 1290 1291 // Create the new alias itself, but don't set a name yet. 1292 auto *GA = 1293 llvm::GlobalAlias::create(DeclTy, 0, LT, "", Addr, &CGM.getModule()); 1294 1295 if (OrigAddr) { 1296 assert(OrigAddr->isDeclaration() && "Expected declaration"); 1297 1298 GA->takeName(OrigAddr); 1299 OrigAddr->replaceAllUsesWith( 1300 llvm::ConstantExpr::getBitCast(GA, OrigAddr->getType())); 1301 OrigAddr->eraseFromParent(); 1302 } else { 1303 GA->setName(CGM.getMangledName(OldGD)); 1304 } 1305 1306 // Set attributes which are particular to an alias; this is a 1307 // specialization of the attributes which may be set on a global function. 1308 if (D->hasAttr<WeakAttr>() || D->hasAttr<WeakRefAttr>() || 1309 D->isWeakImported()) 1310 GA->setLinkage(llvm::Function::WeakAnyLinkage); 1311 1312 CGM.SetCommonAttributes(OldGD, GA); 1313 return true; 1314 } 1315 return false; 1316 } 1317 1318 void CGOpenMPRuntime::clear() { 1319 InternalVars.clear(); 1320 // Clean non-target variable declarations possibly used only in debug info. 1321 for (const auto &Data : EmittedNonTargetVariables) { 1322 if (!Data.getValue().pointsToAliveValue()) 1323 continue; 1324 auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue()); 1325 if (!GV) 1326 continue; 1327 if (!GV->isDeclaration() || GV->getNumUses() > 0) 1328 continue; 1329 GV->eraseFromParent(); 1330 } 1331 // Emit aliases for the deferred aliasees. 1332 for (const auto &Pair : DeferredVariantFunction) { 1333 StringRef MangledName = CGM.getMangledName(Pair.second.second); 1334 llvm::GlobalValue *Addr = CGM.GetGlobalValue(MangledName); 1335 // If not able to emit alias, just emit original declaration. 1336 (void)tryEmitDeclareVariant(Pair.second.first, Pair.second.second, Addr, 1337 /*IsForDefinition=*/false); 1338 } 1339 } 1340 1341 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const { 1342 SmallString<128> Buffer; 1343 llvm::raw_svector_ostream OS(Buffer); 1344 StringRef Sep = FirstSeparator; 1345 for (StringRef Part : Parts) { 1346 OS << Sep << Part; 1347 Sep = Separator; 1348 } 1349 return std::string(OS.str()); 1350 } 1351 1352 static llvm::Function * 1353 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty, 1354 const Expr *CombinerInitializer, const VarDecl *In, 1355 const VarDecl *Out, bool IsCombiner) { 1356 // void .omp_combiner.(Ty *in, Ty *out); 1357 ASTContext &C = CGM.getContext(); 1358 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 1359 FunctionArgList Args; 1360 ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(), 1361 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1362 ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(), 1363 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1364 Args.push_back(&OmpOutParm); 1365 Args.push_back(&OmpInParm); 1366 const CGFunctionInfo &FnInfo = 1367 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 1368 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 1369 std::string Name = CGM.getOpenMPRuntime().getName( 1370 {IsCombiner ? "omp_combiner" : "omp_initializer", ""}); 1371 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 1372 Name, &CGM.getModule()); 1373 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 1374 if (CGM.getLangOpts().Optimize) { 1375 Fn->removeFnAttr(llvm::Attribute::NoInline); 1376 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 1377 Fn->addFnAttr(llvm::Attribute::AlwaysInline); 1378 } 1379 CodeGenFunction CGF(CGM); 1380 // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions. 1381 // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions. 1382 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(), 1383 Out->getLocation()); 1384 CodeGenFunction::OMPPrivateScope Scope(CGF); 1385 Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm); 1386 Scope.addPrivate(In, [&CGF, AddrIn, PtrTy]() { 1387 return CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>()) 1388 .getAddress(CGF); 1389 }); 1390 Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm); 1391 Scope.addPrivate(Out, [&CGF, AddrOut, PtrTy]() { 1392 return CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>()) 1393 .getAddress(CGF); 1394 }); 1395 (void)Scope.Privatize(); 1396 if (!IsCombiner && Out->hasInit() && 1397 !CGF.isTrivialInitializer(Out->getInit())) { 1398 CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out), 1399 Out->getType().getQualifiers(), 1400 /*IsInitializer=*/true); 1401 } 1402 if (CombinerInitializer) 1403 CGF.EmitIgnoredExpr(CombinerInitializer); 1404 Scope.ForceCleanup(); 1405 CGF.FinishFunction(); 1406 return Fn; 1407 } 1408 1409 void CGOpenMPRuntime::emitUserDefinedReduction( 1410 CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) { 1411 if (UDRMap.count(D) > 0) 1412 return; 1413 llvm::Function *Combiner = emitCombinerOrInitializer( 1414 CGM, D->getType(), D->getCombiner(), 1415 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()), 1416 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()), 1417 /*IsCombiner=*/true); 1418 llvm::Function *Initializer = nullptr; 1419 if (const Expr *Init = D->getInitializer()) { 1420 Initializer = emitCombinerOrInitializer( 1421 CGM, D->getType(), 1422 D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init 1423 : nullptr, 1424 cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()), 1425 cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()), 1426 /*IsCombiner=*/false); 1427 } 1428 UDRMap.try_emplace(D, Combiner, Initializer); 1429 if (CGF) { 1430 auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn); 1431 Decls.second.push_back(D); 1432 } 1433 } 1434 1435 std::pair<llvm::Function *, llvm::Function *> 1436 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) { 1437 auto I = UDRMap.find(D); 1438 if (I != UDRMap.end()) 1439 return I->second; 1440 emitUserDefinedReduction(/*CGF=*/nullptr, D); 1441 return UDRMap.lookup(D); 1442 } 1443 1444 namespace { 1445 // Temporary RAII solution to perform a push/pop stack event on the OpenMP IR 1446 // Builder if one is present. 1447 struct PushAndPopStackRAII { 1448 PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF, 1449 bool HasCancel) 1450 : OMPBuilder(OMPBuilder) { 1451 if (!OMPBuilder) 1452 return; 1453 1454 // The following callback is the crucial part of clangs cleanup process. 1455 // 1456 // NOTE: 1457 // Once the OpenMPIRBuilder is used to create parallel regions (and 1458 // similar), the cancellation destination (Dest below) is determined via 1459 // IP. That means if we have variables to finalize we split the block at IP, 1460 // use the new block (=BB) as destination to build a JumpDest (via 1461 // getJumpDestInCurrentScope(BB)) which then is fed to 1462 // EmitBranchThroughCleanup. Furthermore, there will not be the need 1463 // to push & pop an FinalizationInfo object. 1464 // The FiniCB will still be needed but at the point where the 1465 // OpenMPIRBuilder is asked to construct a parallel (or similar) construct. 1466 auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) { 1467 assert(IP.getBlock()->end() == IP.getPoint() && 1468 "Clang CG should cause non-terminated block!"); 1469 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1470 CGF.Builder.restoreIP(IP); 1471 CodeGenFunction::JumpDest Dest = 1472 CGF.getOMPCancelDestination(OMPD_parallel); 1473 CGF.EmitBranchThroughCleanup(Dest); 1474 }; 1475 1476 // TODO: Remove this once we emit parallel regions through the 1477 // OpenMPIRBuilder as it can do this setup internally. 1478 llvm::OpenMPIRBuilder::FinalizationInfo FI( 1479 {FiniCB, OMPD_parallel, HasCancel}); 1480 OMPBuilder->pushFinalizationCB(std::move(FI)); 1481 } 1482 ~PushAndPopStackRAII() { 1483 if (OMPBuilder) 1484 OMPBuilder->popFinalizationCB(); 1485 } 1486 llvm::OpenMPIRBuilder *OMPBuilder; 1487 }; 1488 } // namespace 1489 1490 static llvm::Function *emitParallelOrTeamsOutlinedFunction( 1491 CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS, 1492 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, 1493 const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) { 1494 assert(ThreadIDVar->getType()->isPointerType() && 1495 "thread id variable must be of type kmp_int32 *"); 1496 CodeGenFunction CGF(CGM, true); 1497 bool HasCancel = false; 1498 if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D)) 1499 HasCancel = OPD->hasCancel(); 1500 else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D)) 1501 HasCancel = OPSD->hasCancel(); 1502 else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D)) 1503 HasCancel = OPFD->hasCancel(); 1504 else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D)) 1505 HasCancel = OPFD->hasCancel(); 1506 else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D)) 1507 HasCancel = OPFD->hasCancel(); 1508 else if (const auto *OPFD = 1509 dyn_cast<OMPTeamsDistributeParallelForDirective>(&D)) 1510 HasCancel = OPFD->hasCancel(); 1511 else if (const auto *OPFD = 1512 dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D)) 1513 HasCancel = OPFD->hasCancel(); 1514 1515 // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new 1516 // parallel region to make cancellation barriers work properly. 1517 llvm::OpenMPIRBuilder *OMPBuilder = CGM.getOpenMPIRBuilder(); 1518 PushAndPopStackRAII PSR(OMPBuilder, CGF, HasCancel); 1519 CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind, 1520 HasCancel, OutlinedHelperName); 1521 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1522 return CGF.GenerateOpenMPCapturedStmtFunction(*CS, D.getBeginLoc()); 1523 } 1524 1525 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction( 1526 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1527 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1528 const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel); 1529 return emitParallelOrTeamsOutlinedFunction( 1530 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1531 } 1532 1533 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction( 1534 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1535 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1536 const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams); 1537 return emitParallelOrTeamsOutlinedFunction( 1538 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1539 } 1540 1541 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction( 1542 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1543 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 1544 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 1545 bool Tied, unsigned &NumberOfParts) { 1546 auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF, 1547 PrePostActionTy &) { 1548 llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc()); 1549 llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 1550 llvm::Value *TaskArgs[] = { 1551 UpLoc, ThreadID, 1552 CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar), 1553 TaskTVar->getType()->castAs<PointerType>()) 1554 .getPointer(CGF)}; 1555 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs); 1556 }; 1557 CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar, 1558 UntiedCodeGen); 1559 CodeGen.setAction(Action); 1560 assert(!ThreadIDVar->getType()->isPointerType() && 1561 "thread id variable must be of type kmp_int32 for tasks"); 1562 const OpenMPDirectiveKind Region = 1563 isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop 1564 : OMPD_task; 1565 const CapturedStmt *CS = D.getCapturedStmt(Region); 1566 const auto *TD = dyn_cast<OMPTaskDirective>(&D); 1567 CodeGenFunction CGF(CGM, true); 1568 CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, 1569 InnermostKind, 1570 TD ? TD->hasCancel() : false, Action); 1571 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1572 llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS); 1573 if (!Tied) 1574 NumberOfParts = Action.getNumberOfParts(); 1575 return Res; 1576 } 1577 1578 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM, 1579 const RecordDecl *RD, const CGRecordLayout &RL, 1580 ArrayRef<llvm::Constant *> Data) { 1581 llvm::StructType *StructTy = RL.getLLVMType(); 1582 unsigned PrevIdx = 0; 1583 ConstantInitBuilder CIBuilder(CGM); 1584 auto DI = Data.begin(); 1585 for (const FieldDecl *FD : RD->fields()) { 1586 unsigned Idx = RL.getLLVMFieldNo(FD); 1587 // Fill the alignment. 1588 for (unsigned I = PrevIdx; I < Idx; ++I) 1589 Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I))); 1590 PrevIdx = Idx + 1; 1591 Fields.add(*DI); 1592 ++DI; 1593 } 1594 } 1595 1596 template <class... As> 1597 static llvm::GlobalVariable * 1598 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant, 1599 ArrayRef<llvm::Constant *> Data, const Twine &Name, 1600 As &&... Args) { 1601 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1602 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1603 ConstantInitBuilder CIBuilder(CGM); 1604 ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType()); 1605 buildStructValue(Fields, CGM, RD, RL, Data); 1606 return Fields.finishAndCreateGlobal( 1607 Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant, 1608 std::forward<As>(Args)...); 1609 } 1610 1611 template <typename T> 1612 static void 1613 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty, 1614 ArrayRef<llvm::Constant *> Data, 1615 T &Parent) { 1616 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1617 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1618 ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType()); 1619 buildStructValue(Fields, CGM, RD, RL, Data); 1620 Fields.finishAndAddTo(Parent); 1621 } 1622 1623 Address CGOpenMPRuntime::getOrCreateDefaultLocation(unsigned Flags) { 1624 CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy); 1625 unsigned Reserved2Flags = getDefaultLocationReserved2Flags(); 1626 FlagsTy FlagsKey(Flags, Reserved2Flags); 1627 llvm::Value *Entry = OpenMPDefaultLocMap.lookup(FlagsKey); 1628 if (!Entry) { 1629 if (!DefaultOpenMPPSource) { 1630 // Initialize default location for psource field of ident_t structure of 1631 // all ident_t objects. Format is ";file;function;line;column;;". 1632 // Taken from 1633 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp_str.cpp 1634 DefaultOpenMPPSource = 1635 CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;").getPointer(); 1636 DefaultOpenMPPSource = 1637 llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy); 1638 } 1639 1640 llvm::Constant *Data[] = { 1641 llvm::ConstantInt::getNullValue(CGM.Int32Ty), 1642 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 1643 llvm::ConstantInt::get(CGM.Int32Ty, Reserved2Flags), 1644 llvm::ConstantInt::getNullValue(CGM.Int32Ty), DefaultOpenMPPSource}; 1645 llvm::GlobalValue *DefaultOpenMPLocation = 1646 createGlobalStruct(CGM, IdentQTy, isDefaultLocationConstant(), Data, "", 1647 llvm::GlobalValue::PrivateLinkage); 1648 DefaultOpenMPLocation->setUnnamedAddr( 1649 llvm::GlobalValue::UnnamedAddr::Global); 1650 1651 OpenMPDefaultLocMap[FlagsKey] = Entry = DefaultOpenMPLocation; 1652 } 1653 return Address(Entry, Align); 1654 } 1655 1656 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF, 1657 bool AtCurrentPoint) { 1658 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1659 assert(!Elem.second.ServiceInsertPt && "Insert point is set already."); 1660 1661 llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty); 1662 if (AtCurrentPoint) { 1663 Elem.second.ServiceInsertPt = new llvm::BitCastInst( 1664 Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock()); 1665 } else { 1666 Elem.second.ServiceInsertPt = 1667 new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt"); 1668 Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt); 1669 } 1670 } 1671 1672 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) { 1673 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1674 if (Elem.second.ServiceInsertPt) { 1675 llvm::Instruction *Ptr = Elem.second.ServiceInsertPt; 1676 Elem.second.ServiceInsertPt = nullptr; 1677 Ptr->eraseFromParent(); 1678 } 1679 } 1680 1681 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF, 1682 SourceLocation Loc, 1683 unsigned Flags) { 1684 Flags |= OMP_IDENT_KMPC; 1685 // If no debug info is generated - return global default location. 1686 if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo || 1687 Loc.isInvalid()) 1688 return getOrCreateDefaultLocation(Flags).getPointer(); 1689 1690 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1691 1692 CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy); 1693 Address LocValue = Address::invalid(); 1694 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1695 if (I != OpenMPLocThreadIDMap.end()) 1696 LocValue = Address(I->second.DebugLoc, Align); 1697 1698 // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if 1699 // GetOpenMPThreadID was called before this routine. 1700 if (!LocValue.isValid()) { 1701 // Generate "ident_t .kmpc_loc.addr;" 1702 Address AI = CGF.CreateMemTemp(IdentQTy, ".kmpc_loc.addr"); 1703 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1704 Elem.second.DebugLoc = AI.getPointer(); 1705 LocValue = AI; 1706 1707 if (!Elem.second.ServiceInsertPt) 1708 setLocThreadIdInsertPt(CGF); 1709 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1710 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1711 CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags), 1712 CGF.getTypeSize(IdentQTy)); 1713 } 1714 1715 // char **psource = &.kmpc_loc_<flags>.addr.psource; 1716 LValue Base = CGF.MakeAddrLValue(LocValue, IdentQTy); 1717 auto Fields = cast<RecordDecl>(IdentQTy->getAsTagDecl())->field_begin(); 1718 LValue PSource = 1719 CGF.EmitLValueForField(Base, *std::next(Fields, IdentField_PSource)); 1720 1721 llvm::Value *OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding()); 1722 if (OMPDebugLoc == nullptr) { 1723 SmallString<128> Buffer2; 1724 llvm::raw_svector_ostream OS2(Buffer2); 1725 // Build debug location 1726 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc); 1727 OS2 << ";" << PLoc.getFilename() << ";"; 1728 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) 1729 OS2 << FD->getQualifiedNameAsString(); 1730 OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;"; 1731 OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str()); 1732 OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc; 1733 } 1734 // *psource = ";<File>;<Function>;<Line>;<Column>;;"; 1735 CGF.EmitStoreOfScalar(OMPDebugLoc, PSource); 1736 1737 // Our callers always pass this to a runtime function, so for 1738 // convenience, go ahead and return a naked pointer. 1739 return LocValue.getPointer(); 1740 } 1741 1742 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF, 1743 SourceLocation Loc) { 1744 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1745 1746 llvm::Value *ThreadID = nullptr; 1747 // Check whether we've already cached a load of the thread id in this 1748 // function. 1749 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1750 if (I != OpenMPLocThreadIDMap.end()) { 1751 ThreadID = I->second.ThreadID; 1752 if (ThreadID != nullptr) 1753 return ThreadID; 1754 } 1755 // If exceptions are enabled, do not use parameter to avoid possible crash. 1756 if (auto *OMPRegionInfo = 1757 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 1758 if (OMPRegionInfo->getThreadIDVariable()) { 1759 // Check if this an outlined function with thread id passed as argument. 1760 LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF); 1761 llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent(); 1762 if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions || 1763 !CGF.getLangOpts().CXXExceptions || 1764 CGF.Builder.GetInsertBlock() == TopBlock || 1765 !isa<llvm::Instruction>(LVal.getPointer(CGF)) || 1766 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() == 1767 TopBlock || 1768 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() == 1769 CGF.Builder.GetInsertBlock()) { 1770 ThreadID = CGF.EmitLoadOfScalar(LVal, Loc); 1771 // If value loaded in entry block, cache it and use it everywhere in 1772 // function. 1773 if (CGF.Builder.GetInsertBlock() == TopBlock) { 1774 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1775 Elem.second.ThreadID = ThreadID; 1776 } 1777 return ThreadID; 1778 } 1779 } 1780 } 1781 1782 // This is not an outlined function region - need to call __kmpc_int32 1783 // kmpc_global_thread_num(ident_t *loc). 1784 // Generate thread id value and cache this value for use across the 1785 // function. 1786 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1787 if (!Elem.second.ServiceInsertPt) 1788 setLocThreadIdInsertPt(CGF); 1789 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1790 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1791 llvm::CallInst *Call = CGF.Builder.CreateCall( 1792 createRuntimeFunction(OMPRTL__kmpc_global_thread_num), 1793 emitUpdateLocation(CGF, Loc)); 1794 Call->setCallingConv(CGF.getRuntimeCC()); 1795 Elem.second.ThreadID = Call; 1796 return Call; 1797 } 1798 1799 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) { 1800 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1801 if (OpenMPLocThreadIDMap.count(CGF.CurFn)) { 1802 clearLocThreadIdInsertPt(CGF); 1803 OpenMPLocThreadIDMap.erase(CGF.CurFn); 1804 } 1805 if (FunctionUDRMap.count(CGF.CurFn) > 0) { 1806 for(const auto *D : FunctionUDRMap[CGF.CurFn]) 1807 UDRMap.erase(D); 1808 FunctionUDRMap.erase(CGF.CurFn); 1809 } 1810 auto I = FunctionUDMMap.find(CGF.CurFn); 1811 if (I != FunctionUDMMap.end()) { 1812 for(const auto *D : I->second) 1813 UDMMap.erase(D); 1814 FunctionUDMMap.erase(I); 1815 } 1816 LastprivateConditionalToTypes.erase(CGF.CurFn); 1817 } 1818 1819 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() { 1820 return IdentTy->getPointerTo(); 1821 } 1822 1823 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() { 1824 if (!Kmpc_MicroTy) { 1825 // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...) 1826 llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty), 1827 llvm::PointerType::getUnqual(CGM.Int32Ty)}; 1828 Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true); 1829 } 1830 return llvm::PointerType::getUnqual(Kmpc_MicroTy); 1831 } 1832 1833 llvm::FunctionCallee CGOpenMPRuntime::createRuntimeFunction(unsigned Function) { 1834 llvm::FunctionCallee RTLFn = nullptr; 1835 switch (static_cast<OpenMPRTLFunction>(Function)) { 1836 case OMPRTL__kmpc_fork_call: { 1837 // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro 1838 // microtask, ...); 1839 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1840 getKmpc_MicroPointerTy()}; 1841 auto *FnTy = 1842 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true); 1843 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call"); 1844 if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) { 1845 if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) { 1846 llvm::LLVMContext &Ctx = F->getContext(); 1847 llvm::MDBuilder MDB(Ctx); 1848 // Annotate the callback behavior of the __kmpc_fork_call: 1849 // - The callback callee is argument number 2 (microtask). 1850 // - The first two arguments of the callback callee are unknown (-1). 1851 // - All variadic arguments to the __kmpc_fork_call are passed to the 1852 // callback callee. 1853 F->addMetadata( 1854 llvm::LLVMContext::MD_callback, 1855 *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding( 1856 2, {-1, -1}, 1857 /* VarArgsArePassed */ true)})); 1858 } 1859 } 1860 break; 1861 } 1862 case OMPRTL__kmpc_global_thread_num: { 1863 // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc); 1864 llvm::Type *TypeParams[] = {getIdentTyPointerTy()}; 1865 auto *FnTy = 1866 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1867 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num"); 1868 break; 1869 } 1870 case OMPRTL__kmpc_threadprivate_cached: { 1871 // Build void *__kmpc_threadprivate_cached(ident_t *loc, 1872 // kmp_int32 global_tid, void *data, size_t size, void ***cache); 1873 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1874 CGM.VoidPtrTy, CGM.SizeTy, 1875 CGM.VoidPtrTy->getPointerTo()->getPointerTo()}; 1876 auto *FnTy = 1877 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false); 1878 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached"); 1879 break; 1880 } 1881 case OMPRTL__kmpc_critical: { 1882 // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid, 1883 // kmp_critical_name *crit); 1884 llvm::Type *TypeParams[] = { 1885 getIdentTyPointerTy(), CGM.Int32Ty, 1886 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1887 auto *FnTy = 1888 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1889 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical"); 1890 break; 1891 } 1892 case OMPRTL__kmpc_critical_with_hint: { 1893 // Build void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 global_tid, 1894 // kmp_critical_name *crit, uintptr_t hint); 1895 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1896 llvm::PointerType::getUnqual(KmpCriticalNameTy), 1897 CGM.IntPtrTy}; 1898 auto *FnTy = 1899 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1900 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical_with_hint"); 1901 break; 1902 } 1903 case OMPRTL__kmpc_threadprivate_register: { 1904 // Build void __kmpc_threadprivate_register(ident_t *, void *data, 1905 // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor); 1906 // typedef void *(*kmpc_ctor)(void *); 1907 auto *KmpcCtorTy = 1908 llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 1909 /*isVarArg*/ false)->getPointerTo(); 1910 // typedef void *(*kmpc_cctor)(void *, void *); 1911 llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 1912 auto *KmpcCopyCtorTy = 1913 llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs, 1914 /*isVarArg*/ false) 1915 ->getPointerTo(); 1916 // typedef void (*kmpc_dtor)(void *); 1917 auto *KmpcDtorTy = 1918 llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false) 1919 ->getPointerTo(); 1920 llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy, 1921 KmpcCopyCtorTy, KmpcDtorTy}; 1922 auto *FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs, 1923 /*isVarArg*/ false); 1924 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register"); 1925 break; 1926 } 1927 case OMPRTL__kmpc_end_critical: { 1928 // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid, 1929 // kmp_critical_name *crit); 1930 llvm::Type *TypeParams[] = { 1931 getIdentTyPointerTy(), CGM.Int32Ty, 1932 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1933 auto *FnTy = 1934 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1935 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical"); 1936 break; 1937 } 1938 case OMPRTL__kmpc_cancel_barrier: { 1939 // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32 1940 // global_tid); 1941 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1942 auto *FnTy = 1943 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1944 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier"); 1945 break; 1946 } 1947 case OMPRTL__kmpc_barrier: { 1948 // Build void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid); 1949 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1950 auto *FnTy = 1951 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1952 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_barrier"); 1953 break; 1954 } 1955 case OMPRTL__kmpc_for_static_fini: { 1956 // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid); 1957 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1958 auto *FnTy = 1959 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1960 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini"); 1961 break; 1962 } 1963 case OMPRTL__kmpc_push_num_threads: { 1964 // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid, 1965 // kmp_int32 num_threads) 1966 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1967 CGM.Int32Ty}; 1968 auto *FnTy = 1969 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1970 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads"); 1971 break; 1972 } 1973 case OMPRTL__kmpc_serialized_parallel: { 1974 // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32 1975 // global_tid); 1976 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1977 auto *FnTy = 1978 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1979 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel"); 1980 break; 1981 } 1982 case OMPRTL__kmpc_end_serialized_parallel: { 1983 // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32 1984 // global_tid); 1985 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1986 auto *FnTy = 1987 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1988 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel"); 1989 break; 1990 } 1991 case OMPRTL__kmpc_flush: { 1992 // Build void __kmpc_flush(ident_t *loc); 1993 llvm::Type *TypeParams[] = {getIdentTyPointerTy()}; 1994 auto *FnTy = 1995 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1996 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush"); 1997 break; 1998 } 1999 case OMPRTL__kmpc_master: { 2000 // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid); 2001 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2002 auto *FnTy = 2003 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2004 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master"); 2005 break; 2006 } 2007 case OMPRTL__kmpc_end_master: { 2008 // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid); 2009 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2010 auto *FnTy = 2011 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2012 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master"); 2013 break; 2014 } 2015 case OMPRTL__kmpc_omp_taskyield: { 2016 // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid, 2017 // int end_part); 2018 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2019 auto *FnTy = 2020 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2021 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield"); 2022 break; 2023 } 2024 case OMPRTL__kmpc_single: { 2025 // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid); 2026 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2027 auto *FnTy = 2028 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2029 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single"); 2030 break; 2031 } 2032 case OMPRTL__kmpc_end_single: { 2033 // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid); 2034 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2035 auto *FnTy = 2036 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2037 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single"); 2038 break; 2039 } 2040 case OMPRTL__kmpc_omp_task_alloc: { 2041 // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 2042 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 2043 // kmp_routine_entry_t *task_entry); 2044 assert(KmpRoutineEntryPtrTy != nullptr && 2045 "Type kmp_routine_entry_t must be created."); 2046 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2047 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy}; 2048 // Return void * and then cast to particular kmp_task_t type. 2049 auto *FnTy = 2050 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2051 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc"); 2052 break; 2053 } 2054 case OMPRTL__kmpc_omp_target_task_alloc: { 2055 // Build kmp_task_t *__kmpc_omp_target_task_alloc(ident_t *, kmp_int32 gtid, 2056 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 2057 // kmp_routine_entry_t *task_entry, kmp_int64 device_id); 2058 assert(KmpRoutineEntryPtrTy != nullptr && 2059 "Type kmp_routine_entry_t must be created."); 2060 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2061 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy, 2062 CGM.Int64Ty}; 2063 // Return void * and then cast to particular kmp_task_t type. 2064 auto *FnTy = 2065 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2066 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_target_task_alloc"); 2067 break; 2068 } 2069 case OMPRTL__kmpc_omp_task: { 2070 // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2071 // *new_task); 2072 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2073 CGM.VoidPtrTy}; 2074 auto *FnTy = 2075 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2076 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task"); 2077 break; 2078 } 2079 case OMPRTL__kmpc_copyprivate: { 2080 // Build void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid, 2081 // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *), 2082 // kmp_int32 didit); 2083 llvm::Type *CpyTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2084 auto *CpyFnTy = 2085 llvm::FunctionType::get(CGM.VoidTy, CpyTypeParams, /*isVarArg=*/false); 2086 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.SizeTy, 2087 CGM.VoidPtrTy, CpyFnTy->getPointerTo(), 2088 CGM.Int32Ty}; 2089 auto *FnTy = 2090 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2091 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_copyprivate"); 2092 break; 2093 } 2094 case OMPRTL__kmpc_reduce: { 2095 // Build kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid, 2096 // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void 2097 // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck); 2098 llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2099 auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams, 2100 /*isVarArg=*/false); 2101 llvm::Type *TypeParams[] = { 2102 getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy, 2103 CGM.VoidPtrTy, ReduceFnTy->getPointerTo(), 2104 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2105 auto *FnTy = 2106 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2107 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce"); 2108 break; 2109 } 2110 case OMPRTL__kmpc_reduce_nowait: { 2111 // Build kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32 2112 // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, 2113 // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name 2114 // *lck); 2115 llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2116 auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams, 2117 /*isVarArg=*/false); 2118 llvm::Type *TypeParams[] = { 2119 getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy, 2120 CGM.VoidPtrTy, ReduceFnTy->getPointerTo(), 2121 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2122 auto *FnTy = 2123 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2124 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce_nowait"); 2125 break; 2126 } 2127 case OMPRTL__kmpc_end_reduce: { 2128 // Build void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid, 2129 // kmp_critical_name *lck); 2130 llvm::Type *TypeParams[] = { 2131 getIdentTyPointerTy(), CGM.Int32Ty, 2132 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2133 auto *FnTy = 2134 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2135 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce"); 2136 break; 2137 } 2138 case OMPRTL__kmpc_end_reduce_nowait: { 2139 // Build __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid, 2140 // kmp_critical_name *lck); 2141 llvm::Type *TypeParams[] = { 2142 getIdentTyPointerTy(), CGM.Int32Ty, 2143 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2144 auto *FnTy = 2145 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2146 RTLFn = 2147 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce_nowait"); 2148 break; 2149 } 2150 case OMPRTL__kmpc_omp_task_begin_if0: { 2151 // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2152 // *new_task); 2153 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2154 CGM.VoidPtrTy}; 2155 auto *FnTy = 2156 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2157 RTLFn = 2158 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_begin_if0"); 2159 break; 2160 } 2161 case OMPRTL__kmpc_omp_task_complete_if0: { 2162 // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2163 // *new_task); 2164 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2165 CGM.VoidPtrTy}; 2166 auto *FnTy = 2167 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2168 RTLFn = CGM.CreateRuntimeFunction(FnTy, 2169 /*Name=*/"__kmpc_omp_task_complete_if0"); 2170 break; 2171 } 2172 case OMPRTL__kmpc_ordered: { 2173 // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid); 2174 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2175 auto *FnTy = 2176 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2177 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_ordered"); 2178 break; 2179 } 2180 case OMPRTL__kmpc_end_ordered: { 2181 // Build void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid); 2182 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2183 auto *FnTy = 2184 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2185 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_ordered"); 2186 break; 2187 } 2188 case OMPRTL__kmpc_omp_taskwait: { 2189 // Build kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 global_tid); 2190 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2191 auto *FnTy = 2192 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2193 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_omp_taskwait"); 2194 break; 2195 } 2196 case OMPRTL__kmpc_taskgroup: { 2197 // Build void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid); 2198 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2199 auto *FnTy = 2200 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2201 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_taskgroup"); 2202 break; 2203 } 2204 case OMPRTL__kmpc_end_taskgroup: { 2205 // Build void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid); 2206 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2207 auto *FnTy = 2208 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2209 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_taskgroup"); 2210 break; 2211 } 2212 case OMPRTL__kmpc_push_proc_bind: { 2213 // Build void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid, 2214 // int proc_bind) 2215 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2216 auto *FnTy = 2217 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2218 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_proc_bind"); 2219 break; 2220 } 2221 case OMPRTL__kmpc_omp_task_with_deps: { 2222 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 2223 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 2224 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 2225 llvm::Type *TypeParams[] = { 2226 getIdentTyPointerTy(), CGM.Int32Ty, CGM.VoidPtrTy, CGM.Int32Ty, 2227 CGM.VoidPtrTy, CGM.Int32Ty, CGM.VoidPtrTy}; 2228 auto *FnTy = 2229 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2230 RTLFn = 2231 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_with_deps"); 2232 break; 2233 } 2234 case OMPRTL__kmpc_omp_wait_deps: { 2235 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 2236 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 ndeps_noalias, 2237 // kmp_depend_info_t *noalias_dep_list); 2238 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2239 CGM.Int32Ty, CGM.VoidPtrTy, 2240 CGM.Int32Ty, CGM.VoidPtrTy}; 2241 auto *FnTy = 2242 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2243 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_wait_deps"); 2244 break; 2245 } 2246 case OMPRTL__kmpc_cancellationpoint: { 2247 // Build kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 2248 // global_tid, kmp_int32 cncl_kind) 2249 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2250 auto *FnTy = 2251 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2252 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancellationpoint"); 2253 break; 2254 } 2255 case OMPRTL__kmpc_cancel: { 2256 // Build kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 2257 // kmp_int32 cncl_kind) 2258 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2259 auto *FnTy = 2260 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2261 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancel"); 2262 break; 2263 } 2264 case OMPRTL__kmpc_push_num_teams: { 2265 // Build void kmpc_push_num_teams (ident_t loc, kmp_int32 global_tid, 2266 // kmp_int32 num_teams, kmp_int32 num_threads) 2267 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2268 CGM.Int32Ty}; 2269 auto *FnTy = 2270 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2271 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_teams"); 2272 break; 2273 } 2274 case OMPRTL__kmpc_fork_teams: { 2275 // Build void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro 2276 // microtask, ...); 2277 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2278 getKmpc_MicroPointerTy()}; 2279 auto *FnTy = 2280 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true); 2281 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_teams"); 2282 if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) { 2283 if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) { 2284 llvm::LLVMContext &Ctx = F->getContext(); 2285 llvm::MDBuilder MDB(Ctx); 2286 // Annotate the callback behavior of the __kmpc_fork_teams: 2287 // - The callback callee is argument number 2 (microtask). 2288 // - The first two arguments of the callback callee are unknown (-1). 2289 // - All variadic arguments to the __kmpc_fork_teams are passed to the 2290 // callback callee. 2291 F->addMetadata( 2292 llvm::LLVMContext::MD_callback, 2293 *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding( 2294 2, {-1, -1}, 2295 /* VarArgsArePassed */ true)})); 2296 } 2297 } 2298 break; 2299 } 2300 case OMPRTL__kmpc_taskloop: { 2301 // Build void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 2302 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 2303 // sched, kmp_uint64 grainsize, void *task_dup); 2304 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), 2305 CGM.IntTy, 2306 CGM.VoidPtrTy, 2307 CGM.IntTy, 2308 CGM.Int64Ty->getPointerTo(), 2309 CGM.Int64Ty->getPointerTo(), 2310 CGM.Int64Ty, 2311 CGM.IntTy, 2312 CGM.IntTy, 2313 CGM.Int64Ty, 2314 CGM.VoidPtrTy}; 2315 auto *FnTy = 2316 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2317 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_taskloop"); 2318 break; 2319 } 2320 case OMPRTL__kmpc_doacross_init: { 2321 // Build void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32 2322 // num_dims, struct kmp_dim *dims); 2323 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), 2324 CGM.Int32Ty, 2325 CGM.Int32Ty, 2326 CGM.VoidPtrTy}; 2327 auto *FnTy = 2328 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2329 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_init"); 2330 break; 2331 } 2332 case OMPRTL__kmpc_doacross_fini: { 2333 // Build void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid); 2334 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2335 auto *FnTy = 2336 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2337 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_fini"); 2338 break; 2339 } 2340 case OMPRTL__kmpc_doacross_post: { 2341 // Build void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64 2342 // *vec); 2343 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2344 CGM.Int64Ty->getPointerTo()}; 2345 auto *FnTy = 2346 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2347 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_post"); 2348 break; 2349 } 2350 case OMPRTL__kmpc_doacross_wait: { 2351 // Build void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64 2352 // *vec); 2353 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2354 CGM.Int64Ty->getPointerTo()}; 2355 auto *FnTy = 2356 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2357 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_wait"); 2358 break; 2359 } 2360 case OMPRTL__kmpc_task_reduction_init: { 2361 // Build void *__kmpc_task_reduction_init(int gtid, int num_data, void 2362 // *data); 2363 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.IntTy, CGM.VoidPtrTy}; 2364 auto *FnTy = 2365 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2366 RTLFn = 2367 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_task_reduction_init"); 2368 break; 2369 } 2370 case OMPRTL__kmpc_task_reduction_get_th_data: { 2371 // Build void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 2372 // *d); 2373 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy}; 2374 auto *FnTy = 2375 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2376 RTLFn = CGM.CreateRuntimeFunction( 2377 FnTy, /*Name=*/"__kmpc_task_reduction_get_th_data"); 2378 break; 2379 } 2380 case OMPRTL__kmpc_alloc: { 2381 // Build to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t 2382 // al); omp_allocator_handle_t type is void *. 2383 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.SizeTy, CGM.VoidPtrTy}; 2384 auto *FnTy = 2385 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2386 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_alloc"); 2387 break; 2388 } 2389 case OMPRTL__kmpc_free: { 2390 // Build to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t 2391 // al); omp_allocator_handle_t type is void *. 2392 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy}; 2393 auto *FnTy = 2394 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2395 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_free"); 2396 break; 2397 } 2398 case OMPRTL__kmpc_push_target_tripcount: { 2399 // Build void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64 2400 // size); 2401 llvm::Type *TypeParams[] = {CGM.Int64Ty, CGM.Int64Ty}; 2402 llvm::FunctionType *FnTy = 2403 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2404 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_target_tripcount"); 2405 break; 2406 } 2407 case OMPRTL__tgt_target: { 2408 // Build int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t 2409 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2410 // *arg_types); 2411 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2412 CGM.VoidPtrTy, 2413 CGM.Int32Ty, 2414 CGM.VoidPtrPtrTy, 2415 CGM.VoidPtrPtrTy, 2416 CGM.Int64Ty->getPointerTo(), 2417 CGM.Int64Ty->getPointerTo()}; 2418 auto *FnTy = 2419 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2420 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target"); 2421 break; 2422 } 2423 case OMPRTL__tgt_target_nowait: { 2424 // Build int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr, 2425 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, 2426 // int64_t *arg_types); 2427 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2428 CGM.VoidPtrTy, 2429 CGM.Int32Ty, 2430 CGM.VoidPtrPtrTy, 2431 CGM.VoidPtrPtrTy, 2432 CGM.Int64Ty->getPointerTo(), 2433 CGM.Int64Ty->getPointerTo()}; 2434 auto *FnTy = 2435 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2436 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_nowait"); 2437 break; 2438 } 2439 case OMPRTL__tgt_target_teams: { 2440 // Build int32_t __tgt_target_teams(int64_t device_id, void *host_ptr, 2441 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, 2442 // int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 2443 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2444 CGM.VoidPtrTy, 2445 CGM.Int32Ty, 2446 CGM.VoidPtrPtrTy, 2447 CGM.VoidPtrPtrTy, 2448 CGM.Int64Ty->getPointerTo(), 2449 CGM.Int64Ty->getPointerTo(), 2450 CGM.Int32Ty, 2451 CGM.Int32Ty}; 2452 auto *FnTy = 2453 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2454 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams"); 2455 break; 2456 } 2457 case OMPRTL__tgt_target_teams_nowait: { 2458 // Build int32_t __tgt_target_teams_nowait(int64_t device_id, void 2459 // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t 2460 // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 2461 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2462 CGM.VoidPtrTy, 2463 CGM.Int32Ty, 2464 CGM.VoidPtrPtrTy, 2465 CGM.VoidPtrPtrTy, 2466 CGM.Int64Ty->getPointerTo(), 2467 CGM.Int64Ty->getPointerTo(), 2468 CGM.Int32Ty, 2469 CGM.Int32Ty}; 2470 auto *FnTy = 2471 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2472 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams_nowait"); 2473 break; 2474 } 2475 case OMPRTL__tgt_register_requires: { 2476 // Build void __tgt_register_requires(int64_t flags); 2477 llvm::Type *TypeParams[] = {CGM.Int64Ty}; 2478 auto *FnTy = 2479 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2480 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_requires"); 2481 break; 2482 } 2483 case OMPRTL__tgt_target_data_begin: { 2484 // Build void __tgt_target_data_begin(int64_t device_id, int32_t arg_num, 2485 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2486 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2487 CGM.Int32Ty, 2488 CGM.VoidPtrPtrTy, 2489 CGM.VoidPtrPtrTy, 2490 CGM.Int64Ty->getPointerTo(), 2491 CGM.Int64Ty->getPointerTo()}; 2492 auto *FnTy = 2493 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2494 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin"); 2495 break; 2496 } 2497 case OMPRTL__tgt_target_data_begin_nowait: { 2498 // Build void __tgt_target_data_begin_nowait(int64_t device_id, int32_t 2499 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2500 // *arg_types); 2501 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2502 CGM.Int32Ty, 2503 CGM.VoidPtrPtrTy, 2504 CGM.VoidPtrPtrTy, 2505 CGM.Int64Ty->getPointerTo(), 2506 CGM.Int64Ty->getPointerTo()}; 2507 auto *FnTy = 2508 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2509 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin_nowait"); 2510 break; 2511 } 2512 case OMPRTL__tgt_target_data_end: { 2513 // Build void __tgt_target_data_end(int64_t device_id, int32_t arg_num, 2514 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2515 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2516 CGM.Int32Ty, 2517 CGM.VoidPtrPtrTy, 2518 CGM.VoidPtrPtrTy, 2519 CGM.Int64Ty->getPointerTo(), 2520 CGM.Int64Ty->getPointerTo()}; 2521 auto *FnTy = 2522 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2523 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end"); 2524 break; 2525 } 2526 case OMPRTL__tgt_target_data_end_nowait: { 2527 // Build void __tgt_target_data_end_nowait(int64_t device_id, int32_t 2528 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2529 // *arg_types); 2530 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2531 CGM.Int32Ty, 2532 CGM.VoidPtrPtrTy, 2533 CGM.VoidPtrPtrTy, 2534 CGM.Int64Ty->getPointerTo(), 2535 CGM.Int64Ty->getPointerTo()}; 2536 auto *FnTy = 2537 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2538 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end_nowait"); 2539 break; 2540 } 2541 case OMPRTL__tgt_target_data_update: { 2542 // Build void __tgt_target_data_update(int64_t device_id, int32_t arg_num, 2543 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2544 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2545 CGM.Int32Ty, 2546 CGM.VoidPtrPtrTy, 2547 CGM.VoidPtrPtrTy, 2548 CGM.Int64Ty->getPointerTo(), 2549 CGM.Int64Ty->getPointerTo()}; 2550 auto *FnTy = 2551 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2552 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update"); 2553 break; 2554 } 2555 case OMPRTL__tgt_target_data_update_nowait: { 2556 // Build void __tgt_target_data_update_nowait(int64_t device_id, int32_t 2557 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2558 // *arg_types); 2559 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2560 CGM.Int32Ty, 2561 CGM.VoidPtrPtrTy, 2562 CGM.VoidPtrPtrTy, 2563 CGM.Int64Ty->getPointerTo(), 2564 CGM.Int64Ty->getPointerTo()}; 2565 auto *FnTy = 2566 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2567 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update_nowait"); 2568 break; 2569 } 2570 case OMPRTL__tgt_mapper_num_components: { 2571 // Build int64_t __tgt_mapper_num_components(void *rt_mapper_handle); 2572 llvm::Type *TypeParams[] = {CGM.VoidPtrTy}; 2573 auto *FnTy = 2574 llvm::FunctionType::get(CGM.Int64Ty, TypeParams, /*isVarArg*/ false); 2575 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_mapper_num_components"); 2576 break; 2577 } 2578 case OMPRTL__tgt_push_mapper_component: { 2579 // Build void __tgt_push_mapper_component(void *rt_mapper_handle, void 2580 // *base, void *begin, int64_t size, int64_t type); 2581 llvm::Type *TypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy, CGM.VoidPtrTy, 2582 CGM.Int64Ty, CGM.Int64Ty}; 2583 auto *FnTy = 2584 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2585 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_push_mapper_component"); 2586 break; 2587 } 2588 } 2589 assert(RTLFn && "Unable to find OpenMP runtime function"); 2590 return RTLFn; 2591 } 2592 2593 llvm::FunctionCallee 2594 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned) { 2595 assert((IVSize == 32 || IVSize == 64) && 2596 "IV size is not compatible with the omp runtime"); 2597 StringRef Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4" 2598 : "__kmpc_for_static_init_4u") 2599 : (IVSigned ? "__kmpc_for_static_init_8" 2600 : "__kmpc_for_static_init_8u"); 2601 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2602 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 2603 llvm::Type *TypeParams[] = { 2604 getIdentTyPointerTy(), // loc 2605 CGM.Int32Ty, // tid 2606 CGM.Int32Ty, // schedtype 2607 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 2608 PtrTy, // p_lower 2609 PtrTy, // p_upper 2610 PtrTy, // p_stride 2611 ITy, // incr 2612 ITy // chunk 2613 }; 2614 auto *FnTy = 2615 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2616 return CGM.CreateRuntimeFunction(FnTy, Name); 2617 } 2618 2619 llvm::FunctionCallee 2620 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) { 2621 assert((IVSize == 32 || IVSize == 64) && 2622 "IV size is not compatible with the omp runtime"); 2623 StringRef Name = 2624 IVSize == 32 2625 ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u") 2626 : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u"); 2627 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2628 llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc 2629 CGM.Int32Ty, // tid 2630 CGM.Int32Ty, // schedtype 2631 ITy, // lower 2632 ITy, // upper 2633 ITy, // stride 2634 ITy // chunk 2635 }; 2636 auto *FnTy = 2637 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2638 return CGM.CreateRuntimeFunction(FnTy, Name); 2639 } 2640 2641 llvm::FunctionCallee 2642 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) { 2643 assert((IVSize == 32 || IVSize == 64) && 2644 "IV size is not compatible with the omp runtime"); 2645 StringRef Name = 2646 IVSize == 32 2647 ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u") 2648 : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u"); 2649 llvm::Type *TypeParams[] = { 2650 getIdentTyPointerTy(), // loc 2651 CGM.Int32Ty, // tid 2652 }; 2653 auto *FnTy = 2654 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2655 return CGM.CreateRuntimeFunction(FnTy, Name); 2656 } 2657 2658 llvm::FunctionCallee 2659 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) { 2660 assert((IVSize == 32 || IVSize == 64) && 2661 "IV size is not compatible with the omp runtime"); 2662 StringRef Name = 2663 IVSize == 32 2664 ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u") 2665 : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u"); 2666 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2667 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 2668 llvm::Type *TypeParams[] = { 2669 getIdentTyPointerTy(), // loc 2670 CGM.Int32Ty, // tid 2671 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 2672 PtrTy, // p_lower 2673 PtrTy, // p_upper 2674 PtrTy // p_stride 2675 }; 2676 auto *FnTy = 2677 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2678 return CGM.CreateRuntimeFunction(FnTy, Name); 2679 } 2680 2681 /// Obtain information that uniquely identifies a target entry. This 2682 /// consists of the file and device IDs as well as line number associated with 2683 /// the relevant entry source location. 2684 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc, 2685 unsigned &DeviceID, unsigned &FileID, 2686 unsigned &LineNum) { 2687 SourceManager &SM = C.getSourceManager(); 2688 2689 // The loc should be always valid and have a file ID (the user cannot use 2690 // #pragma directives in macros) 2691 2692 assert(Loc.isValid() && "Source location is expected to be always valid."); 2693 2694 PresumedLoc PLoc = SM.getPresumedLoc(Loc); 2695 assert(PLoc.isValid() && "Source location is expected to be always valid."); 2696 2697 llvm::sys::fs::UniqueID ID; 2698 if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID)) 2699 SM.getDiagnostics().Report(diag::err_cannot_open_file) 2700 << PLoc.getFilename() << EC.message(); 2701 2702 DeviceID = ID.getDevice(); 2703 FileID = ID.getFile(); 2704 LineNum = PLoc.getLine(); 2705 } 2706 2707 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) { 2708 if (CGM.getLangOpts().OpenMPSimd) 2709 return Address::invalid(); 2710 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 2711 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 2712 if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link || 2713 (*Res == OMPDeclareTargetDeclAttr::MT_To && 2714 HasRequiresUnifiedSharedMemory))) { 2715 SmallString<64> PtrName; 2716 { 2717 llvm::raw_svector_ostream OS(PtrName); 2718 OS << CGM.getMangledName(GlobalDecl(VD)); 2719 if (!VD->isExternallyVisible()) { 2720 unsigned DeviceID, FileID, Line; 2721 getTargetEntryUniqueInfo(CGM.getContext(), 2722 VD->getCanonicalDecl()->getBeginLoc(), 2723 DeviceID, FileID, Line); 2724 OS << llvm::format("_%x", FileID); 2725 } 2726 OS << "_decl_tgt_ref_ptr"; 2727 } 2728 llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName); 2729 if (!Ptr) { 2730 QualType PtrTy = CGM.getContext().getPointerType(VD->getType()); 2731 Ptr = getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(PtrTy), 2732 PtrName); 2733 2734 auto *GV = cast<llvm::GlobalVariable>(Ptr); 2735 GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 2736 2737 if (!CGM.getLangOpts().OpenMPIsDevice) 2738 GV->setInitializer(CGM.GetAddrOfGlobal(VD)); 2739 registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr)); 2740 } 2741 return Address(Ptr, CGM.getContext().getDeclAlign(VD)); 2742 } 2743 return Address::invalid(); 2744 } 2745 2746 llvm::Constant * 2747 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) { 2748 assert(!CGM.getLangOpts().OpenMPUseTLS || 2749 !CGM.getContext().getTargetInfo().isTLSSupported()); 2750 // Lookup the entry, lazily creating it if necessary. 2751 std::string Suffix = getName({"cache", ""}); 2752 return getOrCreateInternalVariable( 2753 CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix)); 2754 } 2755 2756 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 2757 const VarDecl *VD, 2758 Address VDAddr, 2759 SourceLocation Loc) { 2760 if (CGM.getLangOpts().OpenMPUseTLS && 2761 CGM.getContext().getTargetInfo().isTLSSupported()) 2762 return VDAddr; 2763 2764 llvm::Type *VarTy = VDAddr.getElementType(); 2765 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2766 CGF.Builder.CreatePointerCast(VDAddr.getPointer(), 2767 CGM.Int8PtrTy), 2768 CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)), 2769 getOrCreateThreadPrivateCache(VD)}; 2770 return Address(CGF.EmitRuntimeCall( 2771 createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args), 2772 VDAddr.getAlignment()); 2773 } 2774 2775 void CGOpenMPRuntime::emitThreadPrivateVarInit( 2776 CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor, 2777 llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) { 2778 // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime 2779 // library. 2780 llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc); 2781 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num), 2782 OMPLoc); 2783 // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor) 2784 // to register constructor/destructor for variable. 2785 llvm::Value *Args[] = { 2786 OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy), 2787 Ctor, CopyCtor, Dtor}; 2788 CGF.EmitRuntimeCall( 2789 createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args); 2790 } 2791 2792 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition( 2793 const VarDecl *VD, Address VDAddr, SourceLocation Loc, 2794 bool PerformInit, CodeGenFunction *CGF) { 2795 if (CGM.getLangOpts().OpenMPUseTLS && 2796 CGM.getContext().getTargetInfo().isTLSSupported()) 2797 return nullptr; 2798 2799 VD = VD->getDefinition(CGM.getContext()); 2800 if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) { 2801 QualType ASTTy = VD->getType(); 2802 2803 llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr; 2804 const Expr *Init = VD->getAnyInitializer(); 2805 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 2806 // Generate function that re-emits the declaration's initializer into the 2807 // threadprivate copy of the variable VD 2808 CodeGenFunction CtorCGF(CGM); 2809 FunctionArgList Args; 2810 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 2811 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 2812 ImplicitParamDecl::Other); 2813 Args.push_back(&Dst); 2814 2815 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 2816 CGM.getContext().VoidPtrTy, Args); 2817 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2818 std::string Name = getName({"__kmpc_global_ctor_", ""}); 2819 llvm::Function *Fn = 2820 CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc); 2821 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI, 2822 Args, Loc, Loc); 2823 llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar( 2824 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 2825 CGM.getContext().VoidPtrTy, Dst.getLocation()); 2826 Address Arg = Address(ArgVal, VDAddr.getAlignment()); 2827 Arg = CtorCGF.Builder.CreateElementBitCast( 2828 Arg, CtorCGF.ConvertTypeForMem(ASTTy)); 2829 CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(), 2830 /*IsInitializer=*/true); 2831 ArgVal = CtorCGF.EmitLoadOfScalar( 2832 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 2833 CGM.getContext().VoidPtrTy, Dst.getLocation()); 2834 CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue); 2835 CtorCGF.FinishFunction(); 2836 Ctor = Fn; 2837 } 2838 if (VD->getType().isDestructedType() != QualType::DK_none) { 2839 // Generate function that emits destructor call for the threadprivate copy 2840 // of the variable VD 2841 CodeGenFunction DtorCGF(CGM); 2842 FunctionArgList Args; 2843 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 2844 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 2845 ImplicitParamDecl::Other); 2846 Args.push_back(&Dst); 2847 2848 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 2849 CGM.getContext().VoidTy, Args); 2850 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2851 std::string Name = getName({"__kmpc_global_dtor_", ""}); 2852 llvm::Function *Fn = 2853 CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc); 2854 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 2855 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args, 2856 Loc, Loc); 2857 // Create a scope with an artificial location for the body of this function. 2858 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 2859 llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar( 2860 DtorCGF.GetAddrOfLocalVar(&Dst), 2861 /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation()); 2862 DtorCGF.emitDestroy(Address(ArgVal, VDAddr.getAlignment()), ASTTy, 2863 DtorCGF.getDestroyer(ASTTy.isDestructedType()), 2864 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 2865 DtorCGF.FinishFunction(); 2866 Dtor = Fn; 2867 } 2868 // Do not emit init function if it is not required. 2869 if (!Ctor && !Dtor) 2870 return nullptr; 2871 2872 llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2873 auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs, 2874 /*isVarArg=*/false) 2875 ->getPointerTo(); 2876 // Copying constructor for the threadprivate variable. 2877 // Must be NULL - reserved by runtime, but currently it requires that this 2878 // parameter is always NULL. Otherwise it fires assertion. 2879 CopyCtor = llvm::Constant::getNullValue(CopyCtorTy); 2880 if (Ctor == nullptr) { 2881 auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 2882 /*isVarArg=*/false) 2883 ->getPointerTo(); 2884 Ctor = llvm::Constant::getNullValue(CtorTy); 2885 } 2886 if (Dtor == nullptr) { 2887 auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, 2888 /*isVarArg=*/false) 2889 ->getPointerTo(); 2890 Dtor = llvm::Constant::getNullValue(DtorTy); 2891 } 2892 if (!CGF) { 2893 auto *InitFunctionTy = 2894 llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false); 2895 std::string Name = getName({"__omp_threadprivate_init_", ""}); 2896 llvm::Function *InitFunction = CGM.CreateGlobalInitOrDestructFunction( 2897 InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction()); 2898 CodeGenFunction InitCGF(CGM); 2899 FunctionArgList ArgList; 2900 InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction, 2901 CGM.getTypes().arrangeNullaryFunction(), ArgList, 2902 Loc, Loc); 2903 emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 2904 InitCGF.FinishFunction(); 2905 return InitFunction; 2906 } 2907 emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 2908 } 2909 return nullptr; 2910 } 2911 2912 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD, 2913 llvm::GlobalVariable *Addr, 2914 bool PerformInit) { 2915 if (CGM.getLangOpts().OMPTargetTriples.empty() && 2916 !CGM.getLangOpts().OpenMPIsDevice) 2917 return false; 2918 Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 2919 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 2920 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 2921 (*Res == OMPDeclareTargetDeclAttr::MT_To && 2922 HasRequiresUnifiedSharedMemory)) 2923 return CGM.getLangOpts().OpenMPIsDevice; 2924 VD = VD->getDefinition(CGM.getContext()); 2925 if (VD && !DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second) 2926 return CGM.getLangOpts().OpenMPIsDevice; 2927 2928 QualType ASTTy = VD->getType(); 2929 2930 SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc(); 2931 // Produce the unique prefix to identify the new target regions. We use 2932 // the source location of the variable declaration which we know to not 2933 // conflict with any target region. 2934 unsigned DeviceID; 2935 unsigned FileID; 2936 unsigned Line; 2937 getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line); 2938 SmallString<128> Buffer, Out; 2939 { 2940 llvm::raw_svector_ostream OS(Buffer); 2941 OS << "__omp_offloading_" << llvm::format("_%x", DeviceID) 2942 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 2943 } 2944 2945 const Expr *Init = VD->getAnyInitializer(); 2946 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 2947 llvm::Constant *Ctor; 2948 llvm::Constant *ID; 2949 if (CGM.getLangOpts().OpenMPIsDevice) { 2950 // Generate function that re-emits the declaration's initializer into 2951 // the threadprivate copy of the variable VD 2952 CodeGenFunction CtorCGF(CGM); 2953 2954 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 2955 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2956 llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction( 2957 FTy, Twine(Buffer, "_ctor"), FI, Loc); 2958 auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF); 2959 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 2960 FunctionArgList(), Loc, Loc); 2961 auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF); 2962 CtorCGF.EmitAnyExprToMem(Init, 2963 Address(Addr, CGM.getContext().getDeclAlign(VD)), 2964 Init->getType().getQualifiers(), 2965 /*IsInitializer=*/true); 2966 CtorCGF.FinishFunction(); 2967 Ctor = Fn; 2968 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 2969 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor)); 2970 } else { 2971 Ctor = new llvm::GlobalVariable( 2972 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 2973 llvm::GlobalValue::PrivateLinkage, 2974 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor")); 2975 ID = Ctor; 2976 } 2977 2978 // Register the information for the entry associated with the constructor. 2979 Out.clear(); 2980 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 2981 DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor, 2982 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor); 2983 } 2984 if (VD->getType().isDestructedType() != QualType::DK_none) { 2985 llvm::Constant *Dtor; 2986 llvm::Constant *ID; 2987 if (CGM.getLangOpts().OpenMPIsDevice) { 2988 // Generate function that emits destructor call for the threadprivate 2989 // copy of the variable VD 2990 CodeGenFunction DtorCGF(CGM); 2991 2992 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 2993 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2994 llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction( 2995 FTy, Twine(Buffer, "_dtor"), FI, Loc); 2996 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 2997 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 2998 FunctionArgList(), Loc, Loc); 2999 // Create a scope with an artificial location for the body of this 3000 // function. 3001 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 3002 DtorCGF.emitDestroy(Address(Addr, CGM.getContext().getDeclAlign(VD)), 3003 ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()), 3004 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 3005 DtorCGF.FinishFunction(); 3006 Dtor = Fn; 3007 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 3008 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor)); 3009 } else { 3010 Dtor = new llvm::GlobalVariable( 3011 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 3012 llvm::GlobalValue::PrivateLinkage, 3013 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor")); 3014 ID = Dtor; 3015 } 3016 // Register the information for the entry associated with the destructor. 3017 Out.clear(); 3018 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 3019 DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor, 3020 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor); 3021 } 3022 return CGM.getLangOpts().OpenMPIsDevice; 3023 } 3024 3025 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF, 3026 QualType VarType, 3027 StringRef Name) { 3028 std::string Suffix = getName({"artificial", ""}); 3029 llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType); 3030 llvm::Value *GAddr = 3031 getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix)); 3032 if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS && 3033 CGM.getTarget().isTLSSupported()) { 3034 cast<llvm::GlobalVariable>(GAddr)->setThreadLocal(/*Val=*/true); 3035 return Address(GAddr, CGM.getContext().getTypeAlignInChars(VarType)); 3036 } 3037 std::string CacheSuffix = getName({"cache", ""}); 3038 llvm::Value *Args[] = { 3039 emitUpdateLocation(CGF, SourceLocation()), 3040 getThreadID(CGF, SourceLocation()), 3041 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy), 3042 CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy, 3043 /*isSigned=*/false), 3044 getOrCreateInternalVariable( 3045 CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))}; 3046 return Address( 3047 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3048 CGF.EmitRuntimeCall( 3049 createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args), 3050 VarLVType->getPointerTo(/*AddrSpace=*/0)), 3051 CGM.getContext().getTypeAlignInChars(VarType)); 3052 } 3053 3054 void CGOpenMPRuntime::emitIfClause(CodeGenFunction &CGF, const Expr *Cond, 3055 const RegionCodeGenTy &ThenGen, 3056 const RegionCodeGenTy &ElseGen) { 3057 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange()); 3058 3059 // If the condition constant folds and can be elided, try to avoid emitting 3060 // the condition and the dead arm of the if/else. 3061 bool CondConstant; 3062 if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) { 3063 if (CondConstant) 3064 ThenGen(CGF); 3065 else 3066 ElseGen(CGF); 3067 return; 3068 } 3069 3070 // Otherwise, the condition did not fold, or we couldn't elide it. Just 3071 // emit the conditional branch. 3072 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then"); 3073 llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else"); 3074 llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end"); 3075 CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0); 3076 3077 // Emit the 'then' code. 3078 CGF.EmitBlock(ThenBlock); 3079 ThenGen(CGF); 3080 CGF.EmitBranch(ContBlock); 3081 // Emit the 'else' code if present. 3082 // There is no need to emit line number for unconditional branch. 3083 (void)ApplyDebugLocation::CreateEmpty(CGF); 3084 CGF.EmitBlock(ElseBlock); 3085 ElseGen(CGF); 3086 // There is no need to emit line number for unconditional branch. 3087 (void)ApplyDebugLocation::CreateEmpty(CGF); 3088 CGF.EmitBranch(ContBlock); 3089 // Emit the continuation block for code after the if. 3090 CGF.EmitBlock(ContBlock, /*IsFinished=*/true); 3091 } 3092 3093 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc, 3094 llvm::Function *OutlinedFn, 3095 ArrayRef<llvm::Value *> CapturedVars, 3096 const Expr *IfCond) { 3097 if (!CGF.HaveInsertPoint()) 3098 return; 3099 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 3100 auto &&ThenGen = [OutlinedFn, CapturedVars, RTLoc](CodeGenFunction &CGF, 3101 PrePostActionTy &) { 3102 // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn); 3103 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 3104 llvm::Value *Args[] = { 3105 RTLoc, 3106 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 3107 CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())}; 3108 llvm::SmallVector<llvm::Value *, 16> RealArgs; 3109 RealArgs.append(std::begin(Args), std::end(Args)); 3110 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 3111 3112 llvm::FunctionCallee RTLFn = 3113 RT.createRuntimeFunction(OMPRTL__kmpc_fork_call); 3114 CGF.EmitRuntimeCall(RTLFn, RealArgs); 3115 }; 3116 auto &&ElseGen = [OutlinedFn, CapturedVars, RTLoc, Loc](CodeGenFunction &CGF, 3117 PrePostActionTy &) { 3118 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 3119 llvm::Value *ThreadID = RT.getThreadID(CGF, Loc); 3120 // Build calls: 3121 // __kmpc_serialized_parallel(&Loc, GTid); 3122 llvm::Value *Args[] = {RTLoc, ThreadID}; 3123 CGF.EmitRuntimeCall( 3124 RT.createRuntimeFunction(OMPRTL__kmpc_serialized_parallel), Args); 3125 3126 // OutlinedFn(>id, &zero_bound, CapturedStruct); 3127 Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc); 3128 Address ZeroAddrBound = 3129 CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty, 3130 /*Name=*/".bound.zero.addr"); 3131 CGF.InitTempAlloca(ZeroAddrBound, CGF.Builder.getInt32(/*C*/ 0)); 3132 llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs; 3133 // ThreadId for serialized parallels is 0. 3134 OutlinedFnArgs.push_back(ThreadIDAddr.getPointer()); 3135 OutlinedFnArgs.push_back(ZeroAddrBound.getPointer()); 3136 OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end()); 3137 RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs); 3138 3139 // __kmpc_end_serialized_parallel(&Loc, GTid); 3140 llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID}; 3141 CGF.EmitRuntimeCall( 3142 RT.createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel), 3143 EndArgs); 3144 }; 3145 if (IfCond) { 3146 emitIfClause(CGF, IfCond, ThenGen, ElseGen); 3147 } else { 3148 RegionCodeGenTy ThenRCG(ThenGen); 3149 ThenRCG(CGF); 3150 } 3151 } 3152 3153 // If we're inside an (outlined) parallel region, use the region info's 3154 // thread-ID variable (it is passed in a first argument of the outlined function 3155 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in 3156 // regular serial code region, get thread ID by calling kmp_int32 3157 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and 3158 // return the address of that temp. 3159 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF, 3160 SourceLocation Loc) { 3161 if (auto *OMPRegionInfo = 3162 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 3163 if (OMPRegionInfo->getThreadIDVariable()) 3164 return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(CGF); 3165 3166 llvm::Value *ThreadID = getThreadID(CGF, Loc); 3167 QualType Int32Ty = 3168 CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true); 3169 Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp."); 3170 CGF.EmitStoreOfScalar(ThreadID, 3171 CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty)); 3172 3173 return ThreadIDTemp; 3174 } 3175 3176 llvm::Constant *CGOpenMPRuntime::getOrCreateInternalVariable( 3177 llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) { 3178 SmallString<256> Buffer; 3179 llvm::raw_svector_ostream Out(Buffer); 3180 Out << Name; 3181 StringRef RuntimeName = Out.str(); 3182 auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first; 3183 if (Elem.second) { 3184 assert(Elem.second->getType()->getPointerElementType() == Ty && 3185 "OMP internal variable has different type than requested"); 3186 return &*Elem.second; 3187 } 3188 3189 return Elem.second = new llvm::GlobalVariable( 3190 CGM.getModule(), Ty, /*IsConstant*/ false, 3191 llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty), 3192 Elem.first(), /*InsertBefore=*/nullptr, 3193 llvm::GlobalValue::NotThreadLocal, AddressSpace); 3194 } 3195 3196 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) { 3197 std::string Prefix = Twine("gomp_critical_user_", CriticalName).str(); 3198 std::string Name = getName({Prefix, "var"}); 3199 return getOrCreateInternalVariable(KmpCriticalNameTy, Name); 3200 } 3201 3202 namespace { 3203 /// Common pre(post)-action for different OpenMP constructs. 3204 class CommonActionTy final : public PrePostActionTy { 3205 llvm::FunctionCallee EnterCallee; 3206 ArrayRef<llvm::Value *> EnterArgs; 3207 llvm::FunctionCallee ExitCallee; 3208 ArrayRef<llvm::Value *> ExitArgs; 3209 bool Conditional; 3210 llvm::BasicBlock *ContBlock = nullptr; 3211 3212 public: 3213 CommonActionTy(llvm::FunctionCallee EnterCallee, 3214 ArrayRef<llvm::Value *> EnterArgs, 3215 llvm::FunctionCallee ExitCallee, 3216 ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false) 3217 : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee), 3218 ExitArgs(ExitArgs), Conditional(Conditional) {} 3219 void Enter(CodeGenFunction &CGF) override { 3220 llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs); 3221 if (Conditional) { 3222 llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes); 3223 auto *ThenBlock = CGF.createBasicBlock("omp_if.then"); 3224 ContBlock = CGF.createBasicBlock("omp_if.end"); 3225 // Generate the branch (If-stmt) 3226 CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock); 3227 CGF.EmitBlock(ThenBlock); 3228 } 3229 } 3230 void Done(CodeGenFunction &CGF) { 3231 // Emit the rest of blocks/branches 3232 CGF.EmitBranch(ContBlock); 3233 CGF.EmitBlock(ContBlock, true); 3234 } 3235 void Exit(CodeGenFunction &CGF) override { 3236 CGF.EmitRuntimeCall(ExitCallee, ExitArgs); 3237 } 3238 }; 3239 } // anonymous namespace 3240 3241 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF, 3242 StringRef CriticalName, 3243 const RegionCodeGenTy &CriticalOpGen, 3244 SourceLocation Loc, const Expr *Hint) { 3245 // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]); 3246 // CriticalOpGen(); 3247 // __kmpc_end_critical(ident_t *, gtid, Lock); 3248 // Prepare arguments and build a call to __kmpc_critical 3249 if (!CGF.HaveInsertPoint()) 3250 return; 3251 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3252 getCriticalRegionLock(CriticalName)}; 3253 llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args), 3254 std::end(Args)); 3255 if (Hint) { 3256 EnterArgs.push_back(CGF.Builder.CreateIntCast( 3257 CGF.EmitScalarExpr(Hint), CGM.IntPtrTy, /*isSigned=*/false)); 3258 } 3259 CommonActionTy Action( 3260 createRuntimeFunction(Hint ? OMPRTL__kmpc_critical_with_hint 3261 : OMPRTL__kmpc_critical), 3262 EnterArgs, createRuntimeFunction(OMPRTL__kmpc_end_critical), Args); 3263 CriticalOpGen.setAction(Action); 3264 emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen); 3265 } 3266 3267 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF, 3268 const RegionCodeGenTy &MasterOpGen, 3269 SourceLocation Loc) { 3270 if (!CGF.HaveInsertPoint()) 3271 return; 3272 // if(__kmpc_master(ident_t *, gtid)) { 3273 // MasterOpGen(); 3274 // __kmpc_end_master(ident_t *, gtid); 3275 // } 3276 // Prepare arguments and build a call to __kmpc_master 3277 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3278 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_master), Args, 3279 createRuntimeFunction(OMPRTL__kmpc_end_master), Args, 3280 /*Conditional=*/true); 3281 MasterOpGen.setAction(Action); 3282 emitInlinedDirective(CGF, OMPD_master, MasterOpGen); 3283 Action.Done(CGF); 3284 } 3285 3286 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 3287 SourceLocation Loc) { 3288 if (!CGF.HaveInsertPoint()) 3289 return; 3290 // Build call __kmpc_omp_taskyield(loc, thread_id, 0); 3291 llvm::Value *Args[] = { 3292 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3293 llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)}; 3294 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield), Args); 3295 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 3296 Region->emitUntiedSwitch(CGF); 3297 } 3298 3299 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF, 3300 const RegionCodeGenTy &TaskgroupOpGen, 3301 SourceLocation Loc) { 3302 if (!CGF.HaveInsertPoint()) 3303 return; 3304 // __kmpc_taskgroup(ident_t *, gtid); 3305 // TaskgroupOpGen(); 3306 // __kmpc_end_taskgroup(ident_t *, gtid); 3307 // Prepare arguments and build a call to __kmpc_taskgroup 3308 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3309 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_taskgroup), Args, 3310 createRuntimeFunction(OMPRTL__kmpc_end_taskgroup), 3311 Args); 3312 TaskgroupOpGen.setAction(Action); 3313 emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen); 3314 } 3315 3316 /// Given an array of pointers to variables, project the address of a 3317 /// given variable. 3318 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array, 3319 unsigned Index, const VarDecl *Var) { 3320 // Pull out the pointer to the variable. 3321 Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index); 3322 llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr); 3323 3324 Address Addr = Address(Ptr, CGF.getContext().getDeclAlign(Var)); 3325 Addr = CGF.Builder.CreateElementBitCast( 3326 Addr, CGF.ConvertTypeForMem(Var->getType())); 3327 return Addr; 3328 } 3329 3330 static llvm::Value *emitCopyprivateCopyFunction( 3331 CodeGenModule &CGM, llvm::Type *ArgsType, 3332 ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs, 3333 ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps, 3334 SourceLocation Loc) { 3335 ASTContext &C = CGM.getContext(); 3336 // void copy_func(void *LHSArg, void *RHSArg); 3337 FunctionArgList Args; 3338 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 3339 ImplicitParamDecl::Other); 3340 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 3341 ImplicitParamDecl::Other); 3342 Args.push_back(&LHSArg); 3343 Args.push_back(&RHSArg); 3344 const auto &CGFI = 3345 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 3346 std::string Name = 3347 CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"}); 3348 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 3349 llvm::GlobalValue::InternalLinkage, Name, 3350 &CGM.getModule()); 3351 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 3352 Fn->setDoesNotRecurse(); 3353 CodeGenFunction CGF(CGM); 3354 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 3355 // Dest = (void*[n])(LHSArg); 3356 // Src = (void*[n])(RHSArg); 3357 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3358 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 3359 ArgsType), CGF.getPointerAlign()); 3360 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3361 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 3362 ArgsType), CGF.getPointerAlign()); 3363 // *(Type0*)Dst[0] = *(Type0*)Src[0]; 3364 // *(Type1*)Dst[1] = *(Type1*)Src[1]; 3365 // ... 3366 // *(Typen*)Dst[n] = *(Typen*)Src[n]; 3367 for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) { 3368 const auto *DestVar = 3369 cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl()); 3370 Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar); 3371 3372 const auto *SrcVar = 3373 cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl()); 3374 Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar); 3375 3376 const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl(); 3377 QualType Type = VD->getType(); 3378 CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]); 3379 } 3380 CGF.FinishFunction(); 3381 return Fn; 3382 } 3383 3384 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF, 3385 const RegionCodeGenTy &SingleOpGen, 3386 SourceLocation Loc, 3387 ArrayRef<const Expr *> CopyprivateVars, 3388 ArrayRef<const Expr *> SrcExprs, 3389 ArrayRef<const Expr *> DstExprs, 3390 ArrayRef<const Expr *> AssignmentOps) { 3391 if (!CGF.HaveInsertPoint()) 3392 return; 3393 assert(CopyprivateVars.size() == SrcExprs.size() && 3394 CopyprivateVars.size() == DstExprs.size() && 3395 CopyprivateVars.size() == AssignmentOps.size()); 3396 ASTContext &C = CGM.getContext(); 3397 // int32 did_it = 0; 3398 // if(__kmpc_single(ident_t *, gtid)) { 3399 // SingleOpGen(); 3400 // __kmpc_end_single(ident_t *, gtid); 3401 // did_it = 1; 3402 // } 3403 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 3404 // <copy_func>, did_it); 3405 3406 Address DidIt = Address::invalid(); 3407 if (!CopyprivateVars.empty()) { 3408 // int32 did_it = 0; 3409 QualType KmpInt32Ty = 3410 C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 3411 DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it"); 3412 CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt); 3413 } 3414 // Prepare arguments and build a call to __kmpc_single 3415 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3416 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_single), Args, 3417 createRuntimeFunction(OMPRTL__kmpc_end_single), Args, 3418 /*Conditional=*/true); 3419 SingleOpGen.setAction(Action); 3420 emitInlinedDirective(CGF, OMPD_single, SingleOpGen); 3421 if (DidIt.isValid()) { 3422 // did_it = 1; 3423 CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt); 3424 } 3425 Action.Done(CGF); 3426 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 3427 // <copy_func>, did_it); 3428 if (DidIt.isValid()) { 3429 llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size()); 3430 QualType CopyprivateArrayTy = C.getConstantArrayType( 3431 C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 3432 /*IndexTypeQuals=*/0); 3433 // Create a list of all private variables for copyprivate. 3434 Address CopyprivateList = 3435 CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list"); 3436 for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) { 3437 Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I); 3438 CGF.Builder.CreateStore( 3439 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3440 CGF.EmitLValue(CopyprivateVars[I]).getPointer(CGF), 3441 CGF.VoidPtrTy), 3442 Elem); 3443 } 3444 // Build function that copies private values from single region to all other 3445 // threads in the corresponding parallel region. 3446 llvm::Value *CpyFn = emitCopyprivateCopyFunction( 3447 CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(), 3448 CopyprivateVars, SrcExprs, DstExprs, AssignmentOps, Loc); 3449 llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy); 3450 Address CL = 3451 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList, 3452 CGF.VoidPtrTy); 3453 llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt); 3454 llvm::Value *Args[] = { 3455 emitUpdateLocation(CGF, Loc), // ident_t *<loc> 3456 getThreadID(CGF, Loc), // i32 <gtid> 3457 BufSize, // size_t <buf_size> 3458 CL.getPointer(), // void *<copyprivate list> 3459 CpyFn, // void (*) (void *, void *) <copy_func> 3460 DidItVal // i32 did_it 3461 }; 3462 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_copyprivate), Args); 3463 } 3464 } 3465 3466 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF, 3467 const RegionCodeGenTy &OrderedOpGen, 3468 SourceLocation Loc, bool IsThreads) { 3469 if (!CGF.HaveInsertPoint()) 3470 return; 3471 // __kmpc_ordered(ident_t *, gtid); 3472 // OrderedOpGen(); 3473 // __kmpc_end_ordered(ident_t *, gtid); 3474 // Prepare arguments and build a call to __kmpc_ordered 3475 if (IsThreads) { 3476 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3477 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_ordered), Args, 3478 createRuntimeFunction(OMPRTL__kmpc_end_ordered), 3479 Args); 3480 OrderedOpGen.setAction(Action); 3481 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 3482 return; 3483 } 3484 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 3485 } 3486 3487 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) { 3488 unsigned Flags; 3489 if (Kind == OMPD_for) 3490 Flags = OMP_IDENT_BARRIER_IMPL_FOR; 3491 else if (Kind == OMPD_sections) 3492 Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS; 3493 else if (Kind == OMPD_single) 3494 Flags = OMP_IDENT_BARRIER_IMPL_SINGLE; 3495 else if (Kind == OMPD_barrier) 3496 Flags = OMP_IDENT_BARRIER_EXPL; 3497 else 3498 Flags = OMP_IDENT_BARRIER_IMPL; 3499 return Flags; 3500 } 3501 3502 void CGOpenMPRuntime::getDefaultScheduleAndChunk( 3503 CodeGenFunction &CGF, const OMPLoopDirective &S, 3504 OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const { 3505 // Check if the loop directive is actually a doacross loop directive. In this 3506 // case choose static, 1 schedule. 3507 if (llvm::any_of( 3508 S.getClausesOfKind<OMPOrderedClause>(), 3509 [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) { 3510 ScheduleKind = OMPC_SCHEDULE_static; 3511 // Chunk size is 1 in this case. 3512 llvm::APInt ChunkSize(32, 1); 3513 ChunkExpr = IntegerLiteral::Create( 3514 CGF.getContext(), ChunkSize, 3515 CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0), 3516 SourceLocation()); 3517 } 3518 } 3519 3520 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc, 3521 OpenMPDirectiveKind Kind, bool EmitChecks, 3522 bool ForceSimpleCall) { 3523 // Check if we should use the OMPBuilder 3524 auto *OMPRegionInfo = 3525 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo); 3526 llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder(); 3527 if (OMPBuilder) { 3528 CGF.Builder.restoreIP(OMPBuilder->CreateBarrier( 3529 CGF.Builder, Kind, ForceSimpleCall, EmitChecks)); 3530 return; 3531 } 3532 3533 if (!CGF.HaveInsertPoint()) 3534 return; 3535 // Build call __kmpc_cancel_barrier(loc, thread_id); 3536 // Build call __kmpc_barrier(loc, thread_id); 3537 unsigned Flags = getDefaultFlagsForBarriers(Kind); 3538 // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc, 3539 // thread_id); 3540 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags), 3541 getThreadID(CGF, Loc)}; 3542 if (OMPRegionInfo) { 3543 if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) { 3544 llvm::Value *Result = CGF.EmitRuntimeCall( 3545 createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args); 3546 if (EmitChecks) { 3547 // if (__kmpc_cancel_barrier()) { 3548 // exit from construct; 3549 // } 3550 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 3551 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 3552 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 3553 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 3554 CGF.EmitBlock(ExitBB); 3555 // exit from construct; 3556 CodeGenFunction::JumpDest CancelDestination = 3557 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 3558 CGF.EmitBranchThroughCleanup(CancelDestination); 3559 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 3560 } 3561 return; 3562 } 3563 } 3564 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_barrier), Args); 3565 } 3566 3567 /// Map the OpenMP loop schedule to the runtime enumeration. 3568 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind, 3569 bool Chunked, bool Ordered) { 3570 switch (ScheduleKind) { 3571 case OMPC_SCHEDULE_static: 3572 return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked) 3573 : (Ordered ? OMP_ord_static : OMP_sch_static); 3574 case OMPC_SCHEDULE_dynamic: 3575 return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked; 3576 case OMPC_SCHEDULE_guided: 3577 return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked; 3578 case OMPC_SCHEDULE_runtime: 3579 return Ordered ? OMP_ord_runtime : OMP_sch_runtime; 3580 case OMPC_SCHEDULE_auto: 3581 return Ordered ? OMP_ord_auto : OMP_sch_auto; 3582 case OMPC_SCHEDULE_unknown: 3583 assert(!Chunked && "chunk was specified but schedule kind not known"); 3584 return Ordered ? OMP_ord_static : OMP_sch_static; 3585 } 3586 llvm_unreachable("Unexpected runtime schedule"); 3587 } 3588 3589 /// Map the OpenMP distribute schedule to the runtime enumeration. 3590 static OpenMPSchedType 3591 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) { 3592 // only static is allowed for dist_schedule 3593 return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static; 3594 } 3595 3596 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind, 3597 bool Chunked) const { 3598 OpenMPSchedType Schedule = 3599 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 3600 return Schedule == OMP_sch_static; 3601 } 3602 3603 bool CGOpenMPRuntime::isStaticNonchunked( 3604 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 3605 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 3606 return Schedule == OMP_dist_sch_static; 3607 } 3608 3609 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind, 3610 bool Chunked) const { 3611 OpenMPSchedType Schedule = 3612 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 3613 return Schedule == OMP_sch_static_chunked; 3614 } 3615 3616 bool CGOpenMPRuntime::isStaticChunked( 3617 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 3618 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 3619 return Schedule == OMP_dist_sch_static_chunked; 3620 } 3621 3622 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const { 3623 OpenMPSchedType Schedule = 3624 getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false); 3625 assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here"); 3626 return Schedule != OMP_sch_static; 3627 } 3628 3629 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule, 3630 OpenMPScheduleClauseModifier M1, 3631 OpenMPScheduleClauseModifier M2) { 3632 int Modifier = 0; 3633 switch (M1) { 3634 case OMPC_SCHEDULE_MODIFIER_monotonic: 3635 Modifier = OMP_sch_modifier_monotonic; 3636 break; 3637 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 3638 Modifier = OMP_sch_modifier_nonmonotonic; 3639 break; 3640 case OMPC_SCHEDULE_MODIFIER_simd: 3641 if (Schedule == OMP_sch_static_chunked) 3642 Schedule = OMP_sch_static_balanced_chunked; 3643 break; 3644 case OMPC_SCHEDULE_MODIFIER_last: 3645 case OMPC_SCHEDULE_MODIFIER_unknown: 3646 break; 3647 } 3648 switch (M2) { 3649 case OMPC_SCHEDULE_MODIFIER_monotonic: 3650 Modifier = OMP_sch_modifier_monotonic; 3651 break; 3652 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 3653 Modifier = OMP_sch_modifier_nonmonotonic; 3654 break; 3655 case OMPC_SCHEDULE_MODIFIER_simd: 3656 if (Schedule == OMP_sch_static_chunked) 3657 Schedule = OMP_sch_static_balanced_chunked; 3658 break; 3659 case OMPC_SCHEDULE_MODIFIER_last: 3660 case OMPC_SCHEDULE_MODIFIER_unknown: 3661 break; 3662 } 3663 // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription. 3664 // If the static schedule kind is specified or if the ordered clause is 3665 // specified, and if the nonmonotonic modifier is not specified, the effect is 3666 // as if the monotonic modifier is specified. Otherwise, unless the monotonic 3667 // modifier is specified, the effect is as if the nonmonotonic modifier is 3668 // specified. 3669 if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) { 3670 if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static || 3671 Schedule == OMP_sch_static_balanced_chunked || 3672 Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static || 3673 Schedule == OMP_dist_sch_static_chunked || 3674 Schedule == OMP_dist_sch_static)) 3675 Modifier = OMP_sch_modifier_nonmonotonic; 3676 } 3677 return Schedule | Modifier; 3678 } 3679 3680 void CGOpenMPRuntime::emitForDispatchInit( 3681 CodeGenFunction &CGF, SourceLocation Loc, 3682 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 3683 bool Ordered, const DispatchRTInput &DispatchValues) { 3684 if (!CGF.HaveInsertPoint()) 3685 return; 3686 OpenMPSchedType Schedule = getRuntimeSchedule( 3687 ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered); 3688 assert(Ordered || 3689 (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked && 3690 Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked && 3691 Schedule != OMP_sch_static_balanced_chunked)); 3692 // Call __kmpc_dispatch_init( 3693 // ident_t *loc, kmp_int32 tid, kmp_int32 schedule, 3694 // kmp_int[32|64] lower, kmp_int[32|64] upper, 3695 // kmp_int[32|64] stride, kmp_int[32|64] chunk); 3696 3697 // If the Chunk was not specified in the clause - use default value 1. 3698 llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk 3699 : CGF.Builder.getIntN(IVSize, 1); 3700 llvm::Value *Args[] = { 3701 emitUpdateLocation(CGF, Loc), 3702 getThreadID(CGF, Loc), 3703 CGF.Builder.getInt32(addMonoNonMonoModifier( 3704 CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type 3705 DispatchValues.LB, // Lower 3706 DispatchValues.UB, // Upper 3707 CGF.Builder.getIntN(IVSize, 1), // Stride 3708 Chunk // Chunk 3709 }; 3710 CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args); 3711 } 3712 3713 static void emitForStaticInitCall( 3714 CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId, 3715 llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule, 3716 OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2, 3717 const CGOpenMPRuntime::StaticRTInput &Values) { 3718 if (!CGF.HaveInsertPoint()) 3719 return; 3720 3721 assert(!Values.Ordered); 3722 assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked || 3723 Schedule == OMP_sch_static_balanced_chunked || 3724 Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked || 3725 Schedule == OMP_dist_sch_static || 3726 Schedule == OMP_dist_sch_static_chunked); 3727 3728 // Call __kmpc_for_static_init( 3729 // ident_t *loc, kmp_int32 tid, kmp_int32 schedtype, 3730 // kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower, 3731 // kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride, 3732 // kmp_int[32|64] incr, kmp_int[32|64] chunk); 3733 llvm::Value *Chunk = Values.Chunk; 3734 if (Chunk == nullptr) { 3735 assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static || 3736 Schedule == OMP_dist_sch_static) && 3737 "expected static non-chunked schedule"); 3738 // If the Chunk was not specified in the clause - use default value 1. 3739 Chunk = CGF.Builder.getIntN(Values.IVSize, 1); 3740 } else { 3741 assert((Schedule == OMP_sch_static_chunked || 3742 Schedule == OMP_sch_static_balanced_chunked || 3743 Schedule == OMP_ord_static_chunked || 3744 Schedule == OMP_dist_sch_static_chunked) && 3745 "expected static chunked schedule"); 3746 } 3747 llvm::Value *Args[] = { 3748 UpdateLocation, 3749 ThreadId, 3750 CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1, 3751 M2)), // Schedule type 3752 Values.IL.getPointer(), // &isLastIter 3753 Values.LB.getPointer(), // &LB 3754 Values.UB.getPointer(), // &UB 3755 Values.ST.getPointer(), // &Stride 3756 CGF.Builder.getIntN(Values.IVSize, 1), // Incr 3757 Chunk // Chunk 3758 }; 3759 CGF.EmitRuntimeCall(ForStaticInitFunction, Args); 3760 } 3761 3762 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF, 3763 SourceLocation Loc, 3764 OpenMPDirectiveKind DKind, 3765 const OpenMPScheduleTy &ScheduleKind, 3766 const StaticRTInput &Values) { 3767 OpenMPSchedType ScheduleNum = getRuntimeSchedule( 3768 ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered); 3769 assert(isOpenMPWorksharingDirective(DKind) && 3770 "Expected loop-based or sections-based directive."); 3771 llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc, 3772 isOpenMPLoopDirective(DKind) 3773 ? OMP_IDENT_WORK_LOOP 3774 : OMP_IDENT_WORK_SECTIONS); 3775 llvm::Value *ThreadId = getThreadID(CGF, Loc); 3776 llvm::FunctionCallee StaticInitFunction = 3777 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 3778 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 3779 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 3780 ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values); 3781 } 3782 3783 void CGOpenMPRuntime::emitDistributeStaticInit( 3784 CodeGenFunction &CGF, SourceLocation Loc, 3785 OpenMPDistScheduleClauseKind SchedKind, 3786 const CGOpenMPRuntime::StaticRTInput &Values) { 3787 OpenMPSchedType ScheduleNum = 3788 getRuntimeSchedule(SchedKind, Values.Chunk != nullptr); 3789 llvm::Value *UpdatedLocation = 3790 emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE); 3791 llvm::Value *ThreadId = getThreadID(CGF, Loc); 3792 llvm::FunctionCallee StaticInitFunction = 3793 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 3794 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 3795 ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown, 3796 OMPC_SCHEDULE_MODIFIER_unknown, Values); 3797 } 3798 3799 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF, 3800 SourceLocation Loc, 3801 OpenMPDirectiveKind DKind) { 3802 if (!CGF.HaveInsertPoint()) 3803 return; 3804 // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid); 3805 llvm::Value *Args[] = { 3806 emitUpdateLocation(CGF, Loc, 3807 isOpenMPDistributeDirective(DKind) 3808 ? OMP_IDENT_WORK_DISTRIBUTE 3809 : isOpenMPLoopDirective(DKind) 3810 ? OMP_IDENT_WORK_LOOP 3811 : OMP_IDENT_WORK_SECTIONS), 3812 getThreadID(CGF, Loc)}; 3813 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 3814 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini), 3815 Args); 3816 } 3817 3818 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 3819 SourceLocation Loc, 3820 unsigned IVSize, 3821 bool IVSigned) { 3822 if (!CGF.HaveInsertPoint()) 3823 return; 3824 // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid); 3825 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3826 CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args); 3827 } 3828 3829 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF, 3830 SourceLocation Loc, unsigned IVSize, 3831 bool IVSigned, Address IL, 3832 Address LB, Address UB, 3833 Address ST) { 3834 // Call __kmpc_dispatch_next( 3835 // ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter, 3836 // kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper, 3837 // kmp_int[32|64] *p_stride); 3838 llvm::Value *Args[] = { 3839 emitUpdateLocation(CGF, Loc), 3840 getThreadID(CGF, Loc), 3841 IL.getPointer(), // &isLastIter 3842 LB.getPointer(), // &Lower 3843 UB.getPointer(), // &Upper 3844 ST.getPointer() // &Stride 3845 }; 3846 llvm::Value *Call = 3847 CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args); 3848 return CGF.EmitScalarConversion( 3849 Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1), 3850 CGF.getContext().BoolTy, Loc); 3851 } 3852 3853 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 3854 llvm::Value *NumThreads, 3855 SourceLocation Loc) { 3856 if (!CGF.HaveInsertPoint()) 3857 return; 3858 // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads) 3859 llvm::Value *Args[] = { 3860 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3861 CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)}; 3862 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads), 3863 Args); 3864 } 3865 3866 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF, 3867 ProcBindKind ProcBind, 3868 SourceLocation Loc) { 3869 if (!CGF.HaveInsertPoint()) 3870 return; 3871 assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value."); 3872 // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind) 3873 llvm::Value *Args[] = { 3874 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3875 llvm::ConstantInt::get(CGM.IntTy, unsigned(ProcBind), /*isSigned=*/true)}; 3876 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_proc_bind), Args); 3877 } 3878 3879 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>, 3880 SourceLocation Loc, llvm::AtomicOrdering AO) { 3881 llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder(); 3882 if (OMPBuilder) { 3883 OMPBuilder->CreateFlush(CGF.Builder); 3884 } else { 3885 if (!CGF.HaveInsertPoint()) 3886 return; 3887 // Build call void __kmpc_flush(ident_t *loc) 3888 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush), 3889 emitUpdateLocation(CGF, Loc)); 3890 } 3891 } 3892 3893 namespace { 3894 /// Indexes of fields for type kmp_task_t. 3895 enum KmpTaskTFields { 3896 /// List of shared variables. 3897 KmpTaskTShareds, 3898 /// Task routine. 3899 KmpTaskTRoutine, 3900 /// Partition id for the untied tasks. 3901 KmpTaskTPartId, 3902 /// Function with call of destructors for private variables. 3903 Data1, 3904 /// Task priority. 3905 Data2, 3906 /// (Taskloops only) Lower bound. 3907 KmpTaskTLowerBound, 3908 /// (Taskloops only) Upper bound. 3909 KmpTaskTUpperBound, 3910 /// (Taskloops only) Stride. 3911 KmpTaskTStride, 3912 /// (Taskloops only) Is last iteration flag. 3913 KmpTaskTLastIter, 3914 /// (Taskloops only) Reduction data. 3915 KmpTaskTReductions, 3916 }; 3917 } // anonymous namespace 3918 3919 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const { 3920 return OffloadEntriesTargetRegion.empty() && 3921 OffloadEntriesDeviceGlobalVar.empty(); 3922 } 3923 3924 /// Initialize target region entry. 3925 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3926 initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3927 StringRef ParentName, unsigned LineNum, 3928 unsigned Order) { 3929 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3930 "only required for the device " 3931 "code generation."); 3932 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = 3933 OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr, 3934 OMPTargetRegionEntryTargetRegion); 3935 ++OffloadingEntriesNum; 3936 } 3937 3938 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3939 registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3940 StringRef ParentName, unsigned LineNum, 3941 llvm::Constant *Addr, llvm::Constant *ID, 3942 OMPTargetRegionEntryKind Flags) { 3943 // If we are emitting code for a target, the entry is already initialized, 3944 // only has to be registered. 3945 if (CGM.getLangOpts().OpenMPIsDevice) { 3946 if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) { 3947 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3948 DiagnosticsEngine::Error, 3949 "Unable to find target region on line '%0' in the device code."); 3950 CGM.getDiags().Report(DiagID) << LineNum; 3951 return; 3952 } 3953 auto &Entry = 3954 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum]; 3955 assert(Entry.isValid() && "Entry not initialized!"); 3956 Entry.setAddress(Addr); 3957 Entry.setID(ID); 3958 Entry.setFlags(Flags); 3959 } else { 3960 OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags); 3961 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry; 3962 ++OffloadingEntriesNum; 3963 } 3964 } 3965 3966 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo( 3967 unsigned DeviceID, unsigned FileID, StringRef ParentName, 3968 unsigned LineNum) const { 3969 auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID); 3970 if (PerDevice == OffloadEntriesTargetRegion.end()) 3971 return false; 3972 auto PerFile = PerDevice->second.find(FileID); 3973 if (PerFile == PerDevice->second.end()) 3974 return false; 3975 auto PerParentName = PerFile->second.find(ParentName); 3976 if (PerParentName == PerFile->second.end()) 3977 return false; 3978 auto PerLine = PerParentName->second.find(LineNum); 3979 if (PerLine == PerParentName->second.end()) 3980 return false; 3981 // Fail if this entry is already registered. 3982 if (PerLine->second.getAddress() || PerLine->second.getID()) 3983 return false; 3984 return true; 3985 } 3986 3987 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo( 3988 const OffloadTargetRegionEntryInfoActTy &Action) { 3989 // Scan all target region entries and perform the provided action. 3990 for (const auto &D : OffloadEntriesTargetRegion) 3991 for (const auto &F : D.second) 3992 for (const auto &P : F.second) 3993 for (const auto &L : P.second) 3994 Action(D.first, F.first, P.first(), L.first, L.second); 3995 } 3996 3997 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3998 initializeDeviceGlobalVarEntryInfo(StringRef Name, 3999 OMPTargetGlobalVarEntryKind Flags, 4000 unsigned Order) { 4001 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 4002 "only required for the device " 4003 "code generation."); 4004 OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags); 4005 ++OffloadingEntriesNum; 4006 } 4007 4008 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 4009 registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr, 4010 CharUnits VarSize, 4011 OMPTargetGlobalVarEntryKind Flags, 4012 llvm::GlobalValue::LinkageTypes Linkage) { 4013 if (CGM.getLangOpts().OpenMPIsDevice) { 4014 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 4015 assert(Entry.isValid() && Entry.getFlags() == Flags && 4016 "Entry not initialized!"); 4017 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 4018 "Resetting with the new address."); 4019 if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) { 4020 if (Entry.getVarSize().isZero()) { 4021 Entry.setVarSize(VarSize); 4022 Entry.setLinkage(Linkage); 4023 } 4024 return; 4025 } 4026 Entry.setVarSize(VarSize); 4027 Entry.setLinkage(Linkage); 4028 Entry.setAddress(Addr); 4029 } else { 4030 if (hasDeviceGlobalVarEntryInfo(VarName)) { 4031 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 4032 assert(Entry.isValid() && Entry.getFlags() == Flags && 4033 "Entry not initialized!"); 4034 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 4035 "Resetting with the new address."); 4036 if (Entry.getVarSize().isZero()) { 4037 Entry.setVarSize(VarSize); 4038 Entry.setLinkage(Linkage); 4039 } 4040 return; 4041 } 4042 OffloadEntriesDeviceGlobalVar.try_emplace( 4043 VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage); 4044 ++OffloadingEntriesNum; 4045 } 4046 } 4047 4048 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 4049 actOnDeviceGlobalVarEntriesInfo( 4050 const OffloadDeviceGlobalVarEntryInfoActTy &Action) { 4051 // Scan all target region entries and perform the provided action. 4052 for (const auto &E : OffloadEntriesDeviceGlobalVar) 4053 Action(E.getKey(), E.getValue()); 4054 } 4055 4056 void CGOpenMPRuntime::createOffloadEntry( 4057 llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags, 4058 llvm::GlobalValue::LinkageTypes Linkage) { 4059 StringRef Name = Addr->getName(); 4060 llvm::Module &M = CGM.getModule(); 4061 llvm::LLVMContext &C = M.getContext(); 4062 4063 // Create constant string with the name. 4064 llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name); 4065 4066 std::string StringName = getName({"omp_offloading", "entry_name"}); 4067 auto *Str = new llvm::GlobalVariable( 4068 M, StrPtrInit->getType(), /*isConstant=*/true, 4069 llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName); 4070 Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 4071 4072 llvm::Constant *Data[] = {llvm::ConstantExpr::getBitCast(ID, CGM.VoidPtrTy), 4073 llvm::ConstantExpr::getBitCast(Str, CGM.Int8PtrTy), 4074 llvm::ConstantInt::get(CGM.SizeTy, Size), 4075 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 4076 llvm::ConstantInt::get(CGM.Int32Ty, 0)}; 4077 std::string EntryName = getName({"omp_offloading", "entry", ""}); 4078 llvm::GlobalVariable *Entry = createGlobalStruct( 4079 CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data, 4080 Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage); 4081 4082 // The entry has to be created in the section the linker expects it to be. 4083 Entry->setSection("omp_offloading_entries"); 4084 } 4085 4086 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() { 4087 // Emit the offloading entries and metadata so that the device codegen side 4088 // can easily figure out what to emit. The produced metadata looks like 4089 // this: 4090 // 4091 // !omp_offload.info = !{!1, ...} 4092 // 4093 // Right now we only generate metadata for function that contain target 4094 // regions. 4095 4096 // If we are in simd mode or there are no entries, we don't need to do 4097 // anything. 4098 if (CGM.getLangOpts().OpenMPSimd || OffloadEntriesInfoManager.empty()) 4099 return; 4100 4101 llvm::Module &M = CGM.getModule(); 4102 llvm::LLVMContext &C = M.getContext(); 4103 SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *, 4104 SourceLocation, StringRef>, 4105 16> 4106 OrderedEntries(OffloadEntriesInfoManager.size()); 4107 llvm::SmallVector<StringRef, 16> ParentFunctions( 4108 OffloadEntriesInfoManager.size()); 4109 4110 // Auxiliary methods to create metadata values and strings. 4111 auto &&GetMDInt = [this](unsigned V) { 4112 return llvm::ConstantAsMetadata::get( 4113 llvm::ConstantInt::get(CGM.Int32Ty, V)); 4114 }; 4115 4116 auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); }; 4117 4118 // Create the offloading info metadata node. 4119 llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info"); 4120 4121 // Create function that emits metadata for each target region entry; 4122 auto &&TargetRegionMetadataEmitter = 4123 [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt, 4124 &GetMDString]( 4125 unsigned DeviceID, unsigned FileID, StringRef ParentName, 4126 unsigned Line, 4127 const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) { 4128 // Generate metadata for target regions. Each entry of this metadata 4129 // contains: 4130 // - Entry 0 -> Kind of this type of metadata (0). 4131 // - Entry 1 -> Device ID of the file where the entry was identified. 4132 // - Entry 2 -> File ID of the file where the entry was identified. 4133 // - Entry 3 -> Mangled name of the function where the entry was 4134 // identified. 4135 // - Entry 4 -> Line in the file where the entry was identified. 4136 // - Entry 5 -> Order the entry was created. 4137 // The first element of the metadata node is the kind. 4138 llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID), 4139 GetMDInt(FileID), GetMDString(ParentName), 4140 GetMDInt(Line), GetMDInt(E.getOrder())}; 4141 4142 SourceLocation Loc; 4143 for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(), 4144 E = CGM.getContext().getSourceManager().fileinfo_end(); 4145 I != E; ++I) { 4146 if (I->getFirst()->getUniqueID().getDevice() == DeviceID && 4147 I->getFirst()->getUniqueID().getFile() == FileID) { 4148 Loc = CGM.getContext().getSourceManager().translateFileLineCol( 4149 I->getFirst(), Line, 1); 4150 break; 4151 } 4152 } 4153 // Save this entry in the right position of the ordered entries array. 4154 OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName); 4155 ParentFunctions[E.getOrder()] = ParentName; 4156 4157 // Add metadata to the named metadata node. 4158 MD->addOperand(llvm::MDNode::get(C, Ops)); 4159 }; 4160 4161 OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo( 4162 TargetRegionMetadataEmitter); 4163 4164 // Create function that emits metadata for each device global variable entry; 4165 auto &&DeviceGlobalVarMetadataEmitter = 4166 [&C, &OrderedEntries, &GetMDInt, &GetMDString, 4167 MD](StringRef MangledName, 4168 const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar 4169 &E) { 4170 // Generate metadata for global variables. Each entry of this metadata 4171 // contains: 4172 // - Entry 0 -> Kind of this type of metadata (1). 4173 // - Entry 1 -> Mangled name of the variable. 4174 // - Entry 2 -> Declare target kind. 4175 // - Entry 3 -> Order the entry was created. 4176 // The first element of the metadata node is the kind. 4177 llvm::Metadata *Ops[] = { 4178 GetMDInt(E.getKind()), GetMDString(MangledName), 4179 GetMDInt(E.getFlags()), GetMDInt(E.getOrder())}; 4180 4181 // Save this entry in the right position of the ordered entries array. 4182 OrderedEntries[E.getOrder()] = 4183 std::make_tuple(&E, SourceLocation(), MangledName); 4184 4185 // Add metadata to the named metadata node. 4186 MD->addOperand(llvm::MDNode::get(C, Ops)); 4187 }; 4188 4189 OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo( 4190 DeviceGlobalVarMetadataEmitter); 4191 4192 for (const auto &E : OrderedEntries) { 4193 assert(std::get<0>(E) && "All ordered entries must exist!"); 4194 if (const auto *CE = 4195 dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>( 4196 std::get<0>(E))) { 4197 if (!CE->getID() || !CE->getAddress()) { 4198 // Do not blame the entry if the parent funtion is not emitted. 4199 StringRef FnName = ParentFunctions[CE->getOrder()]; 4200 if (!CGM.GetGlobalValue(FnName)) 4201 continue; 4202 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4203 DiagnosticsEngine::Error, 4204 "Offloading entry for target region in %0 is incorrect: either the " 4205 "address or the ID is invalid."); 4206 CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName; 4207 continue; 4208 } 4209 createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0, 4210 CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage); 4211 } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy:: 4212 OffloadEntryInfoDeviceGlobalVar>( 4213 std::get<0>(E))) { 4214 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags = 4215 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 4216 CE->getFlags()); 4217 switch (Flags) { 4218 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: { 4219 if (CGM.getLangOpts().OpenMPIsDevice && 4220 CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory()) 4221 continue; 4222 if (!CE->getAddress()) { 4223 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4224 DiagnosticsEngine::Error, "Offloading entry for declare target " 4225 "variable %0 is incorrect: the " 4226 "address is invalid."); 4227 CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E); 4228 continue; 4229 } 4230 // The vaiable has no definition - no need to add the entry. 4231 if (CE->getVarSize().isZero()) 4232 continue; 4233 break; 4234 } 4235 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink: 4236 assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) || 4237 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) && 4238 "Declaret target link address is set."); 4239 if (CGM.getLangOpts().OpenMPIsDevice) 4240 continue; 4241 if (!CE->getAddress()) { 4242 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4243 DiagnosticsEngine::Error, 4244 "Offloading entry for declare target variable is incorrect: the " 4245 "address is invalid."); 4246 CGM.getDiags().Report(DiagID); 4247 continue; 4248 } 4249 break; 4250 } 4251 createOffloadEntry(CE->getAddress(), CE->getAddress(), 4252 CE->getVarSize().getQuantity(), Flags, 4253 CE->getLinkage()); 4254 } else { 4255 llvm_unreachable("Unsupported entry kind."); 4256 } 4257 } 4258 } 4259 4260 /// Loads all the offload entries information from the host IR 4261 /// metadata. 4262 void CGOpenMPRuntime::loadOffloadInfoMetadata() { 4263 // If we are in target mode, load the metadata from the host IR. This code has 4264 // to match the metadaata creation in createOffloadEntriesAndInfoMetadata(). 4265 4266 if (!CGM.getLangOpts().OpenMPIsDevice) 4267 return; 4268 4269 if (CGM.getLangOpts().OMPHostIRFile.empty()) 4270 return; 4271 4272 auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile); 4273 if (auto EC = Buf.getError()) { 4274 CGM.getDiags().Report(diag::err_cannot_open_file) 4275 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 4276 return; 4277 } 4278 4279 llvm::LLVMContext C; 4280 auto ME = expectedToErrorOrAndEmitErrors( 4281 C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C)); 4282 4283 if (auto EC = ME.getError()) { 4284 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4285 DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'"); 4286 CGM.getDiags().Report(DiagID) 4287 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 4288 return; 4289 } 4290 4291 llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info"); 4292 if (!MD) 4293 return; 4294 4295 for (llvm::MDNode *MN : MD->operands()) { 4296 auto &&GetMDInt = [MN](unsigned Idx) { 4297 auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx)); 4298 return cast<llvm::ConstantInt>(V->getValue())->getZExtValue(); 4299 }; 4300 4301 auto &&GetMDString = [MN](unsigned Idx) { 4302 auto *V = cast<llvm::MDString>(MN->getOperand(Idx)); 4303 return V->getString(); 4304 }; 4305 4306 switch (GetMDInt(0)) { 4307 default: 4308 llvm_unreachable("Unexpected metadata!"); 4309 break; 4310 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 4311 OffloadingEntryInfoTargetRegion: 4312 OffloadEntriesInfoManager.initializeTargetRegionEntryInfo( 4313 /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2), 4314 /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4), 4315 /*Order=*/GetMDInt(5)); 4316 break; 4317 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 4318 OffloadingEntryInfoDeviceGlobalVar: 4319 OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo( 4320 /*MangledName=*/GetMDString(1), 4321 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 4322 /*Flags=*/GetMDInt(2)), 4323 /*Order=*/GetMDInt(3)); 4324 break; 4325 } 4326 } 4327 } 4328 4329 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) { 4330 if (!KmpRoutineEntryPtrTy) { 4331 // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type. 4332 ASTContext &C = CGM.getContext(); 4333 QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy}; 4334 FunctionProtoType::ExtProtoInfo EPI; 4335 KmpRoutineEntryPtrQTy = C.getPointerType( 4336 C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI)); 4337 KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy); 4338 } 4339 } 4340 4341 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() { 4342 // Make sure the type of the entry is already created. This is the type we 4343 // have to create: 4344 // struct __tgt_offload_entry{ 4345 // void *addr; // Pointer to the offload entry info. 4346 // // (function or global) 4347 // char *name; // Name of the function or global. 4348 // size_t size; // Size of the entry info (0 if it a function). 4349 // int32_t flags; // Flags associated with the entry, e.g. 'link'. 4350 // int32_t reserved; // Reserved, to use by the runtime library. 4351 // }; 4352 if (TgtOffloadEntryQTy.isNull()) { 4353 ASTContext &C = CGM.getContext(); 4354 RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry"); 4355 RD->startDefinition(); 4356 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4357 addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy)); 4358 addFieldToRecordDecl(C, RD, C.getSizeType()); 4359 addFieldToRecordDecl( 4360 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4361 addFieldToRecordDecl( 4362 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4363 RD->completeDefinition(); 4364 RD->addAttr(PackedAttr::CreateImplicit(C)); 4365 TgtOffloadEntryQTy = C.getRecordType(RD); 4366 } 4367 return TgtOffloadEntryQTy; 4368 } 4369 4370 namespace { 4371 struct PrivateHelpersTy { 4372 PrivateHelpersTy(const VarDecl *Original, const VarDecl *PrivateCopy, 4373 const VarDecl *PrivateElemInit) 4374 : Original(Original), PrivateCopy(PrivateCopy), 4375 PrivateElemInit(PrivateElemInit) {} 4376 const VarDecl *Original; 4377 const VarDecl *PrivateCopy; 4378 const VarDecl *PrivateElemInit; 4379 }; 4380 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy; 4381 } // anonymous namespace 4382 4383 static RecordDecl * 4384 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) { 4385 if (!Privates.empty()) { 4386 ASTContext &C = CGM.getContext(); 4387 // Build struct .kmp_privates_t. { 4388 // /* private vars */ 4389 // }; 4390 RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t"); 4391 RD->startDefinition(); 4392 for (const auto &Pair : Privates) { 4393 const VarDecl *VD = Pair.second.Original; 4394 QualType Type = VD->getType().getNonReferenceType(); 4395 FieldDecl *FD = addFieldToRecordDecl(C, RD, Type); 4396 if (VD->hasAttrs()) { 4397 for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()), 4398 E(VD->getAttrs().end()); 4399 I != E; ++I) 4400 FD->addAttr(*I); 4401 } 4402 } 4403 RD->completeDefinition(); 4404 return RD; 4405 } 4406 return nullptr; 4407 } 4408 4409 static RecordDecl * 4410 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind, 4411 QualType KmpInt32Ty, 4412 QualType KmpRoutineEntryPointerQTy) { 4413 ASTContext &C = CGM.getContext(); 4414 // Build struct kmp_task_t { 4415 // void * shareds; 4416 // kmp_routine_entry_t routine; 4417 // kmp_int32 part_id; 4418 // kmp_cmplrdata_t data1; 4419 // kmp_cmplrdata_t data2; 4420 // For taskloops additional fields: 4421 // kmp_uint64 lb; 4422 // kmp_uint64 ub; 4423 // kmp_int64 st; 4424 // kmp_int32 liter; 4425 // void * reductions; 4426 // }; 4427 RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union); 4428 UD->startDefinition(); 4429 addFieldToRecordDecl(C, UD, KmpInt32Ty); 4430 addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy); 4431 UD->completeDefinition(); 4432 QualType KmpCmplrdataTy = C.getRecordType(UD); 4433 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t"); 4434 RD->startDefinition(); 4435 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4436 addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy); 4437 addFieldToRecordDecl(C, RD, KmpInt32Ty); 4438 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 4439 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 4440 if (isOpenMPTaskLoopDirective(Kind)) { 4441 QualType KmpUInt64Ty = 4442 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0); 4443 QualType KmpInt64Ty = 4444 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 4445 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 4446 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 4447 addFieldToRecordDecl(C, RD, KmpInt64Ty); 4448 addFieldToRecordDecl(C, RD, KmpInt32Ty); 4449 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4450 } 4451 RD->completeDefinition(); 4452 return RD; 4453 } 4454 4455 static RecordDecl * 4456 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy, 4457 ArrayRef<PrivateDataTy> Privates) { 4458 ASTContext &C = CGM.getContext(); 4459 // Build struct kmp_task_t_with_privates { 4460 // kmp_task_t task_data; 4461 // .kmp_privates_t. privates; 4462 // }; 4463 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates"); 4464 RD->startDefinition(); 4465 addFieldToRecordDecl(C, RD, KmpTaskTQTy); 4466 if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates)) 4467 addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD)); 4468 RD->completeDefinition(); 4469 return RD; 4470 } 4471 4472 /// Emit a proxy function which accepts kmp_task_t as the second 4473 /// argument. 4474 /// \code 4475 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) { 4476 /// TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt, 4477 /// For taskloops: 4478 /// tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 4479 /// tt->reductions, tt->shareds); 4480 /// return 0; 4481 /// } 4482 /// \endcode 4483 static llvm::Function * 4484 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc, 4485 OpenMPDirectiveKind Kind, QualType KmpInt32Ty, 4486 QualType KmpTaskTWithPrivatesPtrQTy, 4487 QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy, 4488 QualType SharedsPtrTy, llvm::Function *TaskFunction, 4489 llvm::Value *TaskPrivatesMap) { 4490 ASTContext &C = CGM.getContext(); 4491 FunctionArgList Args; 4492 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 4493 ImplicitParamDecl::Other); 4494 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4495 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 4496 ImplicitParamDecl::Other); 4497 Args.push_back(&GtidArg); 4498 Args.push_back(&TaskTypeArg); 4499 const auto &TaskEntryFnInfo = 4500 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 4501 llvm::FunctionType *TaskEntryTy = 4502 CGM.getTypes().GetFunctionType(TaskEntryFnInfo); 4503 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""}); 4504 auto *TaskEntry = llvm::Function::Create( 4505 TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 4506 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo); 4507 TaskEntry->setDoesNotRecurse(); 4508 CodeGenFunction CGF(CGM); 4509 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args, 4510 Loc, Loc); 4511 4512 // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map, 4513 // tt, 4514 // For taskloops: 4515 // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 4516 // tt->task_data.shareds); 4517 llvm::Value *GtidParam = CGF.EmitLoadOfScalar( 4518 CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc); 4519 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4520 CGF.GetAddrOfLocalVar(&TaskTypeArg), 4521 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4522 const auto *KmpTaskTWithPrivatesQTyRD = 4523 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 4524 LValue Base = 4525 CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4526 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 4527 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 4528 LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI); 4529 llvm::Value *PartidParam = PartIdLVal.getPointer(CGF); 4530 4531 auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds); 4532 LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI); 4533 llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4534 CGF.EmitLoadOfScalar(SharedsLVal, Loc), 4535 CGF.ConvertTypeForMem(SharedsPtrTy)); 4536 4537 auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 4538 llvm::Value *PrivatesParam; 4539 if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) { 4540 LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI); 4541 PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4542 PrivatesLVal.getPointer(CGF), CGF.VoidPtrTy); 4543 } else { 4544 PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 4545 } 4546 4547 llvm::Value *CommonArgs[] = {GtidParam, PartidParam, PrivatesParam, 4548 TaskPrivatesMap, 4549 CGF.Builder 4550 .CreatePointerBitCastOrAddrSpaceCast( 4551 TDBase.getAddress(CGF), CGF.VoidPtrTy) 4552 .getPointer()}; 4553 SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs), 4554 std::end(CommonArgs)); 4555 if (isOpenMPTaskLoopDirective(Kind)) { 4556 auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound); 4557 LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI); 4558 llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc); 4559 auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound); 4560 LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI); 4561 llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc); 4562 auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride); 4563 LValue StLVal = CGF.EmitLValueForField(Base, *StFI); 4564 llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc); 4565 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 4566 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 4567 llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc); 4568 auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions); 4569 LValue RLVal = CGF.EmitLValueForField(Base, *RFI); 4570 llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc); 4571 CallArgs.push_back(LBParam); 4572 CallArgs.push_back(UBParam); 4573 CallArgs.push_back(StParam); 4574 CallArgs.push_back(LIParam); 4575 CallArgs.push_back(RParam); 4576 } 4577 CallArgs.push_back(SharedsParam); 4578 4579 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction, 4580 CallArgs); 4581 CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)), 4582 CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty)); 4583 CGF.FinishFunction(); 4584 return TaskEntry; 4585 } 4586 4587 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM, 4588 SourceLocation Loc, 4589 QualType KmpInt32Ty, 4590 QualType KmpTaskTWithPrivatesPtrQTy, 4591 QualType KmpTaskTWithPrivatesQTy) { 4592 ASTContext &C = CGM.getContext(); 4593 FunctionArgList Args; 4594 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 4595 ImplicitParamDecl::Other); 4596 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4597 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 4598 ImplicitParamDecl::Other); 4599 Args.push_back(&GtidArg); 4600 Args.push_back(&TaskTypeArg); 4601 const auto &DestructorFnInfo = 4602 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 4603 llvm::FunctionType *DestructorFnTy = 4604 CGM.getTypes().GetFunctionType(DestructorFnInfo); 4605 std::string Name = 4606 CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""}); 4607 auto *DestructorFn = 4608 llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage, 4609 Name, &CGM.getModule()); 4610 CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn, 4611 DestructorFnInfo); 4612 DestructorFn->setDoesNotRecurse(); 4613 CodeGenFunction CGF(CGM); 4614 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo, 4615 Args, Loc, Loc); 4616 4617 LValue Base = CGF.EmitLoadOfPointerLValue( 4618 CGF.GetAddrOfLocalVar(&TaskTypeArg), 4619 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4620 const auto *KmpTaskTWithPrivatesQTyRD = 4621 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 4622 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4623 Base = CGF.EmitLValueForField(Base, *FI); 4624 for (const auto *Field : 4625 cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) { 4626 if (QualType::DestructionKind DtorKind = 4627 Field->getType().isDestructedType()) { 4628 LValue FieldLValue = CGF.EmitLValueForField(Base, Field); 4629 CGF.pushDestroy(DtorKind, FieldLValue.getAddress(CGF), Field->getType()); 4630 } 4631 } 4632 CGF.FinishFunction(); 4633 return DestructorFn; 4634 } 4635 4636 /// Emit a privates mapping function for correct handling of private and 4637 /// firstprivate variables. 4638 /// \code 4639 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1> 4640 /// **noalias priv1,..., <tyn> **noalias privn) { 4641 /// *priv1 = &.privates.priv1; 4642 /// ...; 4643 /// *privn = &.privates.privn; 4644 /// } 4645 /// \endcode 4646 static llvm::Value * 4647 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc, 4648 ArrayRef<const Expr *> PrivateVars, 4649 ArrayRef<const Expr *> FirstprivateVars, 4650 ArrayRef<const Expr *> LastprivateVars, 4651 QualType PrivatesQTy, 4652 ArrayRef<PrivateDataTy> Privates) { 4653 ASTContext &C = CGM.getContext(); 4654 FunctionArgList Args; 4655 ImplicitParamDecl TaskPrivatesArg( 4656 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4657 C.getPointerType(PrivatesQTy).withConst().withRestrict(), 4658 ImplicitParamDecl::Other); 4659 Args.push_back(&TaskPrivatesArg); 4660 llvm::DenseMap<const VarDecl *, unsigned> PrivateVarsPos; 4661 unsigned Counter = 1; 4662 for (const Expr *E : PrivateVars) { 4663 Args.push_back(ImplicitParamDecl::Create( 4664 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4665 C.getPointerType(C.getPointerType(E->getType())) 4666 .withConst() 4667 .withRestrict(), 4668 ImplicitParamDecl::Other)); 4669 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4670 PrivateVarsPos[VD] = Counter; 4671 ++Counter; 4672 } 4673 for (const Expr *E : FirstprivateVars) { 4674 Args.push_back(ImplicitParamDecl::Create( 4675 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4676 C.getPointerType(C.getPointerType(E->getType())) 4677 .withConst() 4678 .withRestrict(), 4679 ImplicitParamDecl::Other)); 4680 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4681 PrivateVarsPos[VD] = Counter; 4682 ++Counter; 4683 } 4684 for (const Expr *E : LastprivateVars) { 4685 Args.push_back(ImplicitParamDecl::Create( 4686 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4687 C.getPointerType(C.getPointerType(E->getType())) 4688 .withConst() 4689 .withRestrict(), 4690 ImplicitParamDecl::Other)); 4691 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4692 PrivateVarsPos[VD] = Counter; 4693 ++Counter; 4694 } 4695 const auto &TaskPrivatesMapFnInfo = 4696 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4697 llvm::FunctionType *TaskPrivatesMapTy = 4698 CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo); 4699 std::string Name = 4700 CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""}); 4701 auto *TaskPrivatesMap = llvm::Function::Create( 4702 TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name, 4703 &CGM.getModule()); 4704 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap, 4705 TaskPrivatesMapFnInfo); 4706 if (CGM.getLangOpts().Optimize) { 4707 TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline); 4708 TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone); 4709 TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline); 4710 } 4711 CodeGenFunction CGF(CGM); 4712 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap, 4713 TaskPrivatesMapFnInfo, Args, Loc, Loc); 4714 4715 // *privi = &.privates.privi; 4716 LValue Base = CGF.EmitLoadOfPointerLValue( 4717 CGF.GetAddrOfLocalVar(&TaskPrivatesArg), 4718 TaskPrivatesArg.getType()->castAs<PointerType>()); 4719 const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl()); 4720 Counter = 0; 4721 for (const FieldDecl *Field : PrivatesQTyRD->fields()) { 4722 LValue FieldLVal = CGF.EmitLValueForField(Base, Field); 4723 const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]]; 4724 LValue RefLVal = 4725 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType()); 4726 LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue( 4727 RefLVal.getAddress(CGF), RefLVal.getType()->castAs<PointerType>()); 4728 CGF.EmitStoreOfScalar(FieldLVal.getPointer(CGF), RefLoadLVal); 4729 ++Counter; 4730 } 4731 CGF.FinishFunction(); 4732 return TaskPrivatesMap; 4733 } 4734 4735 /// Emit initialization for private variables in task-based directives. 4736 static void emitPrivatesInit(CodeGenFunction &CGF, 4737 const OMPExecutableDirective &D, 4738 Address KmpTaskSharedsPtr, LValue TDBase, 4739 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 4740 QualType SharedsTy, QualType SharedsPtrTy, 4741 const OMPTaskDataTy &Data, 4742 ArrayRef<PrivateDataTy> Privates, bool ForDup) { 4743 ASTContext &C = CGF.getContext(); 4744 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4745 LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI); 4746 OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind()) 4747 ? OMPD_taskloop 4748 : OMPD_task; 4749 const CapturedStmt &CS = *D.getCapturedStmt(Kind); 4750 CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS); 4751 LValue SrcBase; 4752 bool IsTargetTask = 4753 isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) || 4754 isOpenMPTargetExecutionDirective(D.getDirectiveKind()); 4755 // For target-based directives skip 3 firstprivate arrays BasePointersArray, 4756 // PointersArray and SizesArray. The original variables for these arrays are 4757 // not captured and we get their addresses explicitly. 4758 if ((!IsTargetTask && !Data.FirstprivateVars.empty()) || 4759 (IsTargetTask && KmpTaskSharedsPtr.isValid())) { 4760 SrcBase = CGF.MakeAddrLValue( 4761 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4762 KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)), 4763 SharedsTy); 4764 } 4765 FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin(); 4766 for (const PrivateDataTy &Pair : Privates) { 4767 const VarDecl *VD = Pair.second.PrivateCopy; 4768 const Expr *Init = VD->getAnyInitializer(); 4769 if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) && 4770 !CGF.isTrivialInitializer(Init)))) { 4771 LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI); 4772 if (const VarDecl *Elem = Pair.second.PrivateElemInit) { 4773 const VarDecl *OriginalVD = Pair.second.Original; 4774 // Check if the variable is the target-based BasePointersArray, 4775 // PointersArray or SizesArray. 4776 LValue SharedRefLValue; 4777 QualType Type = PrivateLValue.getType(); 4778 const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD); 4779 if (IsTargetTask && !SharedField) { 4780 assert(isa<ImplicitParamDecl>(OriginalVD) && 4781 isa<CapturedDecl>(OriginalVD->getDeclContext()) && 4782 cast<CapturedDecl>(OriginalVD->getDeclContext()) 4783 ->getNumParams() == 0 && 4784 isa<TranslationUnitDecl>( 4785 cast<CapturedDecl>(OriginalVD->getDeclContext()) 4786 ->getDeclContext()) && 4787 "Expected artificial target data variable."); 4788 SharedRefLValue = 4789 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type); 4790 } else { 4791 SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField); 4792 SharedRefLValue = CGF.MakeAddrLValue( 4793 Address(SharedRefLValue.getPointer(CGF), 4794 C.getDeclAlign(OriginalVD)), 4795 SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl), 4796 SharedRefLValue.getTBAAInfo()); 4797 } 4798 if (Type->isArrayType()) { 4799 // Initialize firstprivate array. 4800 if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) { 4801 // Perform simple memcpy. 4802 CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type); 4803 } else { 4804 // Initialize firstprivate array using element-by-element 4805 // initialization. 4806 CGF.EmitOMPAggregateAssign( 4807 PrivateLValue.getAddress(CGF), SharedRefLValue.getAddress(CGF), 4808 Type, 4809 [&CGF, Elem, Init, &CapturesInfo](Address DestElement, 4810 Address SrcElement) { 4811 // Clean up any temporaries needed by the initialization. 4812 CodeGenFunction::OMPPrivateScope InitScope(CGF); 4813 InitScope.addPrivate( 4814 Elem, [SrcElement]() -> Address { return SrcElement; }); 4815 (void)InitScope.Privatize(); 4816 // Emit initialization for single element. 4817 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII( 4818 CGF, &CapturesInfo); 4819 CGF.EmitAnyExprToMem(Init, DestElement, 4820 Init->getType().getQualifiers(), 4821 /*IsInitializer=*/false); 4822 }); 4823 } 4824 } else { 4825 CodeGenFunction::OMPPrivateScope InitScope(CGF); 4826 InitScope.addPrivate(Elem, [SharedRefLValue, &CGF]() -> Address { 4827 return SharedRefLValue.getAddress(CGF); 4828 }); 4829 (void)InitScope.Privatize(); 4830 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo); 4831 CGF.EmitExprAsInit(Init, VD, PrivateLValue, 4832 /*capturedByInit=*/false); 4833 } 4834 } else { 4835 CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false); 4836 } 4837 } 4838 ++FI; 4839 } 4840 } 4841 4842 /// Check if duplication function is required for taskloops. 4843 static bool checkInitIsRequired(CodeGenFunction &CGF, 4844 ArrayRef<PrivateDataTy> Privates) { 4845 bool InitRequired = false; 4846 for (const PrivateDataTy &Pair : Privates) { 4847 const VarDecl *VD = Pair.second.PrivateCopy; 4848 const Expr *Init = VD->getAnyInitializer(); 4849 InitRequired = InitRequired || (Init && isa<CXXConstructExpr>(Init) && 4850 !CGF.isTrivialInitializer(Init)); 4851 if (InitRequired) 4852 break; 4853 } 4854 return InitRequired; 4855 } 4856 4857 4858 /// Emit task_dup function (for initialization of 4859 /// private/firstprivate/lastprivate vars and last_iter flag) 4860 /// \code 4861 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int 4862 /// lastpriv) { 4863 /// // setup lastprivate flag 4864 /// task_dst->last = lastpriv; 4865 /// // could be constructor calls here... 4866 /// } 4867 /// \endcode 4868 static llvm::Value * 4869 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc, 4870 const OMPExecutableDirective &D, 4871 QualType KmpTaskTWithPrivatesPtrQTy, 4872 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 4873 const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy, 4874 QualType SharedsPtrTy, const OMPTaskDataTy &Data, 4875 ArrayRef<PrivateDataTy> Privates, bool WithLastIter) { 4876 ASTContext &C = CGM.getContext(); 4877 FunctionArgList Args; 4878 ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4879 KmpTaskTWithPrivatesPtrQTy, 4880 ImplicitParamDecl::Other); 4881 ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4882 KmpTaskTWithPrivatesPtrQTy, 4883 ImplicitParamDecl::Other); 4884 ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy, 4885 ImplicitParamDecl::Other); 4886 Args.push_back(&DstArg); 4887 Args.push_back(&SrcArg); 4888 Args.push_back(&LastprivArg); 4889 const auto &TaskDupFnInfo = 4890 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4891 llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo); 4892 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""}); 4893 auto *TaskDup = llvm::Function::Create( 4894 TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 4895 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo); 4896 TaskDup->setDoesNotRecurse(); 4897 CodeGenFunction CGF(CGM); 4898 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc, 4899 Loc); 4900 4901 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4902 CGF.GetAddrOfLocalVar(&DstArg), 4903 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4904 // task_dst->liter = lastpriv; 4905 if (WithLastIter) { 4906 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 4907 LValue Base = CGF.EmitLValueForField( 4908 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4909 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 4910 llvm::Value *Lastpriv = CGF.EmitLoadOfScalar( 4911 CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc); 4912 CGF.EmitStoreOfScalar(Lastpriv, LILVal); 4913 } 4914 4915 // Emit initial values for private copies (if any). 4916 assert(!Privates.empty()); 4917 Address KmpTaskSharedsPtr = Address::invalid(); 4918 if (!Data.FirstprivateVars.empty()) { 4919 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4920 CGF.GetAddrOfLocalVar(&SrcArg), 4921 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4922 LValue Base = CGF.EmitLValueForField( 4923 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4924 KmpTaskSharedsPtr = Address( 4925 CGF.EmitLoadOfScalar(CGF.EmitLValueForField( 4926 Base, *std::next(KmpTaskTQTyRD->field_begin(), 4927 KmpTaskTShareds)), 4928 Loc), 4929 CGF.getNaturalTypeAlignment(SharedsTy)); 4930 } 4931 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD, 4932 SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true); 4933 CGF.FinishFunction(); 4934 return TaskDup; 4935 } 4936 4937 /// Checks if destructor function is required to be generated. 4938 /// \return true if cleanups are required, false otherwise. 4939 static bool 4940 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD) { 4941 bool NeedsCleanup = false; 4942 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 4943 const auto *PrivateRD = cast<RecordDecl>(FI->getType()->getAsTagDecl()); 4944 for (const FieldDecl *FD : PrivateRD->fields()) { 4945 NeedsCleanup = NeedsCleanup || FD->getType().isDestructedType(); 4946 if (NeedsCleanup) 4947 break; 4948 } 4949 return NeedsCleanup; 4950 } 4951 4952 CGOpenMPRuntime::TaskResultTy 4953 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc, 4954 const OMPExecutableDirective &D, 4955 llvm::Function *TaskFunction, QualType SharedsTy, 4956 Address Shareds, const OMPTaskDataTy &Data) { 4957 ASTContext &C = CGM.getContext(); 4958 llvm::SmallVector<PrivateDataTy, 4> Privates; 4959 // Aggregate privates and sort them by the alignment. 4960 auto I = Data.PrivateCopies.begin(); 4961 for (const Expr *E : Data.PrivateVars) { 4962 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4963 Privates.emplace_back( 4964 C.getDeclAlign(VD), 4965 PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4966 /*PrivateElemInit=*/nullptr)); 4967 ++I; 4968 } 4969 I = Data.FirstprivateCopies.begin(); 4970 auto IElemInitRef = Data.FirstprivateInits.begin(); 4971 for (const Expr *E : Data.FirstprivateVars) { 4972 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4973 Privates.emplace_back( 4974 C.getDeclAlign(VD), 4975 PrivateHelpersTy( 4976 VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4977 cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl()))); 4978 ++I; 4979 ++IElemInitRef; 4980 } 4981 I = Data.LastprivateCopies.begin(); 4982 for (const Expr *E : Data.LastprivateVars) { 4983 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4984 Privates.emplace_back( 4985 C.getDeclAlign(VD), 4986 PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4987 /*PrivateElemInit=*/nullptr)); 4988 ++I; 4989 } 4990 llvm::stable_sort(Privates, [](PrivateDataTy L, PrivateDataTy R) { 4991 return L.first > R.first; 4992 }); 4993 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 4994 // Build type kmp_routine_entry_t (if not built yet). 4995 emitKmpRoutineEntryT(KmpInt32Ty); 4996 // Build type kmp_task_t (if not built yet). 4997 if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) { 4998 if (SavedKmpTaskloopTQTy.isNull()) { 4999 SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl( 5000 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 5001 } 5002 KmpTaskTQTy = SavedKmpTaskloopTQTy; 5003 } else { 5004 assert((D.getDirectiveKind() == OMPD_task || 5005 isOpenMPTargetExecutionDirective(D.getDirectiveKind()) || 5006 isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) && 5007 "Expected taskloop, task or target directive"); 5008 if (SavedKmpTaskTQTy.isNull()) { 5009 SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl( 5010 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 5011 } 5012 KmpTaskTQTy = SavedKmpTaskTQTy; 5013 } 5014 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 5015 // Build particular struct kmp_task_t for the given task. 5016 const RecordDecl *KmpTaskTWithPrivatesQTyRD = 5017 createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates); 5018 QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD); 5019 QualType KmpTaskTWithPrivatesPtrQTy = 5020 C.getPointerType(KmpTaskTWithPrivatesQTy); 5021 llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy); 5022 llvm::Type *KmpTaskTWithPrivatesPtrTy = 5023 KmpTaskTWithPrivatesTy->getPointerTo(); 5024 llvm::Value *KmpTaskTWithPrivatesTySize = 5025 CGF.getTypeSize(KmpTaskTWithPrivatesQTy); 5026 QualType SharedsPtrTy = C.getPointerType(SharedsTy); 5027 5028 // Emit initial values for private copies (if any). 5029 llvm::Value *TaskPrivatesMap = nullptr; 5030 llvm::Type *TaskPrivatesMapTy = 5031 std::next(TaskFunction->arg_begin(), 3)->getType(); 5032 if (!Privates.empty()) { 5033 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 5034 TaskPrivatesMap = emitTaskPrivateMappingFunction( 5035 CGM, Loc, Data.PrivateVars, Data.FirstprivateVars, Data.LastprivateVars, 5036 FI->getType(), Privates); 5037 TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5038 TaskPrivatesMap, TaskPrivatesMapTy); 5039 } else { 5040 TaskPrivatesMap = llvm::ConstantPointerNull::get( 5041 cast<llvm::PointerType>(TaskPrivatesMapTy)); 5042 } 5043 // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid, 5044 // kmp_task_t *tt); 5045 llvm::Function *TaskEntry = emitProxyTaskFunction( 5046 CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 5047 KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction, 5048 TaskPrivatesMap); 5049 5050 // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 5051 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 5052 // kmp_routine_entry_t *task_entry); 5053 // Task flags. Format is taken from 5054 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h, 5055 // description of kmp_tasking_flags struct. 5056 enum { 5057 TiedFlag = 0x1, 5058 FinalFlag = 0x2, 5059 DestructorsFlag = 0x8, 5060 PriorityFlag = 0x20 5061 }; 5062 unsigned Flags = Data.Tied ? TiedFlag : 0; 5063 bool NeedsCleanup = false; 5064 if (!Privates.empty()) { 5065 NeedsCleanup = checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD); 5066 if (NeedsCleanup) 5067 Flags = Flags | DestructorsFlag; 5068 } 5069 if (Data.Priority.getInt()) 5070 Flags = Flags | PriorityFlag; 5071 llvm::Value *TaskFlags = 5072 Data.Final.getPointer() 5073 ? CGF.Builder.CreateSelect(Data.Final.getPointer(), 5074 CGF.Builder.getInt32(FinalFlag), 5075 CGF.Builder.getInt32(/*C=*/0)) 5076 : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0); 5077 TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags)); 5078 llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy)); 5079 SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc), 5080 getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize, 5081 SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5082 TaskEntry, KmpRoutineEntryPtrTy)}; 5083 llvm::Value *NewTask; 5084 if (D.hasClausesOfKind<OMPNowaitClause>()) { 5085 // Check if we have any device clause associated with the directive. 5086 const Expr *Device = nullptr; 5087 if (auto *C = D.getSingleClause<OMPDeviceClause>()) 5088 Device = C->getDevice(); 5089 // Emit device ID if any otherwise use default value. 5090 llvm::Value *DeviceID; 5091 if (Device) 5092 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 5093 CGF.Int64Ty, /*isSigned=*/true); 5094 else 5095 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 5096 AllocArgs.push_back(DeviceID); 5097 NewTask = CGF.EmitRuntimeCall( 5098 createRuntimeFunction(OMPRTL__kmpc_omp_target_task_alloc), AllocArgs); 5099 } else { 5100 NewTask = CGF.EmitRuntimeCall( 5101 createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs); 5102 } 5103 llvm::Value *NewTaskNewTaskTTy = 5104 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5105 NewTask, KmpTaskTWithPrivatesPtrTy); 5106 LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy, 5107 KmpTaskTWithPrivatesQTy); 5108 LValue TDBase = 5109 CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin()); 5110 // Fill the data in the resulting kmp_task_t record. 5111 // Copy shareds if there are any. 5112 Address KmpTaskSharedsPtr = Address::invalid(); 5113 if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) { 5114 KmpTaskSharedsPtr = 5115 Address(CGF.EmitLoadOfScalar( 5116 CGF.EmitLValueForField( 5117 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), 5118 KmpTaskTShareds)), 5119 Loc), 5120 CGF.getNaturalTypeAlignment(SharedsTy)); 5121 LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy); 5122 LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy); 5123 CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap); 5124 } 5125 // Emit initial values for private copies (if any). 5126 TaskResultTy Result; 5127 if (!Privates.empty()) { 5128 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD, 5129 SharedsTy, SharedsPtrTy, Data, Privates, 5130 /*ForDup=*/false); 5131 if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) && 5132 (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) { 5133 Result.TaskDupFn = emitTaskDupFunction( 5134 CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD, 5135 KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates, 5136 /*WithLastIter=*/!Data.LastprivateVars.empty()); 5137 } 5138 } 5139 // Fields of union "kmp_cmplrdata_t" for destructors and priority. 5140 enum { Priority = 0, Destructors = 1 }; 5141 // Provide pointer to function with destructors for privates. 5142 auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1); 5143 const RecordDecl *KmpCmplrdataUD = 5144 (*FI)->getType()->getAsUnionType()->getDecl(); 5145 if (NeedsCleanup) { 5146 llvm::Value *DestructorFn = emitDestructorsFunction( 5147 CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 5148 KmpTaskTWithPrivatesQTy); 5149 LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI); 5150 LValue DestructorsLV = CGF.EmitLValueForField( 5151 Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors)); 5152 CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5153 DestructorFn, KmpRoutineEntryPtrTy), 5154 DestructorsLV); 5155 } 5156 // Set priority. 5157 if (Data.Priority.getInt()) { 5158 LValue Data2LV = CGF.EmitLValueForField( 5159 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2)); 5160 LValue PriorityLV = CGF.EmitLValueForField( 5161 Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority)); 5162 CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV); 5163 } 5164 Result.NewTask = NewTask; 5165 Result.TaskEntry = TaskEntry; 5166 Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy; 5167 Result.TDBase = TDBase; 5168 Result.KmpTaskTQTyRD = KmpTaskTQTyRD; 5169 return Result; 5170 } 5171 5172 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 5173 const OMPExecutableDirective &D, 5174 llvm::Function *TaskFunction, 5175 QualType SharedsTy, Address Shareds, 5176 const Expr *IfCond, 5177 const OMPTaskDataTy &Data) { 5178 if (!CGF.HaveInsertPoint()) 5179 return; 5180 5181 TaskResultTy Result = 5182 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5183 llvm::Value *NewTask = Result.NewTask; 5184 llvm::Function *TaskEntry = Result.TaskEntry; 5185 llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy; 5186 LValue TDBase = Result.TDBase; 5187 const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD; 5188 ASTContext &C = CGM.getContext(); 5189 // Process list of dependences. 5190 Address DependenciesArray = Address::invalid(); 5191 unsigned NumDependencies = Data.Dependences.size(); 5192 if (NumDependencies) { 5193 // Dependence kind for RTL. 5194 enum RTLDependenceKindTy { DepIn = 0x01, DepInOut = 0x3, DepMutexInOutSet = 0x4 }; 5195 enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags }; 5196 RecordDecl *KmpDependInfoRD; 5197 QualType FlagsTy = 5198 C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false); 5199 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 5200 if (KmpDependInfoTy.isNull()) { 5201 KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info"); 5202 KmpDependInfoRD->startDefinition(); 5203 addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType()); 5204 addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType()); 5205 addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy); 5206 KmpDependInfoRD->completeDefinition(); 5207 KmpDependInfoTy = C.getRecordType(KmpDependInfoRD); 5208 } else { 5209 KmpDependInfoRD = cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5210 } 5211 // Define type kmp_depend_info[<Dependences.size()>]; 5212 QualType KmpDependInfoArrayTy = C.getConstantArrayType( 5213 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies), 5214 nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 5215 // kmp_depend_info[<Dependences.size()>] deps; 5216 DependenciesArray = 5217 CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr"); 5218 for (unsigned I = 0; I < NumDependencies; ++I) { 5219 const Expr *E = Data.Dependences[I].second; 5220 LValue Addr = CGF.EmitLValue(E); 5221 llvm::Value *Size; 5222 QualType Ty = E->getType(); 5223 if (const auto *ASE = 5224 dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) { 5225 LValue UpAddrLVal = 5226 CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false); 5227 llvm::Value *UpAddr = CGF.Builder.CreateConstGEP1_32( 5228 UpAddrLVal.getPointer(CGF), /*Idx0=*/1); 5229 llvm::Value *LowIntPtr = 5230 CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGM.SizeTy); 5231 llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGM.SizeTy); 5232 Size = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr); 5233 } else { 5234 Size = CGF.getTypeSize(Ty); 5235 } 5236 LValue Base = CGF.MakeAddrLValue( 5237 CGF.Builder.CreateConstArrayGEP(DependenciesArray, I), 5238 KmpDependInfoTy); 5239 // deps[i].base_addr = &<Dependences[i].second>; 5240 LValue BaseAddrLVal = CGF.EmitLValueForField( 5241 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 5242 CGF.EmitStoreOfScalar( 5243 CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGF.IntPtrTy), 5244 BaseAddrLVal); 5245 // deps[i].len = sizeof(<Dependences[i].second>); 5246 LValue LenLVal = CGF.EmitLValueForField( 5247 Base, *std::next(KmpDependInfoRD->field_begin(), Len)); 5248 CGF.EmitStoreOfScalar(Size, LenLVal); 5249 // deps[i].flags = <Dependences[i].first>; 5250 RTLDependenceKindTy DepKind; 5251 switch (Data.Dependences[I].first) { 5252 case OMPC_DEPEND_in: 5253 DepKind = DepIn; 5254 break; 5255 // Out and InOut dependencies must use the same code. 5256 case OMPC_DEPEND_out: 5257 case OMPC_DEPEND_inout: 5258 DepKind = DepInOut; 5259 break; 5260 case OMPC_DEPEND_mutexinoutset: 5261 DepKind = DepMutexInOutSet; 5262 break; 5263 case OMPC_DEPEND_source: 5264 case OMPC_DEPEND_sink: 5265 case OMPC_DEPEND_unknown: 5266 llvm_unreachable("Unknown task dependence type"); 5267 } 5268 LValue FlagsLVal = CGF.EmitLValueForField( 5269 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 5270 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 5271 FlagsLVal); 5272 } 5273 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5274 CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0), CGF.VoidPtrTy); 5275 } 5276 5277 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5278 // libcall. 5279 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 5280 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 5281 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence 5282 // list is not empty 5283 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5284 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5285 llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask }; 5286 llvm::Value *DepTaskArgs[7]; 5287 if (NumDependencies) { 5288 DepTaskArgs[0] = UpLoc; 5289 DepTaskArgs[1] = ThreadID; 5290 DepTaskArgs[2] = NewTask; 5291 DepTaskArgs[3] = CGF.Builder.getInt32(NumDependencies); 5292 DepTaskArgs[4] = DependenciesArray.getPointer(); 5293 DepTaskArgs[5] = CGF.Builder.getInt32(0); 5294 DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5295 } 5296 auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, NumDependencies, 5297 &TaskArgs, 5298 &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) { 5299 if (!Data.Tied) { 5300 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 5301 LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI); 5302 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal); 5303 } 5304 if (NumDependencies) { 5305 CGF.EmitRuntimeCall( 5306 createRuntimeFunction(OMPRTL__kmpc_omp_task_with_deps), DepTaskArgs); 5307 } else { 5308 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), 5309 TaskArgs); 5310 } 5311 // Check if parent region is untied and build return for untied task; 5312 if (auto *Region = 5313 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 5314 Region->emitUntiedSwitch(CGF); 5315 }; 5316 5317 llvm::Value *DepWaitTaskArgs[6]; 5318 if (NumDependencies) { 5319 DepWaitTaskArgs[0] = UpLoc; 5320 DepWaitTaskArgs[1] = ThreadID; 5321 DepWaitTaskArgs[2] = CGF.Builder.getInt32(NumDependencies); 5322 DepWaitTaskArgs[3] = DependenciesArray.getPointer(); 5323 DepWaitTaskArgs[4] = CGF.Builder.getInt32(0); 5324 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5325 } 5326 auto &&ElseCodeGen = [&TaskArgs, ThreadID, NewTaskNewTaskTTy, TaskEntry, 5327 NumDependencies, &DepWaitTaskArgs, 5328 Loc](CodeGenFunction &CGF, PrePostActionTy &) { 5329 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5330 CodeGenFunction::RunCleanupsScope LocalScope(CGF); 5331 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 5332 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 5333 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info 5334 // is specified. 5335 if (NumDependencies) 5336 CGF.EmitRuntimeCall(RT.createRuntimeFunction(OMPRTL__kmpc_omp_wait_deps), 5337 DepWaitTaskArgs); 5338 // Call proxy_task_entry(gtid, new_task); 5339 auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy, 5340 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 5341 Action.Enter(CGF); 5342 llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy}; 5343 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry, 5344 OutlinedFnArgs); 5345 }; 5346 5347 // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 5348 // kmp_task_t *new_task); 5349 // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 5350 // kmp_task_t *new_task); 5351 RegionCodeGenTy RCG(CodeGen); 5352 CommonActionTy Action( 5353 RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_begin_if0), TaskArgs, 5354 RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_complete_if0), TaskArgs); 5355 RCG.setAction(Action); 5356 RCG(CGF); 5357 }; 5358 5359 if (IfCond) { 5360 emitIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen); 5361 } else { 5362 RegionCodeGenTy ThenRCG(ThenCodeGen); 5363 ThenRCG(CGF); 5364 } 5365 } 5366 5367 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc, 5368 const OMPLoopDirective &D, 5369 llvm::Function *TaskFunction, 5370 QualType SharedsTy, Address Shareds, 5371 const Expr *IfCond, 5372 const OMPTaskDataTy &Data) { 5373 if (!CGF.HaveInsertPoint()) 5374 return; 5375 TaskResultTy Result = 5376 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5377 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5378 // libcall. 5379 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 5380 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 5381 // sched, kmp_uint64 grainsize, void *task_dup); 5382 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5383 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5384 llvm::Value *IfVal; 5385 if (IfCond) { 5386 IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy, 5387 /*isSigned=*/true); 5388 } else { 5389 IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1); 5390 } 5391 5392 LValue LBLVal = CGF.EmitLValueForField( 5393 Result.TDBase, 5394 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound)); 5395 const auto *LBVar = 5396 cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl()); 5397 CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(CGF), 5398 LBLVal.getQuals(), 5399 /*IsInitializer=*/true); 5400 LValue UBLVal = CGF.EmitLValueForField( 5401 Result.TDBase, 5402 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound)); 5403 const auto *UBVar = 5404 cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl()); 5405 CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(CGF), 5406 UBLVal.getQuals(), 5407 /*IsInitializer=*/true); 5408 LValue StLVal = CGF.EmitLValueForField( 5409 Result.TDBase, 5410 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride)); 5411 const auto *StVar = 5412 cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl()); 5413 CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(CGF), 5414 StLVal.getQuals(), 5415 /*IsInitializer=*/true); 5416 // Store reductions address. 5417 LValue RedLVal = CGF.EmitLValueForField( 5418 Result.TDBase, 5419 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions)); 5420 if (Data.Reductions) { 5421 CGF.EmitStoreOfScalar(Data.Reductions, RedLVal); 5422 } else { 5423 CGF.EmitNullInitialization(RedLVal.getAddress(CGF), 5424 CGF.getContext().VoidPtrTy); 5425 } 5426 enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 }; 5427 llvm::Value *TaskArgs[] = { 5428 UpLoc, 5429 ThreadID, 5430 Result.NewTask, 5431 IfVal, 5432 LBLVal.getPointer(CGF), 5433 UBLVal.getPointer(CGF), 5434 CGF.EmitLoadOfScalar(StLVal, Loc), 5435 llvm::ConstantInt::getSigned( 5436 CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler 5437 llvm::ConstantInt::getSigned( 5438 CGF.IntTy, Data.Schedule.getPointer() 5439 ? Data.Schedule.getInt() ? NumTasks : Grainsize 5440 : NoSchedule), 5441 Data.Schedule.getPointer() 5442 ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty, 5443 /*isSigned=*/false) 5444 : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0), 5445 Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5446 Result.TaskDupFn, CGF.VoidPtrTy) 5447 : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)}; 5448 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_taskloop), TaskArgs); 5449 } 5450 5451 /// Emit reduction operation for each element of array (required for 5452 /// array sections) LHS op = RHS. 5453 /// \param Type Type of array. 5454 /// \param LHSVar Variable on the left side of the reduction operation 5455 /// (references element of array in original variable). 5456 /// \param RHSVar Variable on the right side of the reduction operation 5457 /// (references element of array in original variable). 5458 /// \param RedOpGen Generator of reduction operation with use of LHSVar and 5459 /// RHSVar. 5460 static void EmitOMPAggregateReduction( 5461 CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar, 5462 const VarDecl *RHSVar, 5463 const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *, 5464 const Expr *, const Expr *)> &RedOpGen, 5465 const Expr *XExpr = nullptr, const Expr *EExpr = nullptr, 5466 const Expr *UpExpr = nullptr) { 5467 // Perform element-by-element initialization. 5468 QualType ElementTy; 5469 Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar); 5470 Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar); 5471 5472 // Drill down to the base element type on both arrays. 5473 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 5474 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr); 5475 5476 llvm::Value *RHSBegin = RHSAddr.getPointer(); 5477 llvm::Value *LHSBegin = LHSAddr.getPointer(); 5478 // Cast from pointer to array type to pointer to single element. 5479 llvm::Value *LHSEnd = CGF.Builder.CreateGEP(LHSBegin, NumElements); 5480 // The basic structure here is a while-do loop. 5481 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body"); 5482 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done"); 5483 llvm::Value *IsEmpty = 5484 CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty"); 5485 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5486 5487 // Enter the loop body, making that address the current address. 5488 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5489 CGF.EmitBlock(BodyBB); 5490 5491 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 5492 5493 llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI( 5494 RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast"); 5495 RHSElementPHI->addIncoming(RHSBegin, EntryBB); 5496 Address RHSElementCurrent = 5497 Address(RHSElementPHI, 5498 RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5499 5500 llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI( 5501 LHSBegin->getType(), 2, "omp.arraycpy.destElementPast"); 5502 LHSElementPHI->addIncoming(LHSBegin, EntryBB); 5503 Address LHSElementCurrent = 5504 Address(LHSElementPHI, 5505 LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5506 5507 // Emit copy. 5508 CodeGenFunction::OMPPrivateScope Scope(CGF); 5509 Scope.addPrivate(LHSVar, [=]() { return LHSElementCurrent; }); 5510 Scope.addPrivate(RHSVar, [=]() { return RHSElementCurrent; }); 5511 Scope.Privatize(); 5512 RedOpGen(CGF, XExpr, EExpr, UpExpr); 5513 Scope.ForceCleanup(); 5514 5515 // Shift the address forward by one element. 5516 llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32( 5517 LHSElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 5518 llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32( 5519 RHSElementPHI, /*Idx0=*/1, "omp.arraycpy.src.element"); 5520 // Check whether we've reached the end. 5521 llvm::Value *Done = 5522 CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done"); 5523 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 5524 LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock()); 5525 RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock()); 5526 5527 // Done. 5528 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5529 } 5530 5531 /// Emit reduction combiner. If the combiner is a simple expression emit it as 5532 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of 5533 /// UDR combiner function. 5534 static void emitReductionCombiner(CodeGenFunction &CGF, 5535 const Expr *ReductionOp) { 5536 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 5537 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 5538 if (const auto *DRE = 5539 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 5540 if (const auto *DRD = 5541 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) { 5542 std::pair<llvm::Function *, llvm::Function *> Reduction = 5543 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 5544 RValue Func = RValue::get(Reduction.first); 5545 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 5546 CGF.EmitIgnoredExpr(ReductionOp); 5547 return; 5548 } 5549 CGF.EmitIgnoredExpr(ReductionOp); 5550 } 5551 5552 llvm::Function *CGOpenMPRuntime::emitReductionFunction( 5553 SourceLocation Loc, llvm::Type *ArgsType, ArrayRef<const Expr *> Privates, 5554 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 5555 ArrayRef<const Expr *> ReductionOps) { 5556 ASTContext &C = CGM.getContext(); 5557 5558 // void reduction_func(void *LHSArg, void *RHSArg); 5559 FunctionArgList Args; 5560 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5561 ImplicitParamDecl::Other); 5562 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5563 ImplicitParamDecl::Other); 5564 Args.push_back(&LHSArg); 5565 Args.push_back(&RHSArg); 5566 const auto &CGFI = 5567 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5568 std::string Name = getName({"omp", "reduction", "reduction_func"}); 5569 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 5570 llvm::GlobalValue::InternalLinkage, Name, 5571 &CGM.getModule()); 5572 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 5573 Fn->setDoesNotRecurse(); 5574 CodeGenFunction CGF(CGM); 5575 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 5576 5577 // Dst = (void*[n])(LHSArg); 5578 // Src = (void*[n])(RHSArg); 5579 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5580 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 5581 ArgsType), CGF.getPointerAlign()); 5582 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5583 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 5584 ArgsType), CGF.getPointerAlign()); 5585 5586 // ... 5587 // *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]); 5588 // ... 5589 CodeGenFunction::OMPPrivateScope Scope(CGF); 5590 auto IPriv = Privates.begin(); 5591 unsigned Idx = 0; 5592 for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) { 5593 const auto *RHSVar = 5594 cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl()); 5595 Scope.addPrivate(RHSVar, [&CGF, RHS, Idx, RHSVar]() { 5596 return emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar); 5597 }); 5598 const auto *LHSVar = 5599 cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl()); 5600 Scope.addPrivate(LHSVar, [&CGF, LHS, Idx, LHSVar]() { 5601 return emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar); 5602 }); 5603 QualType PrivTy = (*IPriv)->getType(); 5604 if (PrivTy->isVariablyModifiedType()) { 5605 // Get array size and emit VLA type. 5606 ++Idx; 5607 Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx); 5608 llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem); 5609 const VariableArrayType *VLA = 5610 CGF.getContext().getAsVariableArrayType(PrivTy); 5611 const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr()); 5612 CodeGenFunction::OpaqueValueMapping OpaqueMap( 5613 CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy))); 5614 CGF.EmitVariablyModifiedType(PrivTy); 5615 } 5616 } 5617 Scope.Privatize(); 5618 IPriv = Privates.begin(); 5619 auto ILHS = LHSExprs.begin(); 5620 auto IRHS = RHSExprs.begin(); 5621 for (const Expr *E : ReductionOps) { 5622 if ((*IPriv)->getType()->isArrayType()) { 5623 // Emit reduction for array section. 5624 const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5625 const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5626 EmitOMPAggregateReduction( 5627 CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5628 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5629 emitReductionCombiner(CGF, E); 5630 }); 5631 } else { 5632 // Emit reduction for array subscript or single variable. 5633 emitReductionCombiner(CGF, E); 5634 } 5635 ++IPriv; 5636 ++ILHS; 5637 ++IRHS; 5638 } 5639 Scope.ForceCleanup(); 5640 CGF.FinishFunction(); 5641 return Fn; 5642 } 5643 5644 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF, 5645 const Expr *ReductionOp, 5646 const Expr *PrivateRef, 5647 const DeclRefExpr *LHS, 5648 const DeclRefExpr *RHS) { 5649 if (PrivateRef->getType()->isArrayType()) { 5650 // Emit reduction for array section. 5651 const auto *LHSVar = cast<VarDecl>(LHS->getDecl()); 5652 const auto *RHSVar = cast<VarDecl>(RHS->getDecl()); 5653 EmitOMPAggregateReduction( 5654 CGF, PrivateRef->getType(), LHSVar, RHSVar, 5655 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5656 emitReductionCombiner(CGF, ReductionOp); 5657 }); 5658 } else { 5659 // Emit reduction for array subscript or single variable. 5660 emitReductionCombiner(CGF, ReductionOp); 5661 } 5662 } 5663 5664 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc, 5665 ArrayRef<const Expr *> Privates, 5666 ArrayRef<const Expr *> LHSExprs, 5667 ArrayRef<const Expr *> RHSExprs, 5668 ArrayRef<const Expr *> ReductionOps, 5669 ReductionOptionsTy Options) { 5670 if (!CGF.HaveInsertPoint()) 5671 return; 5672 5673 bool WithNowait = Options.WithNowait; 5674 bool SimpleReduction = Options.SimpleReduction; 5675 5676 // Next code should be emitted for reduction: 5677 // 5678 // static kmp_critical_name lock = { 0 }; 5679 // 5680 // void reduce_func(void *lhs[<n>], void *rhs[<n>]) { 5681 // *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]); 5682 // ... 5683 // *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1], 5684 // *(Type<n>-1*)rhs[<n>-1]); 5685 // } 5686 // 5687 // ... 5688 // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]}; 5689 // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5690 // RedList, reduce_func, &<lock>)) { 5691 // case 1: 5692 // ... 5693 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5694 // ... 5695 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5696 // break; 5697 // case 2: 5698 // ... 5699 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5700 // ... 5701 // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);] 5702 // break; 5703 // default:; 5704 // } 5705 // 5706 // if SimpleReduction is true, only the next code is generated: 5707 // ... 5708 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5709 // ... 5710 5711 ASTContext &C = CGM.getContext(); 5712 5713 if (SimpleReduction) { 5714 CodeGenFunction::RunCleanupsScope Scope(CGF); 5715 auto IPriv = Privates.begin(); 5716 auto ILHS = LHSExprs.begin(); 5717 auto IRHS = RHSExprs.begin(); 5718 for (const Expr *E : ReductionOps) { 5719 emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5720 cast<DeclRefExpr>(*IRHS)); 5721 ++IPriv; 5722 ++ILHS; 5723 ++IRHS; 5724 } 5725 return; 5726 } 5727 5728 // 1. Build a list of reduction variables. 5729 // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]}; 5730 auto Size = RHSExprs.size(); 5731 for (const Expr *E : Privates) { 5732 if (E->getType()->isVariablyModifiedType()) 5733 // Reserve place for array size. 5734 ++Size; 5735 } 5736 llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size); 5737 QualType ReductionArrayTy = 5738 C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 5739 /*IndexTypeQuals=*/0); 5740 Address ReductionList = 5741 CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list"); 5742 auto IPriv = Privates.begin(); 5743 unsigned Idx = 0; 5744 for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) { 5745 Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 5746 CGF.Builder.CreateStore( 5747 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5748 CGF.EmitLValue(RHSExprs[I]).getPointer(CGF), CGF.VoidPtrTy), 5749 Elem); 5750 if ((*IPriv)->getType()->isVariablyModifiedType()) { 5751 // Store array size. 5752 ++Idx; 5753 Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 5754 llvm::Value *Size = CGF.Builder.CreateIntCast( 5755 CGF.getVLASize( 5756 CGF.getContext().getAsVariableArrayType((*IPriv)->getType())) 5757 .NumElts, 5758 CGF.SizeTy, /*isSigned=*/false); 5759 CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy), 5760 Elem); 5761 } 5762 } 5763 5764 // 2. Emit reduce_func(). 5765 llvm::Function *ReductionFn = emitReductionFunction( 5766 Loc, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), Privates, 5767 LHSExprs, RHSExprs, ReductionOps); 5768 5769 // 3. Create static kmp_critical_name lock = { 0 }; 5770 std::string Name = getName({"reduction"}); 5771 llvm::Value *Lock = getCriticalRegionLock(Name); 5772 5773 // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5774 // RedList, reduce_func, &<lock>); 5775 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE); 5776 llvm::Value *ThreadId = getThreadID(CGF, Loc); 5777 llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy); 5778 llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5779 ReductionList.getPointer(), CGF.VoidPtrTy); 5780 llvm::Value *Args[] = { 5781 IdentTLoc, // ident_t *<loc> 5782 ThreadId, // i32 <gtid> 5783 CGF.Builder.getInt32(RHSExprs.size()), // i32 <n> 5784 ReductionArrayTySize, // size_type sizeof(RedList) 5785 RL, // void *RedList 5786 ReductionFn, // void (*) (void *, void *) <reduce_func> 5787 Lock // kmp_critical_name *&<lock> 5788 }; 5789 llvm::Value *Res = CGF.EmitRuntimeCall( 5790 createRuntimeFunction(WithNowait ? OMPRTL__kmpc_reduce_nowait 5791 : OMPRTL__kmpc_reduce), 5792 Args); 5793 5794 // 5. Build switch(res) 5795 llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default"); 5796 llvm::SwitchInst *SwInst = 5797 CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2); 5798 5799 // 6. Build case 1: 5800 // ... 5801 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5802 // ... 5803 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5804 // break; 5805 llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1"); 5806 SwInst->addCase(CGF.Builder.getInt32(1), Case1BB); 5807 CGF.EmitBlock(Case1BB); 5808 5809 // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5810 llvm::Value *EndArgs[] = { 5811 IdentTLoc, // ident_t *<loc> 5812 ThreadId, // i32 <gtid> 5813 Lock // kmp_critical_name *&<lock> 5814 }; 5815 auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps]( 5816 CodeGenFunction &CGF, PrePostActionTy &Action) { 5817 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5818 auto IPriv = Privates.begin(); 5819 auto ILHS = LHSExprs.begin(); 5820 auto IRHS = RHSExprs.begin(); 5821 for (const Expr *E : ReductionOps) { 5822 RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5823 cast<DeclRefExpr>(*IRHS)); 5824 ++IPriv; 5825 ++ILHS; 5826 ++IRHS; 5827 } 5828 }; 5829 RegionCodeGenTy RCG(CodeGen); 5830 CommonActionTy Action( 5831 nullptr, llvm::None, 5832 createRuntimeFunction(WithNowait ? OMPRTL__kmpc_end_reduce_nowait 5833 : OMPRTL__kmpc_end_reduce), 5834 EndArgs); 5835 RCG.setAction(Action); 5836 RCG(CGF); 5837 5838 CGF.EmitBranch(DefaultBB); 5839 5840 // 7. Build case 2: 5841 // ... 5842 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5843 // ... 5844 // break; 5845 llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2"); 5846 SwInst->addCase(CGF.Builder.getInt32(2), Case2BB); 5847 CGF.EmitBlock(Case2BB); 5848 5849 auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps]( 5850 CodeGenFunction &CGF, PrePostActionTy &Action) { 5851 auto ILHS = LHSExprs.begin(); 5852 auto IRHS = RHSExprs.begin(); 5853 auto IPriv = Privates.begin(); 5854 for (const Expr *E : ReductionOps) { 5855 const Expr *XExpr = nullptr; 5856 const Expr *EExpr = nullptr; 5857 const Expr *UpExpr = nullptr; 5858 BinaryOperatorKind BO = BO_Comma; 5859 if (const auto *BO = dyn_cast<BinaryOperator>(E)) { 5860 if (BO->getOpcode() == BO_Assign) { 5861 XExpr = BO->getLHS(); 5862 UpExpr = BO->getRHS(); 5863 } 5864 } 5865 // Try to emit update expression as a simple atomic. 5866 const Expr *RHSExpr = UpExpr; 5867 if (RHSExpr) { 5868 // Analyze RHS part of the whole expression. 5869 if (const auto *ACO = dyn_cast<AbstractConditionalOperator>( 5870 RHSExpr->IgnoreParenImpCasts())) { 5871 // If this is a conditional operator, analyze its condition for 5872 // min/max reduction operator. 5873 RHSExpr = ACO->getCond(); 5874 } 5875 if (const auto *BORHS = 5876 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) { 5877 EExpr = BORHS->getRHS(); 5878 BO = BORHS->getOpcode(); 5879 } 5880 } 5881 if (XExpr) { 5882 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5883 auto &&AtomicRedGen = [BO, VD, 5884 Loc](CodeGenFunction &CGF, const Expr *XExpr, 5885 const Expr *EExpr, const Expr *UpExpr) { 5886 LValue X = CGF.EmitLValue(XExpr); 5887 RValue E; 5888 if (EExpr) 5889 E = CGF.EmitAnyExpr(EExpr); 5890 CGF.EmitOMPAtomicSimpleUpdateExpr( 5891 X, E, BO, /*IsXLHSInRHSPart=*/true, 5892 llvm::AtomicOrdering::Monotonic, Loc, 5893 [&CGF, UpExpr, VD, Loc](RValue XRValue) { 5894 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 5895 PrivateScope.addPrivate( 5896 VD, [&CGF, VD, XRValue, Loc]() { 5897 Address LHSTemp = CGF.CreateMemTemp(VD->getType()); 5898 CGF.emitOMPSimpleStore( 5899 CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue, 5900 VD->getType().getNonReferenceType(), Loc); 5901 return LHSTemp; 5902 }); 5903 (void)PrivateScope.Privatize(); 5904 return CGF.EmitAnyExpr(UpExpr); 5905 }); 5906 }; 5907 if ((*IPriv)->getType()->isArrayType()) { 5908 // Emit atomic reduction for array section. 5909 const auto *RHSVar = 5910 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5911 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar, 5912 AtomicRedGen, XExpr, EExpr, UpExpr); 5913 } else { 5914 // Emit atomic reduction for array subscript or single variable. 5915 AtomicRedGen(CGF, XExpr, EExpr, UpExpr); 5916 } 5917 } else { 5918 // Emit as a critical region. 5919 auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *, 5920 const Expr *, const Expr *) { 5921 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5922 std::string Name = RT.getName({"atomic_reduction"}); 5923 RT.emitCriticalRegion( 5924 CGF, Name, 5925 [=](CodeGenFunction &CGF, PrePostActionTy &Action) { 5926 Action.Enter(CGF); 5927 emitReductionCombiner(CGF, E); 5928 }, 5929 Loc); 5930 }; 5931 if ((*IPriv)->getType()->isArrayType()) { 5932 const auto *LHSVar = 5933 cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5934 const auto *RHSVar = 5935 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5936 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5937 CritRedGen); 5938 } else { 5939 CritRedGen(CGF, nullptr, nullptr, nullptr); 5940 } 5941 } 5942 ++ILHS; 5943 ++IRHS; 5944 ++IPriv; 5945 } 5946 }; 5947 RegionCodeGenTy AtomicRCG(AtomicCodeGen); 5948 if (!WithNowait) { 5949 // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>); 5950 llvm::Value *EndArgs[] = { 5951 IdentTLoc, // ident_t *<loc> 5952 ThreadId, // i32 <gtid> 5953 Lock // kmp_critical_name *&<lock> 5954 }; 5955 CommonActionTy Action(nullptr, llvm::None, 5956 createRuntimeFunction(OMPRTL__kmpc_end_reduce), 5957 EndArgs); 5958 AtomicRCG.setAction(Action); 5959 AtomicRCG(CGF); 5960 } else { 5961 AtomicRCG(CGF); 5962 } 5963 5964 CGF.EmitBranch(DefaultBB); 5965 CGF.EmitBlock(DefaultBB, /*IsFinished=*/true); 5966 } 5967 5968 /// Generates unique name for artificial threadprivate variables. 5969 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>" 5970 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix, 5971 const Expr *Ref) { 5972 SmallString<256> Buffer; 5973 llvm::raw_svector_ostream Out(Buffer); 5974 const clang::DeclRefExpr *DE; 5975 const VarDecl *D = ::getBaseDecl(Ref, DE); 5976 if (!D) 5977 D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl()); 5978 D = D->getCanonicalDecl(); 5979 std::string Name = CGM.getOpenMPRuntime().getName( 5980 {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)}); 5981 Out << Prefix << Name << "_" 5982 << D->getCanonicalDecl()->getBeginLoc().getRawEncoding(); 5983 return std::string(Out.str()); 5984 } 5985 5986 /// Emits reduction initializer function: 5987 /// \code 5988 /// void @.red_init(void* %arg) { 5989 /// %0 = bitcast void* %arg to <type>* 5990 /// store <type> <init>, <type>* %0 5991 /// ret void 5992 /// } 5993 /// \endcode 5994 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM, 5995 SourceLocation Loc, 5996 ReductionCodeGen &RCG, unsigned N) { 5997 ASTContext &C = CGM.getContext(); 5998 FunctionArgList Args; 5999 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6000 ImplicitParamDecl::Other); 6001 Args.emplace_back(&Param); 6002 const auto &FnInfo = 6003 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6004 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6005 std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""}); 6006 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6007 Name, &CGM.getModule()); 6008 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6009 Fn->setDoesNotRecurse(); 6010 CodeGenFunction CGF(CGM); 6011 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6012 Address PrivateAddr = CGF.EmitLoadOfPointer( 6013 CGF.GetAddrOfLocalVar(&Param), 6014 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6015 llvm::Value *Size = nullptr; 6016 // If the size of the reduction item is non-constant, load it from global 6017 // threadprivate variable. 6018 if (RCG.getSizes(N).second) { 6019 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6020 CGF, CGM.getContext().getSizeType(), 6021 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6022 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6023 CGM.getContext().getSizeType(), Loc); 6024 } 6025 RCG.emitAggregateType(CGF, N, Size); 6026 LValue SharedLVal; 6027 // If initializer uses initializer from declare reduction construct, emit a 6028 // pointer to the address of the original reduction item (reuired by reduction 6029 // initializer) 6030 if (RCG.usesReductionInitializer(N)) { 6031 Address SharedAddr = 6032 CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6033 CGF, CGM.getContext().VoidPtrTy, 6034 generateUniqueName(CGM, "reduction", RCG.getRefExpr(N))); 6035 SharedAddr = CGF.EmitLoadOfPointer( 6036 SharedAddr, 6037 CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr()); 6038 SharedLVal = CGF.MakeAddrLValue(SharedAddr, CGM.getContext().VoidPtrTy); 6039 } else { 6040 SharedLVal = CGF.MakeNaturalAlignAddrLValue( 6041 llvm::ConstantPointerNull::get(CGM.VoidPtrTy), 6042 CGM.getContext().VoidPtrTy); 6043 } 6044 // Emit the initializer: 6045 // %0 = bitcast void* %arg to <type>* 6046 // store <type> <init>, <type>* %0 6047 RCG.emitInitialization(CGF, N, PrivateAddr, SharedLVal, 6048 [](CodeGenFunction &) { return false; }); 6049 CGF.FinishFunction(); 6050 return Fn; 6051 } 6052 6053 /// Emits reduction combiner function: 6054 /// \code 6055 /// void @.red_comb(void* %arg0, void* %arg1) { 6056 /// %lhs = bitcast void* %arg0 to <type>* 6057 /// %rhs = bitcast void* %arg1 to <type>* 6058 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs) 6059 /// store <type> %2, <type>* %lhs 6060 /// ret void 6061 /// } 6062 /// \endcode 6063 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM, 6064 SourceLocation Loc, 6065 ReductionCodeGen &RCG, unsigned N, 6066 const Expr *ReductionOp, 6067 const Expr *LHS, const Expr *RHS, 6068 const Expr *PrivateRef) { 6069 ASTContext &C = CGM.getContext(); 6070 const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl()); 6071 const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl()); 6072 FunctionArgList Args; 6073 ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 6074 C.VoidPtrTy, ImplicitParamDecl::Other); 6075 ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6076 ImplicitParamDecl::Other); 6077 Args.emplace_back(&ParamInOut); 6078 Args.emplace_back(&ParamIn); 6079 const auto &FnInfo = 6080 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6081 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6082 std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""}); 6083 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6084 Name, &CGM.getModule()); 6085 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6086 Fn->setDoesNotRecurse(); 6087 CodeGenFunction CGF(CGM); 6088 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6089 llvm::Value *Size = nullptr; 6090 // If the size of the reduction item is non-constant, load it from global 6091 // threadprivate variable. 6092 if (RCG.getSizes(N).second) { 6093 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6094 CGF, CGM.getContext().getSizeType(), 6095 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6096 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6097 CGM.getContext().getSizeType(), Loc); 6098 } 6099 RCG.emitAggregateType(CGF, N, Size); 6100 // Remap lhs and rhs variables to the addresses of the function arguments. 6101 // %lhs = bitcast void* %arg0 to <type>* 6102 // %rhs = bitcast void* %arg1 to <type>* 6103 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 6104 PrivateScope.addPrivate(LHSVD, [&C, &CGF, &ParamInOut, LHSVD]() { 6105 // Pull out the pointer to the variable. 6106 Address PtrAddr = CGF.EmitLoadOfPointer( 6107 CGF.GetAddrOfLocalVar(&ParamInOut), 6108 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6109 return CGF.Builder.CreateElementBitCast( 6110 PtrAddr, CGF.ConvertTypeForMem(LHSVD->getType())); 6111 }); 6112 PrivateScope.addPrivate(RHSVD, [&C, &CGF, &ParamIn, RHSVD]() { 6113 // Pull out the pointer to the variable. 6114 Address PtrAddr = CGF.EmitLoadOfPointer( 6115 CGF.GetAddrOfLocalVar(&ParamIn), 6116 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6117 return CGF.Builder.CreateElementBitCast( 6118 PtrAddr, CGF.ConvertTypeForMem(RHSVD->getType())); 6119 }); 6120 PrivateScope.Privatize(); 6121 // Emit the combiner body: 6122 // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs) 6123 // store <type> %2, <type>* %lhs 6124 CGM.getOpenMPRuntime().emitSingleReductionCombiner( 6125 CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS), 6126 cast<DeclRefExpr>(RHS)); 6127 CGF.FinishFunction(); 6128 return Fn; 6129 } 6130 6131 /// Emits reduction finalizer function: 6132 /// \code 6133 /// void @.red_fini(void* %arg) { 6134 /// %0 = bitcast void* %arg to <type>* 6135 /// <destroy>(<type>* %0) 6136 /// ret void 6137 /// } 6138 /// \endcode 6139 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM, 6140 SourceLocation Loc, 6141 ReductionCodeGen &RCG, unsigned N) { 6142 if (!RCG.needCleanups(N)) 6143 return nullptr; 6144 ASTContext &C = CGM.getContext(); 6145 FunctionArgList Args; 6146 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6147 ImplicitParamDecl::Other); 6148 Args.emplace_back(&Param); 6149 const auto &FnInfo = 6150 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6151 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6152 std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""}); 6153 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6154 Name, &CGM.getModule()); 6155 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6156 Fn->setDoesNotRecurse(); 6157 CodeGenFunction CGF(CGM); 6158 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6159 Address PrivateAddr = CGF.EmitLoadOfPointer( 6160 CGF.GetAddrOfLocalVar(&Param), 6161 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6162 llvm::Value *Size = nullptr; 6163 // If the size of the reduction item is non-constant, load it from global 6164 // threadprivate variable. 6165 if (RCG.getSizes(N).second) { 6166 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6167 CGF, CGM.getContext().getSizeType(), 6168 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6169 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6170 CGM.getContext().getSizeType(), Loc); 6171 } 6172 RCG.emitAggregateType(CGF, N, Size); 6173 // Emit the finalizer body: 6174 // <destroy>(<type>* %0) 6175 RCG.emitCleanups(CGF, N, PrivateAddr); 6176 CGF.FinishFunction(Loc); 6177 return Fn; 6178 } 6179 6180 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit( 6181 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 6182 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 6183 if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty()) 6184 return nullptr; 6185 6186 // Build typedef struct: 6187 // kmp_task_red_input { 6188 // void *reduce_shar; // shared reduction item 6189 // size_t reduce_size; // size of data item 6190 // void *reduce_init; // data initialization routine 6191 // void *reduce_fini; // data finalization routine 6192 // void *reduce_comb; // data combiner routine 6193 // kmp_task_red_flags_t flags; // flags for additional info from compiler 6194 // } kmp_task_red_input_t; 6195 ASTContext &C = CGM.getContext(); 6196 RecordDecl *RD = C.buildImplicitRecord("kmp_task_red_input_t"); 6197 RD->startDefinition(); 6198 const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6199 const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType()); 6200 const FieldDecl *InitFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6201 const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6202 const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6203 const FieldDecl *FlagsFD = addFieldToRecordDecl( 6204 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false)); 6205 RD->completeDefinition(); 6206 QualType RDType = C.getRecordType(RD); 6207 unsigned Size = Data.ReductionVars.size(); 6208 llvm::APInt ArraySize(/*numBits=*/64, Size); 6209 QualType ArrayRDType = C.getConstantArrayType( 6210 RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 6211 // kmp_task_red_input_t .rd_input.[Size]; 6212 Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input."); 6213 ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionCopies, 6214 Data.ReductionOps); 6215 for (unsigned Cnt = 0; Cnt < Size; ++Cnt) { 6216 // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt]; 6217 llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0), 6218 llvm::ConstantInt::get(CGM.SizeTy, Cnt)}; 6219 llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP( 6220 TaskRedInput.getPointer(), Idxs, 6221 /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc, 6222 ".rd_input.gep."); 6223 LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType); 6224 // ElemLVal.reduce_shar = &Shareds[Cnt]; 6225 LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD); 6226 RCG.emitSharedLValue(CGF, Cnt); 6227 llvm::Value *CastedShared = 6228 CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer(CGF)); 6229 CGF.EmitStoreOfScalar(CastedShared, SharedLVal); 6230 RCG.emitAggregateType(CGF, Cnt); 6231 llvm::Value *SizeValInChars; 6232 llvm::Value *SizeVal; 6233 std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt); 6234 // We use delayed creation/initialization for VLAs, array sections and 6235 // custom reduction initializations. It is required because runtime does not 6236 // provide the way to pass the sizes of VLAs/array sections to 6237 // initializer/combiner/finalizer functions and does not pass the pointer to 6238 // original reduction item to the initializer. Instead threadprivate global 6239 // variables are used to store these values and use them in the functions. 6240 bool DelayedCreation = !!SizeVal; 6241 SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy, 6242 /*isSigned=*/false); 6243 LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD); 6244 CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal); 6245 // ElemLVal.reduce_init = init; 6246 LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD); 6247 llvm::Value *InitAddr = 6248 CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt)); 6249 CGF.EmitStoreOfScalar(InitAddr, InitLVal); 6250 DelayedCreation = DelayedCreation || RCG.usesReductionInitializer(Cnt); 6251 // ElemLVal.reduce_fini = fini; 6252 LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD); 6253 llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt); 6254 llvm::Value *FiniAddr = Fini 6255 ? CGF.EmitCastToVoidPtr(Fini) 6256 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy); 6257 CGF.EmitStoreOfScalar(FiniAddr, FiniLVal); 6258 // ElemLVal.reduce_comb = comb; 6259 LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD); 6260 llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction( 6261 CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt], 6262 RHSExprs[Cnt], Data.ReductionCopies[Cnt])); 6263 CGF.EmitStoreOfScalar(CombAddr, CombLVal); 6264 // ElemLVal.flags = 0; 6265 LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD); 6266 if (DelayedCreation) { 6267 CGF.EmitStoreOfScalar( 6268 llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true), 6269 FlagsLVal); 6270 } else 6271 CGF.EmitNullInitialization(FlagsLVal.getAddress(CGF), 6272 FlagsLVal.getType()); 6273 } 6274 // Build call void *__kmpc_task_reduction_init(int gtid, int num_data, void 6275 // *data); 6276 llvm::Value *Args[] = { 6277 CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy, 6278 /*isSigned=*/true), 6279 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true), 6280 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(), 6281 CGM.VoidPtrTy)}; 6282 return CGF.EmitRuntimeCall( 6283 createRuntimeFunction(OMPRTL__kmpc_task_reduction_init), Args); 6284 } 6285 6286 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 6287 SourceLocation Loc, 6288 ReductionCodeGen &RCG, 6289 unsigned N) { 6290 auto Sizes = RCG.getSizes(N); 6291 // Emit threadprivate global variable if the type is non-constant 6292 // (Sizes.second = nullptr). 6293 if (Sizes.second) { 6294 llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy, 6295 /*isSigned=*/false); 6296 Address SizeAddr = getAddrOfArtificialThreadPrivate( 6297 CGF, CGM.getContext().getSizeType(), 6298 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6299 CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false); 6300 } 6301 // Store address of the original reduction item if custom initializer is used. 6302 if (RCG.usesReductionInitializer(N)) { 6303 Address SharedAddr = getAddrOfArtificialThreadPrivate( 6304 CGF, CGM.getContext().VoidPtrTy, 6305 generateUniqueName(CGM, "reduction", RCG.getRefExpr(N))); 6306 CGF.Builder.CreateStore( 6307 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6308 RCG.getSharedLValue(N).getPointer(CGF), CGM.VoidPtrTy), 6309 SharedAddr, /*IsVolatile=*/false); 6310 } 6311 } 6312 6313 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF, 6314 SourceLocation Loc, 6315 llvm::Value *ReductionsPtr, 6316 LValue SharedLVal) { 6317 // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 6318 // *d); 6319 llvm::Value *Args[] = {CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), 6320 CGM.IntTy, 6321 /*isSigned=*/true), 6322 ReductionsPtr, 6323 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6324 SharedLVal.getPointer(CGF), CGM.VoidPtrTy)}; 6325 return Address( 6326 CGF.EmitRuntimeCall( 6327 createRuntimeFunction(OMPRTL__kmpc_task_reduction_get_th_data), Args), 6328 SharedLVal.getAlignment()); 6329 } 6330 6331 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 6332 SourceLocation Loc) { 6333 if (!CGF.HaveInsertPoint()) 6334 return; 6335 // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 6336 // global_tid); 6337 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 6338 // Ignore return result until untied tasks are supported. 6339 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskwait), Args); 6340 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 6341 Region->emitUntiedSwitch(CGF); 6342 } 6343 6344 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF, 6345 OpenMPDirectiveKind InnerKind, 6346 const RegionCodeGenTy &CodeGen, 6347 bool HasCancel) { 6348 if (!CGF.HaveInsertPoint()) 6349 return; 6350 InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel); 6351 CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr); 6352 } 6353 6354 namespace { 6355 enum RTCancelKind { 6356 CancelNoreq = 0, 6357 CancelParallel = 1, 6358 CancelLoop = 2, 6359 CancelSections = 3, 6360 CancelTaskgroup = 4 6361 }; 6362 } // anonymous namespace 6363 6364 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) { 6365 RTCancelKind CancelKind = CancelNoreq; 6366 if (CancelRegion == OMPD_parallel) 6367 CancelKind = CancelParallel; 6368 else if (CancelRegion == OMPD_for) 6369 CancelKind = CancelLoop; 6370 else if (CancelRegion == OMPD_sections) 6371 CancelKind = CancelSections; 6372 else { 6373 assert(CancelRegion == OMPD_taskgroup); 6374 CancelKind = CancelTaskgroup; 6375 } 6376 return CancelKind; 6377 } 6378 6379 void CGOpenMPRuntime::emitCancellationPointCall( 6380 CodeGenFunction &CGF, SourceLocation Loc, 6381 OpenMPDirectiveKind CancelRegion) { 6382 if (!CGF.HaveInsertPoint()) 6383 return; 6384 // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 6385 // global_tid, kmp_int32 cncl_kind); 6386 if (auto *OMPRegionInfo = 6387 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6388 // For 'cancellation point taskgroup', the task region info may not have a 6389 // cancel. This may instead happen in another adjacent task. 6390 if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) { 6391 llvm::Value *Args[] = { 6392 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 6393 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6394 // Ignore return result until untied tasks are supported. 6395 llvm::Value *Result = CGF.EmitRuntimeCall( 6396 createRuntimeFunction(OMPRTL__kmpc_cancellationpoint), Args); 6397 // if (__kmpc_cancellationpoint()) { 6398 // exit from construct; 6399 // } 6400 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6401 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6402 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6403 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6404 CGF.EmitBlock(ExitBB); 6405 // exit from construct; 6406 CodeGenFunction::JumpDest CancelDest = 6407 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6408 CGF.EmitBranchThroughCleanup(CancelDest); 6409 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6410 } 6411 } 6412 } 6413 6414 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc, 6415 const Expr *IfCond, 6416 OpenMPDirectiveKind CancelRegion) { 6417 if (!CGF.HaveInsertPoint()) 6418 return; 6419 // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 6420 // kmp_int32 cncl_kind); 6421 if (auto *OMPRegionInfo = 6422 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6423 auto &&ThenGen = [Loc, CancelRegion, OMPRegionInfo](CodeGenFunction &CGF, 6424 PrePostActionTy &) { 6425 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6426 llvm::Value *Args[] = { 6427 RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc), 6428 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6429 // Ignore return result until untied tasks are supported. 6430 llvm::Value *Result = CGF.EmitRuntimeCall( 6431 RT.createRuntimeFunction(OMPRTL__kmpc_cancel), Args); 6432 // if (__kmpc_cancel()) { 6433 // exit from construct; 6434 // } 6435 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6436 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6437 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6438 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6439 CGF.EmitBlock(ExitBB); 6440 // exit from construct; 6441 CodeGenFunction::JumpDest CancelDest = 6442 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6443 CGF.EmitBranchThroughCleanup(CancelDest); 6444 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6445 }; 6446 if (IfCond) { 6447 emitIfClause(CGF, IfCond, ThenGen, 6448 [](CodeGenFunction &, PrePostActionTy &) {}); 6449 } else { 6450 RegionCodeGenTy ThenRCG(ThenGen); 6451 ThenRCG(CGF); 6452 } 6453 } 6454 } 6455 6456 void CGOpenMPRuntime::emitTargetOutlinedFunction( 6457 const OMPExecutableDirective &D, StringRef ParentName, 6458 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6459 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6460 assert(!ParentName.empty() && "Invalid target region parent name!"); 6461 HasEmittedTargetRegion = true; 6462 emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID, 6463 IsOffloadEntry, CodeGen); 6464 } 6465 6466 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper( 6467 const OMPExecutableDirective &D, StringRef ParentName, 6468 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6469 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6470 // Create a unique name for the entry function using the source location 6471 // information of the current target region. The name will be something like: 6472 // 6473 // __omp_offloading_DD_FFFF_PP_lBB 6474 // 6475 // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the 6476 // mangled name of the function that encloses the target region and BB is the 6477 // line number of the target region. 6478 6479 unsigned DeviceID; 6480 unsigned FileID; 6481 unsigned Line; 6482 getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID, 6483 Line); 6484 SmallString<64> EntryFnName; 6485 { 6486 llvm::raw_svector_ostream OS(EntryFnName); 6487 OS << "__omp_offloading" << llvm::format("_%x", DeviceID) 6488 << llvm::format("_%x_", FileID) << ParentName << "_l" << Line; 6489 } 6490 6491 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 6492 6493 CodeGenFunction CGF(CGM, true); 6494 CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName); 6495 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6496 6497 OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS, D.getBeginLoc()); 6498 6499 // If this target outline function is not an offload entry, we don't need to 6500 // register it. 6501 if (!IsOffloadEntry) 6502 return; 6503 6504 // The target region ID is used by the runtime library to identify the current 6505 // target region, so it only has to be unique and not necessarily point to 6506 // anything. It could be the pointer to the outlined function that implements 6507 // the target region, but we aren't using that so that the compiler doesn't 6508 // need to keep that, and could therefore inline the host function if proven 6509 // worthwhile during optimization. In the other hand, if emitting code for the 6510 // device, the ID has to be the function address so that it can retrieved from 6511 // the offloading entry and launched by the runtime library. We also mark the 6512 // outlined function to have external linkage in case we are emitting code for 6513 // the device, because these functions will be entry points to the device. 6514 6515 if (CGM.getLangOpts().OpenMPIsDevice) { 6516 OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy); 6517 OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 6518 OutlinedFn->setDSOLocal(false); 6519 } else { 6520 std::string Name = getName({EntryFnName, "region_id"}); 6521 OutlinedFnID = new llvm::GlobalVariable( 6522 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 6523 llvm::GlobalValue::WeakAnyLinkage, 6524 llvm::Constant::getNullValue(CGM.Int8Ty), Name); 6525 } 6526 6527 // Register the information for the entry associated with this target region. 6528 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 6529 DeviceID, FileID, ParentName, Line, OutlinedFn, OutlinedFnID, 6530 OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion); 6531 } 6532 6533 /// Checks if the expression is constant or does not have non-trivial function 6534 /// calls. 6535 static bool isTrivial(ASTContext &Ctx, const Expr * E) { 6536 // We can skip constant expressions. 6537 // We can skip expressions with trivial calls or simple expressions. 6538 return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) || 6539 !E->hasNonTrivialCall(Ctx)) && 6540 !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true); 6541 } 6542 6543 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx, 6544 const Stmt *Body) { 6545 const Stmt *Child = Body->IgnoreContainers(); 6546 while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) { 6547 Child = nullptr; 6548 for (const Stmt *S : C->body()) { 6549 if (const auto *E = dyn_cast<Expr>(S)) { 6550 if (isTrivial(Ctx, E)) 6551 continue; 6552 } 6553 // Some of the statements can be ignored. 6554 if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) || 6555 isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S)) 6556 continue; 6557 // Analyze declarations. 6558 if (const auto *DS = dyn_cast<DeclStmt>(S)) { 6559 if (llvm::all_of(DS->decls(), [&Ctx](const Decl *D) { 6560 if (isa<EmptyDecl>(D) || isa<DeclContext>(D) || 6561 isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) || 6562 isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) || 6563 isa<UsingDirectiveDecl>(D) || 6564 isa<OMPDeclareReductionDecl>(D) || 6565 isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D)) 6566 return true; 6567 const auto *VD = dyn_cast<VarDecl>(D); 6568 if (!VD) 6569 return false; 6570 return VD->isConstexpr() || 6571 ((VD->getType().isTrivialType(Ctx) || 6572 VD->getType()->isReferenceType()) && 6573 (!VD->hasInit() || isTrivial(Ctx, VD->getInit()))); 6574 })) 6575 continue; 6576 } 6577 // Found multiple children - cannot get the one child only. 6578 if (Child) 6579 return nullptr; 6580 Child = S; 6581 } 6582 if (Child) 6583 Child = Child->IgnoreContainers(); 6584 } 6585 return Child; 6586 } 6587 6588 /// Emit the number of teams for a target directive. Inspect the num_teams 6589 /// clause associated with a teams construct combined or closely nested 6590 /// with the target directive. 6591 /// 6592 /// Emit a team of size one for directives such as 'target parallel' that 6593 /// have no associated teams construct. 6594 /// 6595 /// Otherwise, return nullptr. 6596 static llvm::Value * 6597 emitNumTeamsForTargetDirective(CodeGenFunction &CGF, 6598 const OMPExecutableDirective &D) { 6599 assert(!CGF.getLangOpts().OpenMPIsDevice && 6600 "Clauses associated with the teams directive expected to be emitted " 6601 "only for the host!"); 6602 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6603 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 6604 "Expected target-based executable directive."); 6605 CGBuilderTy &Bld = CGF.Builder; 6606 switch (DirectiveKind) { 6607 case OMPD_target: { 6608 const auto *CS = D.getInnermostCapturedStmt(); 6609 const auto *Body = 6610 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 6611 const Stmt *ChildStmt = 6612 CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body); 6613 if (const auto *NestedDir = 6614 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 6615 if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) { 6616 if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) { 6617 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6618 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6619 const Expr *NumTeams = 6620 NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6621 llvm::Value *NumTeamsVal = 6622 CGF.EmitScalarExpr(NumTeams, 6623 /*IgnoreResultAssign*/ true); 6624 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6625 /*isSigned=*/true); 6626 } 6627 return Bld.getInt32(0); 6628 } 6629 if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) || 6630 isOpenMPSimdDirective(NestedDir->getDirectiveKind())) 6631 return Bld.getInt32(1); 6632 return Bld.getInt32(0); 6633 } 6634 return nullptr; 6635 } 6636 case OMPD_target_teams: 6637 case OMPD_target_teams_distribute: 6638 case OMPD_target_teams_distribute_simd: 6639 case OMPD_target_teams_distribute_parallel_for: 6640 case OMPD_target_teams_distribute_parallel_for_simd: { 6641 if (D.hasClausesOfKind<OMPNumTeamsClause>()) { 6642 CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF); 6643 const Expr *NumTeams = 6644 D.getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6645 llvm::Value *NumTeamsVal = 6646 CGF.EmitScalarExpr(NumTeams, 6647 /*IgnoreResultAssign*/ true); 6648 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6649 /*isSigned=*/true); 6650 } 6651 return Bld.getInt32(0); 6652 } 6653 case OMPD_target_parallel: 6654 case OMPD_target_parallel_for: 6655 case OMPD_target_parallel_for_simd: 6656 case OMPD_target_simd: 6657 return Bld.getInt32(1); 6658 case OMPD_parallel: 6659 case OMPD_for: 6660 case OMPD_parallel_for: 6661 case OMPD_parallel_master: 6662 case OMPD_parallel_sections: 6663 case OMPD_for_simd: 6664 case OMPD_parallel_for_simd: 6665 case OMPD_cancel: 6666 case OMPD_cancellation_point: 6667 case OMPD_ordered: 6668 case OMPD_threadprivate: 6669 case OMPD_allocate: 6670 case OMPD_task: 6671 case OMPD_simd: 6672 case OMPD_sections: 6673 case OMPD_section: 6674 case OMPD_single: 6675 case OMPD_master: 6676 case OMPD_critical: 6677 case OMPD_taskyield: 6678 case OMPD_barrier: 6679 case OMPD_taskwait: 6680 case OMPD_taskgroup: 6681 case OMPD_atomic: 6682 case OMPD_flush: 6683 case OMPD_teams: 6684 case OMPD_target_data: 6685 case OMPD_target_exit_data: 6686 case OMPD_target_enter_data: 6687 case OMPD_distribute: 6688 case OMPD_distribute_simd: 6689 case OMPD_distribute_parallel_for: 6690 case OMPD_distribute_parallel_for_simd: 6691 case OMPD_teams_distribute: 6692 case OMPD_teams_distribute_simd: 6693 case OMPD_teams_distribute_parallel_for: 6694 case OMPD_teams_distribute_parallel_for_simd: 6695 case OMPD_target_update: 6696 case OMPD_declare_simd: 6697 case OMPD_declare_variant: 6698 case OMPD_declare_target: 6699 case OMPD_end_declare_target: 6700 case OMPD_declare_reduction: 6701 case OMPD_declare_mapper: 6702 case OMPD_taskloop: 6703 case OMPD_taskloop_simd: 6704 case OMPD_master_taskloop: 6705 case OMPD_master_taskloop_simd: 6706 case OMPD_parallel_master_taskloop: 6707 case OMPD_parallel_master_taskloop_simd: 6708 case OMPD_requires: 6709 case OMPD_unknown: 6710 break; 6711 } 6712 llvm_unreachable("Unexpected directive kind."); 6713 } 6714 6715 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS, 6716 llvm::Value *DefaultThreadLimitVal) { 6717 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6718 CGF.getContext(), CS->getCapturedStmt()); 6719 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6720 if (isOpenMPParallelDirective(Dir->getDirectiveKind())) { 6721 llvm::Value *NumThreads = nullptr; 6722 llvm::Value *CondVal = nullptr; 6723 // Handle if clause. If if clause present, the number of threads is 6724 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 6725 if (Dir->hasClausesOfKind<OMPIfClause>()) { 6726 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6727 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6728 const OMPIfClause *IfClause = nullptr; 6729 for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) { 6730 if (C->getNameModifier() == OMPD_unknown || 6731 C->getNameModifier() == OMPD_parallel) { 6732 IfClause = C; 6733 break; 6734 } 6735 } 6736 if (IfClause) { 6737 const Expr *Cond = IfClause->getCondition(); 6738 bool Result; 6739 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 6740 if (!Result) 6741 return CGF.Builder.getInt32(1); 6742 } else { 6743 CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange()); 6744 if (const auto *PreInit = 6745 cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) { 6746 for (const auto *I : PreInit->decls()) { 6747 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6748 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6749 } else { 6750 CodeGenFunction::AutoVarEmission Emission = 6751 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6752 CGF.EmitAutoVarCleanups(Emission); 6753 } 6754 } 6755 } 6756 CondVal = CGF.EvaluateExprAsBool(Cond); 6757 } 6758 } 6759 } 6760 // Check the value of num_threads clause iff if clause was not specified 6761 // or is not evaluated to false. 6762 if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) { 6763 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6764 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6765 const auto *NumThreadsClause = 6766 Dir->getSingleClause<OMPNumThreadsClause>(); 6767 CodeGenFunction::LexicalScope Scope( 6768 CGF, NumThreadsClause->getNumThreads()->getSourceRange()); 6769 if (const auto *PreInit = 6770 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) { 6771 for (const auto *I : PreInit->decls()) { 6772 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6773 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6774 } else { 6775 CodeGenFunction::AutoVarEmission Emission = 6776 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6777 CGF.EmitAutoVarCleanups(Emission); 6778 } 6779 } 6780 } 6781 NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads()); 6782 NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, 6783 /*isSigned=*/false); 6784 if (DefaultThreadLimitVal) 6785 NumThreads = CGF.Builder.CreateSelect( 6786 CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads), 6787 DefaultThreadLimitVal, NumThreads); 6788 } else { 6789 NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal 6790 : CGF.Builder.getInt32(0); 6791 } 6792 // Process condition of the if clause. 6793 if (CondVal) { 6794 NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads, 6795 CGF.Builder.getInt32(1)); 6796 } 6797 return NumThreads; 6798 } 6799 if (isOpenMPSimdDirective(Dir->getDirectiveKind())) 6800 return CGF.Builder.getInt32(1); 6801 return DefaultThreadLimitVal; 6802 } 6803 return DefaultThreadLimitVal ? DefaultThreadLimitVal 6804 : CGF.Builder.getInt32(0); 6805 } 6806 6807 /// Emit the number of threads for a target directive. Inspect the 6808 /// thread_limit clause associated with a teams construct combined or closely 6809 /// nested with the target directive. 6810 /// 6811 /// Emit the num_threads clause for directives such as 'target parallel' that 6812 /// have no associated teams construct. 6813 /// 6814 /// Otherwise, return nullptr. 6815 static llvm::Value * 6816 emitNumThreadsForTargetDirective(CodeGenFunction &CGF, 6817 const OMPExecutableDirective &D) { 6818 assert(!CGF.getLangOpts().OpenMPIsDevice && 6819 "Clauses associated with the teams directive expected to be emitted " 6820 "only for the host!"); 6821 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6822 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 6823 "Expected target-based executable directive."); 6824 CGBuilderTy &Bld = CGF.Builder; 6825 llvm::Value *ThreadLimitVal = nullptr; 6826 llvm::Value *NumThreadsVal = nullptr; 6827 switch (DirectiveKind) { 6828 case OMPD_target: { 6829 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 6830 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 6831 return NumThreads; 6832 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6833 CGF.getContext(), CS->getCapturedStmt()); 6834 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6835 if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) { 6836 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6837 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6838 const auto *ThreadLimitClause = 6839 Dir->getSingleClause<OMPThreadLimitClause>(); 6840 CodeGenFunction::LexicalScope Scope( 6841 CGF, ThreadLimitClause->getThreadLimit()->getSourceRange()); 6842 if (const auto *PreInit = 6843 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) { 6844 for (const auto *I : PreInit->decls()) { 6845 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6846 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6847 } else { 6848 CodeGenFunction::AutoVarEmission Emission = 6849 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6850 CGF.EmitAutoVarCleanups(Emission); 6851 } 6852 } 6853 } 6854 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 6855 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 6856 ThreadLimitVal = 6857 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 6858 } 6859 if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) && 6860 !isOpenMPDistributeDirective(Dir->getDirectiveKind())) { 6861 CS = Dir->getInnermostCapturedStmt(); 6862 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6863 CGF.getContext(), CS->getCapturedStmt()); 6864 Dir = dyn_cast_or_null<OMPExecutableDirective>(Child); 6865 } 6866 if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) && 6867 !isOpenMPSimdDirective(Dir->getDirectiveKind())) { 6868 CS = Dir->getInnermostCapturedStmt(); 6869 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 6870 return NumThreads; 6871 } 6872 if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind())) 6873 return Bld.getInt32(1); 6874 } 6875 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 6876 } 6877 case OMPD_target_teams: { 6878 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 6879 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 6880 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 6881 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 6882 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 6883 ThreadLimitVal = 6884 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 6885 } 6886 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 6887 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 6888 return NumThreads; 6889 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6890 CGF.getContext(), CS->getCapturedStmt()); 6891 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6892 if (Dir->getDirectiveKind() == OMPD_distribute) { 6893 CS = Dir->getInnermostCapturedStmt(); 6894 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 6895 return NumThreads; 6896 } 6897 } 6898 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 6899 } 6900 case OMPD_target_teams_distribute: 6901 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 6902 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 6903 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 6904 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 6905 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 6906 ThreadLimitVal = 6907 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 6908 } 6909 return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal); 6910 case OMPD_target_parallel: 6911 case OMPD_target_parallel_for: 6912 case OMPD_target_parallel_for_simd: 6913 case OMPD_target_teams_distribute_parallel_for: 6914 case OMPD_target_teams_distribute_parallel_for_simd: { 6915 llvm::Value *CondVal = nullptr; 6916 // Handle if clause. If if clause present, the number of threads is 6917 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 6918 if (D.hasClausesOfKind<OMPIfClause>()) { 6919 const OMPIfClause *IfClause = nullptr; 6920 for (const auto *C : D.getClausesOfKind<OMPIfClause>()) { 6921 if (C->getNameModifier() == OMPD_unknown || 6922 C->getNameModifier() == OMPD_parallel) { 6923 IfClause = C; 6924 break; 6925 } 6926 } 6927 if (IfClause) { 6928 const Expr *Cond = IfClause->getCondition(); 6929 bool Result; 6930 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 6931 if (!Result) 6932 return Bld.getInt32(1); 6933 } else { 6934 CodeGenFunction::RunCleanupsScope Scope(CGF); 6935 CondVal = CGF.EvaluateExprAsBool(Cond); 6936 } 6937 } 6938 } 6939 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 6940 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 6941 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 6942 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 6943 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 6944 ThreadLimitVal = 6945 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 6946 } 6947 if (D.hasClausesOfKind<OMPNumThreadsClause>()) { 6948 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF); 6949 const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>(); 6950 llvm::Value *NumThreads = CGF.EmitScalarExpr( 6951 NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true); 6952 NumThreadsVal = 6953 Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false); 6954 ThreadLimitVal = ThreadLimitVal 6955 ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal, 6956 ThreadLimitVal), 6957 NumThreadsVal, ThreadLimitVal) 6958 : NumThreadsVal; 6959 } 6960 if (!ThreadLimitVal) 6961 ThreadLimitVal = Bld.getInt32(0); 6962 if (CondVal) 6963 return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1)); 6964 return ThreadLimitVal; 6965 } 6966 case OMPD_target_teams_distribute_simd: 6967 case OMPD_target_simd: 6968 return Bld.getInt32(1); 6969 case OMPD_parallel: 6970 case OMPD_for: 6971 case OMPD_parallel_for: 6972 case OMPD_parallel_master: 6973 case OMPD_parallel_sections: 6974 case OMPD_for_simd: 6975 case OMPD_parallel_for_simd: 6976 case OMPD_cancel: 6977 case OMPD_cancellation_point: 6978 case OMPD_ordered: 6979 case OMPD_threadprivate: 6980 case OMPD_allocate: 6981 case OMPD_task: 6982 case OMPD_simd: 6983 case OMPD_sections: 6984 case OMPD_section: 6985 case OMPD_single: 6986 case OMPD_master: 6987 case OMPD_critical: 6988 case OMPD_taskyield: 6989 case OMPD_barrier: 6990 case OMPD_taskwait: 6991 case OMPD_taskgroup: 6992 case OMPD_atomic: 6993 case OMPD_flush: 6994 case OMPD_teams: 6995 case OMPD_target_data: 6996 case OMPD_target_exit_data: 6997 case OMPD_target_enter_data: 6998 case OMPD_distribute: 6999 case OMPD_distribute_simd: 7000 case OMPD_distribute_parallel_for: 7001 case OMPD_distribute_parallel_for_simd: 7002 case OMPD_teams_distribute: 7003 case OMPD_teams_distribute_simd: 7004 case OMPD_teams_distribute_parallel_for: 7005 case OMPD_teams_distribute_parallel_for_simd: 7006 case OMPD_target_update: 7007 case OMPD_declare_simd: 7008 case OMPD_declare_variant: 7009 case OMPD_declare_target: 7010 case OMPD_end_declare_target: 7011 case OMPD_declare_reduction: 7012 case OMPD_declare_mapper: 7013 case OMPD_taskloop: 7014 case OMPD_taskloop_simd: 7015 case OMPD_master_taskloop: 7016 case OMPD_master_taskloop_simd: 7017 case OMPD_parallel_master_taskloop: 7018 case OMPD_parallel_master_taskloop_simd: 7019 case OMPD_requires: 7020 case OMPD_unknown: 7021 break; 7022 } 7023 llvm_unreachable("Unsupported directive kind."); 7024 } 7025 7026 namespace { 7027 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 7028 7029 // Utility to handle information from clauses associated with a given 7030 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause). 7031 // It provides a convenient interface to obtain the information and generate 7032 // code for that information. 7033 class MappableExprsHandler { 7034 public: 7035 /// Values for bit flags used to specify the mapping type for 7036 /// offloading. 7037 enum OpenMPOffloadMappingFlags : uint64_t { 7038 /// No flags 7039 OMP_MAP_NONE = 0x0, 7040 /// Allocate memory on the device and move data from host to device. 7041 OMP_MAP_TO = 0x01, 7042 /// Allocate memory on the device and move data from device to host. 7043 OMP_MAP_FROM = 0x02, 7044 /// Always perform the requested mapping action on the element, even 7045 /// if it was already mapped before. 7046 OMP_MAP_ALWAYS = 0x04, 7047 /// Delete the element from the device environment, ignoring the 7048 /// current reference count associated with the element. 7049 OMP_MAP_DELETE = 0x08, 7050 /// The element being mapped is a pointer-pointee pair; both the 7051 /// pointer and the pointee should be mapped. 7052 OMP_MAP_PTR_AND_OBJ = 0x10, 7053 /// This flags signals that the base address of an entry should be 7054 /// passed to the target kernel as an argument. 7055 OMP_MAP_TARGET_PARAM = 0x20, 7056 /// Signal that the runtime library has to return the device pointer 7057 /// in the current position for the data being mapped. Used when we have the 7058 /// use_device_ptr clause. 7059 OMP_MAP_RETURN_PARAM = 0x40, 7060 /// This flag signals that the reference being passed is a pointer to 7061 /// private data. 7062 OMP_MAP_PRIVATE = 0x80, 7063 /// Pass the element to the device by value. 7064 OMP_MAP_LITERAL = 0x100, 7065 /// Implicit map 7066 OMP_MAP_IMPLICIT = 0x200, 7067 /// Close is a hint to the runtime to allocate memory close to 7068 /// the target device. 7069 OMP_MAP_CLOSE = 0x400, 7070 /// The 16 MSBs of the flags indicate whether the entry is member of some 7071 /// struct/class. 7072 OMP_MAP_MEMBER_OF = 0xffff000000000000, 7073 LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF), 7074 }; 7075 7076 /// Get the offset of the OMP_MAP_MEMBER_OF field. 7077 static unsigned getFlagMemberOffset() { 7078 unsigned Offset = 0; 7079 for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1); 7080 Remain = Remain >> 1) 7081 Offset++; 7082 return Offset; 7083 } 7084 7085 /// Class that associates information with a base pointer to be passed to the 7086 /// runtime library. 7087 class BasePointerInfo { 7088 /// The base pointer. 7089 llvm::Value *Ptr = nullptr; 7090 /// The base declaration that refers to this device pointer, or null if 7091 /// there is none. 7092 const ValueDecl *DevPtrDecl = nullptr; 7093 7094 public: 7095 BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr) 7096 : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {} 7097 llvm::Value *operator*() const { return Ptr; } 7098 const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; } 7099 void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; } 7100 }; 7101 7102 using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>; 7103 using MapValuesArrayTy = SmallVector<llvm::Value *, 4>; 7104 using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>; 7105 7106 /// Map between a struct and the its lowest & highest elements which have been 7107 /// mapped. 7108 /// [ValueDecl *] --> {LE(FieldIndex, Pointer), 7109 /// HE(FieldIndex, Pointer)} 7110 struct StructRangeInfoTy { 7111 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = { 7112 0, Address::invalid()}; 7113 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = { 7114 0, Address::invalid()}; 7115 Address Base = Address::invalid(); 7116 }; 7117 7118 private: 7119 /// Kind that defines how a device pointer has to be returned. 7120 struct MapInfo { 7121 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 7122 OpenMPMapClauseKind MapType = OMPC_MAP_unknown; 7123 ArrayRef<OpenMPMapModifierKind> MapModifiers; 7124 bool ReturnDevicePointer = false; 7125 bool IsImplicit = false; 7126 7127 MapInfo() = default; 7128 MapInfo( 7129 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7130 OpenMPMapClauseKind MapType, 7131 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7132 bool ReturnDevicePointer, bool IsImplicit) 7133 : Components(Components), MapType(MapType), MapModifiers(MapModifiers), 7134 ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit) {} 7135 }; 7136 7137 /// If use_device_ptr is used on a pointer which is a struct member and there 7138 /// is no map information about it, then emission of that entry is deferred 7139 /// until the whole struct has been processed. 7140 struct DeferredDevicePtrEntryTy { 7141 const Expr *IE = nullptr; 7142 const ValueDecl *VD = nullptr; 7143 7144 DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD) 7145 : IE(IE), VD(VD) {} 7146 }; 7147 7148 /// The target directive from where the mappable clauses were extracted. It 7149 /// is either a executable directive or a user-defined mapper directive. 7150 llvm::PointerUnion<const OMPExecutableDirective *, 7151 const OMPDeclareMapperDecl *> 7152 CurDir; 7153 7154 /// Function the directive is being generated for. 7155 CodeGenFunction &CGF; 7156 7157 /// Set of all first private variables in the current directive. 7158 /// bool data is set to true if the variable is implicitly marked as 7159 /// firstprivate, false otherwise. 7160 llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls; 7161 7162 /// Map between device pointer declarations and their expression components. 7163 /// The key value for declarations in 'this' is null. 7164 llvm::DenseMap< 7165 const ValueDecl *, 7166 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>> 7167 DevPointersMap; 7168 7169 llvm::Value *getExprTypeSize(const Expr *E) const { 7170 QualType ExprTy = E->getType().getCanonicalType(); 7171 7172 // Reference types are ignored for mapping purposes. 7173 if (const auto *RefTy = ExprTy->getAs<ReferenceType>()) 7174 ExprTy = RefTy->getPointeeType().getCanonicalType(); 7175 7176 // Given that an array section is considered a built-in type, we need to 7177 // do the calculation based on the length of the section instead of relying 7178 // on CGF.getTypeSize(E->getType()). 7179 if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) { 7180 QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType( 7181 OAE->getBase()->IgnoreParenImpCasts()) 7182 .getCanonicalType(); 7183 7184 // If there is no length associated with the expression and lower bound is 7185 // not specified too, that means we are using the whole length of the 7186 // base. 7187 if (!OAE->getLength() && OAE->getColonLoc().isValid() && 7188 !OAE->getLowerBound()) 7189 return CGF.getTypeSize(BaseTy); 7190 7191 llvm::Value *ElemSize; 7192 if (const auto *PTy = BaseTy->getAs<PointerType>()) { 7193 ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType()); 7194 } else { 7195 const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr()); 7196 assert(ATy && "Expecting array type if not a pointer type."); 7197 ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType()); 7198 } 7199 7200 // If we don't have a length at this point, that is because we have an 7201 // array section with a single element. 7202 if (!OAE->getLength() && OAE->getColonLoc().isInvalid()) 7203 return ElemSize; 7204 7205 if (const Expr *LenExpr = OAE->getLength()) { 7206 llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr); 7207 LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(), 7208 CGF.getContext().getSizeType(), 7209 LenExpr->getExprLoc()); 7210 return CGF.Builder.CreateNUWMul(LengthVal, ElemSize); 7211 } 7212 assert(!OAE->getLength() && OAE->getColonLoc().isValid() && 7213 OAE->getLowerBound() && "expected array_section[lb:]."); 7214 // Size = sizetype - lb * elemtype; 7215 llvm::Value *LengthVal = CGF.getTypeSize(BaseTy); 7216 llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound()); 7217 LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(), 7218 CGF.getContext().getSizeType(), 7219 OAE->getLowerBound()->getExprLoc()); 7220 LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize); 7221 llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal); 7222 llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal); 7223 LengthVal = CGF.Builder.CreateSelect( 7224 Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0)); 7225 return LengthVal; 7226 } 7227 return CGF.getTypeSize(ExprTy); 7228 } 7229 7230 /// Return the corresponding bits for a given map clause modifier. Add 7231 /// a flag marking the map as a pointer if requested. Add a flag marking the 7232 /// map as the first one of a series of maps that relate to the same map 7233 /// expression. 7234 OpenMPOffloadMappingFlags getMapTypeBits( 7235 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers, 7236 bool IsImplicit, bool AddPtrFlag, bool AddIsTargetParamFlag) const { 7237 OpenMPOffloadMappingFlags Bits = 7238 IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE; 7239 switch (MapType) { 7240 case OMPC_MAP_alloc: 7241 case OMPC_MAP_release: 7242 // alloc and release is the default behavior in the runtime library, i.e. 7243 // if we don't pass any bits alloc/release that is what the runtime is 7244 // going to do. Therefore, we don't need to signal anything for these two 7245 // type modifiers. 7246 break; 7247 case OMPC_MAP_to: 7248 Bits |= OMP_MAP_TO; 7249 break; 7250 case OMPC_MAP_from: 7251 Bits |= OMP_MAP_FROM; 7252 break; 7253 case OMPC_MAP_tofrom: 7254 Bits |= OMP_MAP_TO | OMP_MAP_FROM; 7255 break; 7256 case OMPC_MAP_delete: 7257 Bits |= OMP_MAP_DELETE; 7258 break; 7259 case OMPC_MAP_unknown: 7260 llvm_unreachable("Unexpected map type!"); 7261 } 7262 if (AddPtrFlag) 7263 Bits |= OMP_MAP_PTR_AND_OBJ; 7264 if (AddIsTargetParamFlag) 7265 Bits |= OMP_MAP_TARGET_PARAM; 7266 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_always) 7267 != MapModifiers.end()) 7268 Bits |= OMP_MAP_ALWAYS; 7269 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_close) 7270 != MapModifiers.end()) 7271 Bits |= OMP_MAP_CLOSE; 7272 return Bits; 7273 } 7274 7275 /// Return true if the provided expression is a final array section. A 7276 /// final array section, is one whose length can't be proved to be one. 7277 bool isFinalArraySectionExpression(const Expr *E) const { 7278 const auto *OASE = dyn_cast<OMPArraySectionExpr>(E); 7279 7280 // It is not an array section and therefore not a unity-size one. 7281 if (!OASE) 7282 return false; 7283 7284 // An array section with no colon always refer to a single element. 7285 if (OASE->getColonLoc().isInvalid()) 7286 return false; 7287 7288 const Expr *Length = OASE->getLength(); 7289 7290 // If we don't have a length we have to check if the array has size 1 7291 // for this dimension. Also, we should always expect a length if the 7292 // base type is pointer. 7293 if (!Length) { 7294 QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType( 7295 OASE->getBase()->IgnoreParenImpCasts()) 7296 .getCanonicalType(); 7297 if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr())) 7298 return ATy->getSize().getSExtValue() != 1; 7299 // If we don't have a constant dimension length, we have to consider 7300 // the current section as having any size, so it is not necessarily 7301 // unitary. If it happen to be unity size, that's user fault. 7302 return true; 7303 } 7304 7305 // Check if the length evaluates to 1. 7306 Expr::EvalResult Result; 7307 if (!Length->EvaluateAsInt(Result, CGF.getContext())) 7308 return true; // Can have more that size 1. 7309 7310 llvm::APSInt ConstLength = Result.Val.getInt(); 7311 return ConstLength.getSExtValue() != 1; 7312 } 7313 7314 /// Generate the base pointers, section pointers, sizes and map type 7315 /// bits for the provided map type, map modifier, and expression components. 7316 /// \a IsFirstComponent should be set to true if the provided set of 7317 /// components is the first associated with a capture. 7318 void generateInfoForComponentList( 7319 OpenMPMapClauseKind MapType, 7320 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7321 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7322 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 7323 MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types, 7324 StructRangeInfoTy &PartialStruct, bool IsFirstComponentList, 7325 bool IsImplicit, 7326 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 7327 OverlappedElements = llvm::None) const { 7328 // The following summarizes what has to be generated for each map and the 7329 // types below. The generated information is expressed in this order: 7330 // base pointer, section pointer, size, flags 7331 // (to add to the ones that come from the map type and modifier). 7332 // 7333 // double d; 7334 // int i[100]; 7335 // float *p; 7336 // 7337 // struct S1 { 7338 // int i; 7339 // float f[50]; 7340 // } 7341 // struct S2 { 7342 // int i; 7343 // float f[50]; 7344 // S1 s; 7345 // double *p; 7346 // struct S2 *ps; 7347 // } 7348 // S2 s; 7349 // S2 *ps; 7350 // 7351 // map(d) 7352 // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM 7353 // 7354 // map(i) 7355 // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM 7356 // 7357 // map(i[1:23]) 7358 // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM 7359 // 7360 // map(p) 7361 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM 7362 // 7363 // map(p[1:24]) 7364 // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM 7365 // 7366 // map(s) 7367 // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM 7368 // 7369 // map(s.i) 7370 // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM 7371 // 7372 // map(s.s.f) 7373 // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7374 // 7375 // map(s.p) 7376 // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM 7377 // 7378 // map(to: s.p[:22]) 7379 // &s, &(s.p), sizeof(double*), TARGET_PARAM (*) 7380 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**) 7381 // &(s.p), &(s.p[0]), 22*sizeof(double), 7382 // MEMBER_OF(1) | PTR_AND_OBJ | TO (***) 7383 // (*) alloc space for struct members, only this is a target parameter 7384 // (**) map the pointer (nothing to be mapped in this example) (the compiler 7385 // optimizes this entry out, same in the examples below) 7386 // (***) map the pointee (map: to) 7387 // 7388 // map(s.ps) 7389 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7390 // 7391 // map(from: s.ps->s.i) 7392 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7393 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7394 // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7395 // 7396 // map(to: s.ps->ps) 7397 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7398 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7399 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | TO 7400 // 7401 // map(s.ps->ps->ps) 7402 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7403 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7404 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7405 // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7406 // 7407 // map(to: s.ps->ps->s.f[:22]) 7408 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7409 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7410 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7411 // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7412 // 7413 // map(ps) 7414 // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM 7415 // 7416 // map(ps->i) 7417 // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM 7418 // 7419 // map(ps->s.f) 7420 // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7421 // 7422 // map(from: ps->p) 7423 // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM 7424 // 7425 // map(to: ps->p[:22]) 7426 // ps, &(ps->p), sizeof(double*), TARGET_PARAM 7427 // ps, &(ps->p), sizeof(double*), MEMBER_OF(1) 7428 // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO 7429 // 7430 // map(ps->ps) 7431 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7432 // 7433 // map(from: ps->ps->s.i) 7434 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7435 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7436 // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7437 // 7438 // map(from: ps->ps->ps) 7439 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7440 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7441 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7442 // 7443 // map(ps->ps->ps->ps) 7444 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7445 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7446 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7447 // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7448 // 7449 // map(to: ps->ps->ps->s.f[:22]) 7450 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7451 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7452 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7453 // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7454 // 7455 // map(to: s.f[:22]) map(from: s.p[:33]) 7456 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) + 7457 // sizeof(double*) (**), TARGET_PARAM 7458 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO 7459 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) 7460 // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7461 // (*) allocate contiguous space needed to fit all mapped members even if 7462 // we allocate space for members not mapped (in this example, 7463 // s.f[22..49] and s.s are not mapped, yet we must allocate space for 7464 // them as well because they fall between &s.f[0] and &s.p) 7465 // 7466 // map(from: s.f[:22]) map(to: ps->p[:33]) 7467 // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM 7468 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7469 // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*) 7470 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO 7471 // (*) the struct this entry pertains to is the 2nd element in the list of 7472 // arguments, hence MEMBER_OF(2) 7473 // 7474 // map(from: s.f[:22], s.s) map(to: ps->p[:33]) 7475 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM 7476 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM 7477 // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM 7478 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7479 // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*) 7480 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO 7481 // (*) the struct this entry pertains to is the 4th element in the list 7482 // of arguments, hence MEMBER_OF(4) 7483 7484 // Track if the map information being generated is the first for a capture. 7485 bool IsCaptureFirstInfo = IsFirstComponentList; 7486 // When the variable is on a declare target link or in a to clause with 7487 // unified memory, a reference is needed to hold the host/device address 7488 // of the variable. 7489 bool RequiresReference = false; 7490 7491 // Scan the components from the base to the complete expression. 7492 auto CI = Components.rbegin(); 7493 auto CE = Components.rend(); 7494 auto I = CI; 7495 7496 // Track if the map information being generated is the first for a list of 7497 // components. 7498 bool IsExpressionFirstInfo = true; 7499 Address BP = Address::invalid(); 7500 const Expr *AssocExpr = I->getAssociatedExpression(); 7501 const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr); 7502 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 7503 7504 if (isa<MemberExpr>(AssocExpr)) { 7505 // The base is the 'this' pointer. The content of the pointer is going 7506 // to be the base of the field being mapped. 7507 BP = CGF.LoadCXXThisAddress(); 7508 } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) || 7509 (OASE && 7510 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) { 7511 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF); 7512 } else { 7513 // The base is the reference to the variable. 7514 // BP = &Var. 7515 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF); 7516 if (const auto *VD = 7517 dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) { 7518 if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 7519 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) { 7520 if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 7521 (*Res == OMPDeclareTargetDeclAttr::MT_To && 7522 CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) { 7523 RequiresReference = true; 7524 BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 7525 } 7526 } 7527 } 7528 7529 // If the variable is a pointer and is being dereferenced (i.e. is not 7530 // the last component), the base has to be the pointer itself, not its 7531 // reference. References are ignored for mapping purposes. 7532 QualType Ty = 7533 I->getAssociatedDeclaration()->getType().getNonReferenceType(); 7534 if (Ty->isAnyPointerType() && std::next(I) != CE) { 7535 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>()); 7536 7537 // We do not need to generate individual map information for the 7538 // pointer, it can be associated with the combined storage. 7539 ++I; 7540 } 7541 } 7542 7543 // Track whether a component of the list should be marked as MEMBER_OF some 7544 // combined entry (for partial structs). Only the first PTR_AND_OBJ entry 7545 // in a component list should be marked as MEMBER_OF, all subsequent entries 7546 // do not belong to the base struct. E.g. 7547 // struct S2 s; 7548 // s.ps->ps->ps->f[:] 7549 // (1) (2) (3) (4) 7550 // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a 7551 // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3) 7552 // is the pointee of ps(2) which is not member of struct s, so it should not 7553 // be marked as such (it is still PTR_AND_OBJ). 7554 // The variable is initialized to false so that PTR_AND_OBJ entries which 7555 // are not struct members are not considered (e.g. array of pointers to 7556 // data). 7557 bool ShouldBeMemberOf = false; 7558 7559 // Variable keeping track of whether or not we have encountered a component 7560 // in the component list which is a member expression. Useful when we have a 7561 // pointer or a final array section, in which case it is the previous 7562 // component in the list which tells us whether we have a member expression. 7563 // E.g. X.f[:] 7564 // While processing the final array section "[:]" it is "f" which tells us 7565 // whether we are dealing with a member of a declared struct. 7566 const MemberExpr *EncounteredME = nullptr; 7567 7568 for (; I != CE; ++I) { 7569 // If the current component is member of a struct (parent struct) mark it. 7570 if (!EncounteredME) { 7571 EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression()); 7572 // If we encounter a PTR_AND_OBJ entry from now on it should be marked 7573 // as MEMBER_OF the parent struct. 7574 if (EncounteredME) 7575 ShouldBeMemberOf = true; 7576 } 7577 7578 auto Next = std::next(I); 7579 7580 // We need to generate the addresses and sizes if this is the last 7581 // component, if the component is a pointer or if it is an array section 7582 // whose length can't be proved to be one. If this is a pointer, it 7583 // becomes the base address for the following components. 7584 7585 // A final array section, is one whose length can't be proved to be one. 7586 bool IsFinalArraySection = 7587 isFinalArraySectionExpression(I->getAssociatedExpression()); 7588 7589 // Get information on whether the element is a pointer. Have to do a 7590 // special treatment for array sections given that they are built-in 7591 // types. 7592 const auto *OASE = 7593 dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression()); 7594 bool IsPointer = 7595 (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE) 7596 .getCanonicalType() 7597 ->isAnyPointerType()) || 7598 I->getAssociatedExpression()->getType()->isAnyPointerType(); 7599 7600 if (Next == CE || IsPointer || IsFinalArraySection) { 7601 // If this is not the last component, we expect the pointer to be 7602 // associated with an array expression or member expression. 7603 assert((Next == CE || 7604 isa<MemberExpr>(Next->getAssociatedExpression()) || 7605 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) || 7606 isa<OMPArraySectionExpr>(Next->getAssociatedExpression())) && 7607 "Unexpected expression"); 7608 7609 Address LB = CGF.EmitOMPSharedLValue(I->getAssociatedExpression()) 7610 .getAddress(CGF); 7611 7612 // If this component is a pointer inside the base struct then we don't 7613 // need to create any entry for it - it will be combined with the object 7614 // it is pointing to into a single PTR_AND_OBJ entry. 7615 bool IsMemberPointer = 7616 IsPointer && EncounteredME && 7617 (dyn_cast<MemberExpr>(I->getAssociatedExpression()) == 7618 EncounteredME); 7619 if (!OverlappedElements.empty()) { 7620 // Handle base element with the info for overlapped elements. 7621 assert(!PartialStruct.Base.isValid() && "The base element is set."); 7622 assert(Next == CE && 7623 "Expected last element for the overlapped elements."); 7624 assert(!IsPointer && 7625 "Unexpected base element with the pointer type."); 7626 // Mark the whole struct as the struct that requires allocation on the 7627 // device. 7628 PartialStruct.LowestElem = {0, LB}; 7629 CharUnits TypeSize = CGF.getContext().getTypeSizeInChars( 7630 I->getAssociatedExpression()->getType()); 7631 Address HB = CGF.Builder.CreateConstGEP( 7632 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(LB, 7633 CGF.VoidPtrTy), 7634 TypeSize.getQuantity() - 1); 7635 PartialStruct.HighestElem = { 7636 std::numeric_limits<decltype( 7637 PartialStruct.HighestElem.first)>::max(), 7638 HB}; 7639 PartialStruct.Base = BP; 7640 // Emit data for non-overlapped data. 7641 OpenMPOffloadMappingFlags Flags = 7642 OMP_MAP_MEMBER_OF | 7643 getMapTypeBits(MapType, MapModifiers, IsImplicit, 7644 /*AddPtrFlag=*/false, 7645 /*AddIsTargetParamFlag=*/false); 7646 LB = BP; 7647 llvm::Value *Size = nullptr; 7648 // Do bitcopy of all non-overlapped structure elements. 7649 for (OMPClauseMappableExprCommon::MappableExprComponentListRef 7650 Component : OverlappedElements) { 7651 Address ComponentLB = Address::invalid(); 7652 for (const OMPClauseMappableExprCommon::MappableComponent &MC : 7653 Component) { 7654 if (MC.getAssociatedDeclaration()) { 7655 ComponentLB = 7656 CGF.EmitOMPSharedLValue(MC.getAssociatedExpression()) 7657 .getAddress(CGF); 7658 Size = CGF.Builder.CreatePtrDiff( 7659 CGF.EmitCastToVoidPtr(ComponentLB.getPointer()), 7660 CGF.EmitCastToVoidPtr(LB.getPointer())); 7661 break; 7662 } 7663 } 7664 BasePointers.push_back(BP.getPointer()); 7665 Pointers.push_back(LB.getPointer()); 7666 Sizes.push_back(CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, 7667 /*isSigned=*/true)); 7668 Types.push_back(Flags); 7669 LB = CGF.Builder.CreateConstGEP(ComponentLB, 1); 7670 } 7671 BasePointers.push_back(BP.getPointer()); 7672 Pointers.push_back(LB.getPointer()); 7673 Size = CGF.Builder.CreatePtrDiff( 7674 CGF.EmitCastToVoidPtr( 7675 CGF.Builder.CreateConstGEP(HB, 1).getPointer()), 7676 CGF.EmitCastToVoidPtr(LB.getPointer())); 7677 Sizes.push_back( 7678 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 7679 Types.push_back(Flags); 7680 break; 7681 } 7682 llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression()); 7683 if (!IsMemberPointer) { 7684 BasePointers.push_back(BP.getPointer()); 7685 Pointers.push_back(LB.getPointer()); 7686 Sizes.push_back( 7687 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 7688 7689 // We need to add a pointer flag for each map that comes from the 7690 // same expression except for the first one. We also need to signal 7691 // this map is the first one that relates with the current capture 7692 // (there is a set of entries for each capture). 7693 OpenMPOffloadMappingFlags Flags = getMapTypeBits( 7694 MapType, MapModifiers, IsImplicit, 7695 !IsExpressionFirstInfo || RequiresReference, 7696 IsCaptureFirstInfo && !RequiresReference); 7697 7698 if (!IsExpressionFirstInfo) { 7699 // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well, 7700 // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags. 7701 if (IsPointer) 7702 Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS | 7703 OMP_MAP_DELETE | OMP_MAP_CLOSE); 7704 7705 if (ShouldBeMemberOf) { 7706 // Set placeholder value MEMBER_OF=FFFF to indicate that the flag 7707 // should be later updated with the correct value of MEMBER_OF. 7708 Flags |= OMP_MAP_MEMBER_OF; 7709 // From now on, all subsequent PTR_AND_OBJ entries should not be 7710 // marked as MEMBER_OF. 7711 ShouldBeMemberOf = false; 7712 } 7713 } 7714 7715 Types.push_back(Flags); 7716 } 7717 7718 // If we have encountered a member expression so far, keep track of the 7719 // mapped member. If the parent is "*this", then the value declaration 7720 // is nullptr. 7721 if (EncounteredME) { 7722 const auto *FD = dyn_cast<FieldDecl>(EncounteredME->getMemberDecl()); 7723 unsigned FieldIndex = FD->getFieldIndex(); 7724 7725 // Update info about the lowest and highest elements for this struct 7726 if (!PartialStruct.Base.isValid()) { 7727 PartialStruct.LowestElem = {FieldIndex, LB}; 7728 PartialStruct.HighestElem = {FieldIndex, LB}; 7729 PartialStruct.Base = BP; 7730 } else if (FieldIndex < PartialStruct.LowestElem.first) { 7731 PartialStruct.LowestElem = {FieldIndex, LB}; 7732 } else if (FieldIndex > PartialStruct.HighestElem.first) { 7733 PartialStruct.HighestElem = {FieldIndex, LB}; 7734 } 7735 } 7736 7737 // If we have a final array section, we are done with this expression. 7738 if (IsFinalArraySection) 7739 break; 7740 7741 // The pointer becomes the base for the next element. 7742 if (Next != CE) 7743 BP = LB; 7744 7745 IsExpressionFirstInfo = false; 7746 IsCaptureFirstInfo = false; 7747 } 7748 } 7749 } 7750 7751 /// Return the adjusted map modifiers if the declaration a capture refers to 7752 /// appears in a first-private clause. This is expected to be used only with 7753 /// directives that start with 'target'. 7754 MappableExprsHandler::OpenMPOffloadMappingFlags 7755 getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const { 7756 assert(Cap.capturesVariable() && "Expected capture by reference only!"); 7757 7758 // A first private variable captured by reference will use only the 7759 // 'private ptr' and 'map to' flag. Return the right flags if the captured 7760 // declaration is known as first-private in this handler. 7761 if (FirstPrivateDecls.count(Cap.getCapturedVar())) { 7762 if (Cap.getCapturedVar()->getType().isConstant(CGF.getContext()) && 7763 Cap.getCaptureKind() == CapturedStmt::VCK_ByRef) 7764 return MappableExprsHandler::OMP_MAP_ALWAYS | 7765 MappableExprsHandler::OMP_MAP_TO; 7766 if (Cap.getCapturedVar()->getType()->isAnyPointerType()) 7767 return MappableExprsHandler::OMP_MAP_TO | 7768 MappableExprsHandler::OMP_MAP_PTR_AND_OBJ; 7769 return MappableExprsHandler::OMP_MAP_PRIVATE | 7770 MappableExprsHandler::OMP_MAP_TO; 7771 } 7772 return MappableExprsHandler::OMP_MAP_TO | 7773 MappableExprsHandler::OMP_MAP_FROM; 7774 } 7775 7776 static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) { 7777 // Rotate by getFlagMemberOffset() bits. 7778 return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1) 7779 << getFlagMemberOffset()); 7780 } 7781 7782 static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags, 7783 OpenMPOffloadMappingFlags MemberOfFlag) { 7784 // If the entry is PTR_AND_OBJ but has not been marked with the special 7785 // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be 7786 // marked as MEMBER_OF. 7787 if ((Flags & OMP_MAP_PTR_AND_OBJ) && 7788 ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF)) 7789 return; 7790 7791 // Reset the placeholder value to prepare the flag for the assignment of the 7792 // proper MEMBER_OF value. 7793 Flags &= ~OMP_MAP_MEMBER_OF; 7794 Flags |= MemberOfFlag; 7795 } 7796 7797 void getPlainLayout(const CXXRecordDecl *RD, 7798 llvm::SmallVectorImpl<const FieldDecl *> &Layout, 7799 bool AsBase) const { 7800 const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD); 7801 7802 llvm::StructType *St = 7803 AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType(); 7804 7805 unsigned NumElements = St->getNumElements(); 7806 llvm::SmallVector< 7807 llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4> 7808 RecordLayout(NumElements); 7809 7810 // Fill bases. 7811 for (const auto &I : RD->bases()) { 7812 if (I.isVirtual()) 7813 continue; 7814 const auto *Base = I.getType()->getAsCXXRecordDecl(); 7815 // Ignore empty bases. 7816 if (Base->isEmpty() || CGF.getContext() 7817 .getASTRecordLayout(Base) 7818 .getNonVirtualSize() 7819 .isZero()) 7820 continue; 7821 7822 unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base); 7823 RecordLayout[FieldIndex] = Base; 7824 } 7825 // Fill in virtual bases. 7826 for (const auto &I : RD->vbases()) { 7827 const auto *Base = I.getType()->getAsCXXRecordDecl(); 7828 // Ignore empty bases. 7829 if (Base->isEmpty()) 7830 continue; 7831 unsigned FieldIndex = RL.getVirtualBaseIndex(Base); 7832 if (RecordLayout[FieldIndex]) 7833 continue; 7834 RecordLayout[FieldIndex] = Base; 7835 } 7836 // Fill in all the fields. 7837 assert(!RD->isUnion() && "Unexpected union."); 7838 for (const auto *Field : RD->fields()) { 7839 // Fill in non-bitfields. (Bitfields always use a zero pattern, which we 7840 // will fill in later.) 7841 if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) { 7842 unsigned FieldIndex = RL.getLLVMFieldNo(Field); 7843 RecordLayout[FieldIndex] = Field; 7844 } 7845 } 7846 for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *> 7847 &Data : RecordLayout) { 7848 if (Data.isNull()) 7849 continue; 7850 if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>()) 7851 getPlainLayout(Base, Layout, /*AsBase=*/true); 7852 else 7853 Layout.push_back(Data.get<const FieldDecl *>()); 7854 } 7855 } 7856 7857 public: 7858 MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF) 7859 : CurDir(&Dir), CGF(CGF) { 7860 // Extract firstprivate clause information. 7861 for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>()) 7862 for (const auto *D : C->varlists()) 7863 FirstPrivateDecls.try_emplace( 7864 cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit()); 7865 // Extract device pointer clause information. 7866 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>()) 7867 for (auto L : C->component_lists()) 7868 DevPointersMap[L.first].push_back(L.second); 7869 } 7870 7871 /// Constructor for the declare mapper directive. 7872 MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF) 7873 : CurDir(&Dir), CGF(CGF) {} 7874 7875 /// Generate code for the combined entry if we have a partially mapped struct 7876 /// and take care of the mapping flags of the arguments corresponding to 7877 /// individual struct members. 7878 void emitCombinedEntry(MapBaseValuesArrayTy &BasePointers, 7879 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 7880 MapFlagsArrayTy &Types, MapFlagsArrayTy &CurTypes, 7881 const StructRangeInfoTy &PartialStruct) const { 7882 // Base is the base of the struct 7883 BasePointers.push_back(PartialStruct.Base.getPointer()); 7884 // Pointer is the address of the lowest element 7885 llvm::Value *LB = PartialStruct.LowestElem.second.getPointer(); 7886 Pointers.push_back(LB); 7887 // Size is (addr of {highest+1} element) - (addr of lowest element) 7888 llvm::Value *HB = PartialStruct.HighestElem.second.getPointer(); 7889 llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(HB, /*Idx0=*/1); 7890 llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy); 7891 llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy); 7892 llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr); 7893 llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty, 7894 /*isSigned=*/false); 7895 Sizes.push_back(Size); 7896 // Map type is always TARGET_PARAM 7897 Types.push_back(OMP_MAP_TARGET_PARAM); 7898 // Remove TARGET_PARAM flag from the first element 7899 (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM; 7900 7901 // All other current entries will be MEMBER_OF the combined entry 7902 // (except for PTR_AND_OBJ entries which do not have a placeholder value 7903 // 0xFFFF in the MEMBER_OF field). 7904 OpenMPOffloadMappingFlags MemberOfFlag = 7905 getMemberOfFlag(BasePointers.size() - 1); 7906 for (auto &M : CurTypes) 7907 setCorrectMemberOfFlag(M, MemberOfFlag); 7908 } 7909 7910 /// Generate all the base pointers, section pointers, sizes and map 7911 /// types for the extracted mappable expressions. Also, for each item that 7912 /// relates with a device pointer, a pair of the relevant declaration and 7913 /// index where it occurs is appended to the device pointers info array. 7914 void generateAllInfo(MapBaseValuesArrayTy &BasePointers, 7915 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 7916 MapFlagsArrayTy &Types) const { 7917 // We have to process the component lists that relate with the same 7918 // declaration in a single chunk so that we can generate the map flags 7919 // correctly. Therefore, we organize all lists in a map. 7920 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 7921 7922 // Helper function to fill the information map for the different supported 7923 // clauses. 7924 auto &&InfoGen = [&Info]( 7925 const ValueDecl *D, 7926 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 7927 OpenMPMapClauseKind MapType, 7928 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7929 bool ReturnDevicePointer, bool IsImplicit) { 7930 const ValueDecl *VD = 7931 D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr; 7932 Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer, 7933 IsImplicit); 7934 }; 7935 7936 assert(CurDir.is<const OMPExecutableDirective *>() && 7937 "Expect a executable directive"); 7938 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 7939 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) 7940 for (const auto L : C->component_lists()) { 7941 InfoGen(L.first, L.second, C->getMapType(), C->getMapTypeModifiers(), 7942 /*ReturnDevicePointer=*/false, C->isImplicit()); 7943 } 7944 for (const auto *C : CurExecDir->getClausesOfKind<OMPToClause>()) 7945 for (const auto L : C->component_lists()) { 7946 InfoGen(L.first, L.second, OMPC_MAP_to, llvm::None, 7947 /*ReturnDevicePointer=*/false, C->isImplicit()); 7948 } 7949 for (const auto *C : CurExecDir->getClausesOfKind<OMPFromClause>()) 7950 for (const auto L : C->component_lists()) { 7951 InfoGen(L.first, L.second, OMPC_MAP_from, llvm::None, 7952 /*ReturnDevicePointer=*/false, C->isImplicit()); 7953 } 7954 7955 // Look at the use_device_ptr clause information and mark the existing map 7956 // entries as such. If there is no map information for an entry in the 7957 // use_device_ptr list, we create one with map type 'alloc' and zero size 7958 // section. It is the user fault if that was not mapped before. If there is 7959 // no map information and the pointer is a struct member, then we defer the 7960 // emission of that entry until the whole struct has been processed. 7961 llvm::MapVector<const ValueDecl *, SmallVector<DeferredDevicePtrEntryTy, 4>> 7962 DeferredInfo; 7963 7964 for (const auto *C : 7965 CurExecDir->getClausesOfKind<OMPUseDevicePtrClause>()) { 7966 for (const auto L : C->component_lists()) { 7967 assert(!L.second.empty() && "Not expecting empty list of components!"); 7968 const ValueDecl *VD = L.second.back().getAssociatedDeclaration(); 7969 VD = cast<ValueDecl>(VD->getCanonicalDecl()); 7970 const Expr *IE = L.second.back().getAssociatedExpression(); 7971 // If the first component is a member expression, we have to look into 7972 // 'this', which maps to null in the map of map information. Otherwise 7973 // look directly for the information. 7974 auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD); 7975 7976 // We potentially have map information for this declaration already. 7977 // Look for the first set of components that refer to it. 7978 if (It != Info.end()) { 7979 auto CI = std::find_if( 7980 It->second.begin(), It->second.end(), [VD](const MapInfo &MI) { 7981 return MI.Components.back().getAssociatedDeclaration() == VD; 7982 }); 7983 // If we found a map entry, signal that the pointer has to be returned 7984 // and move on to the next declaration. 7985 if (CI != It->second.end()) { 7986 CI->ReturnDevicePointer = true; 7987 continue; 7988 } 7989 } 7990 7991 // We didn't find any match in our map information - generate a zero 7992 // size array section - if the pointer is a struct member we defer this 7993 // action until the whole struct has been processed. 7994 if (isa<MemberExpr>(IE)) { 7995 // Insert the pointer into Info to be processed by 7996 // generateInfoForComponentList. Because it is a member pointer 7997 // without a pointee, no entry will be generated for it, therefore 7998 // we need to generate one after the whole struct has been processed. 7999 // Nonetheless, generateInfoForComponentList must be called to take 8000 // the pointer into account for the calculation of the range of the 8001 // partial struct. 8002 InfoGen(nullptr, L.second, OMPC_MAP_unknown, llvm::None, 8003 /*ReturnDevicePointer=*/false, C->isImplicit()); 8004 DeferredInfo[nullptr].emplace_back(IE, VD); 8005 } else { 8006 llvm::Value *Ptr = 8007 CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc()); 8008 BasePointers.emplace_back(Ptr, VD); 8009 Pointers.push_back(Ptr); 8010 Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty)); 8011 Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_TARGET_PARAM); 8012 } 8013 } 8014 } 8015 8016 for (const auto &M : Info) { 8017 // We need to know when we generate information for the first component 8018 // associated with a capture, because the mapping flags depend on it. 8019 bool IsFirstComponentList = true; 8020 8021 // Temporary versions of arrays 8022 MapBaseValuesArrayTy CurBasePointers; 8023 MapValuesArrayTy CurPointers; 8024 MapValuesArrayTy CurSizes; 8025 MapFlagsArrayTy CurTypes; 8026 StructRangeInfoTy PartialStruct; 8027 8028 for (const MapInfo &L : M.second) { 8029 assert(!L.Components.empty() && 8030 "Not expecting declaration with no component lists."); 8031 8032 // Remember the current base pointer index. 8033 unsigned CurrentBasePointersIdx = CurBasePointers.size(); 8034 generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components, 8035 CurBasePointers, CurPointers, CurSizes, 8036 CurTypes, PartialStruct, 8037 IsFirstComponentList, L.IsImplicit); 8038 8039 // If this entry relates with a device pointer, set the relevant 8040 // declaration and add the 'return pointer' flag. 8041 if (L.ReturnDevicePointer) { 8042 assert(CurBasePointers.size() > CurrentBasePointersIdx && 8043 "Unexpected number of mapped base pointers."); 8044 8045 const ValueDecl *RelevantVD = 8046 L.Components.back().getAssociatedDeclaration(); 8047 assert(RelevantVD && 8048 "No relevant declaration related with device pointer??"); 8049 8050 CurBasePointers[CurrentBasePointersIdx].setDevicePtrDecl(RelevantVD); 8051 CurTypes[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM; 8052 } 8053 IsFirstComponentList = false; 8054 } 8055 8056 // Append any pending zero-length pointers which are struct members and 8057 // used with use_device_ptr. 8058 auto CI = DeferredInfo.find(M.first); 8059 if (CI != DeferredInfo.end()) { 8060 for (const DeferredDevicePtrEntryTy &L : CI->second) { 8061 llvm::Value *BasePtr = this->CGF.EmitLValue(L.IE).getPointer(CGF); 8062 llvm::Value *Ptr = this->CGF.EmitLoadOfScalar( 8063 this->CGF.EmitLValue(L.IE), L.IE->getExprLoc()); 8064 CurBasePointers.emplace_back(BasePtr, L.VD); 8065 CurPointers.push_back(Ptr); 8066 CurSizes.push_back(llvm::Constant::getNullValue(this->CGF.Int64Ty)); 8067 // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the placeholder 8068 // value MEMBER_OF=FFFF so that the entry is later updated with the 8069 // correct value of MEMBER_OF. 8070 CurTypes.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM | 8071 OMP_MAP_MEMBER_OF); 8072 } 8073 } 8074 8075 // If there is an entry in PartialStruct it means we have a struct with 8076 // individual members mapped. Emit an extra combined entry. 8077 if (PartialStruct.Base.isValid()) 8078 emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes, 8079 PartialStruct); 8080 8081 // We need to append the results of this capture to what we already have. 8082 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 8083 Pointers.append(CurPointers.begin(), CurPointers.end()); 8084 Sizes.append(CurSizes.begin(), CurSizes.end()); 8085 Types.append(CurTypes.begin(), CurTypes.end()); 8086 } 8087 } 8088 8089 /// Generate all the base pointers, section pointers, sizes and map types for 8090 /// the extracted map clauses of user-defined mapper. 8091 void generateAllInfoForMapper(MapBaseValuesArrayTy &BasePointers, 8092 MapValuesArrayTy &Pointers, 8093 MapValuesArrayTy &Sizes, 8094 MapFlagsArrayTy &Types) const { 8095 assert(CurDir.is<const OMPDeclareMapperDecl *>() && 8096 "Expect a declare mapper directive"); 8097 const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>(); 8098 // We have to process the component lists that relate with the same 8099 // declaration in a single chunk so that we can generate the map flags 8100 // correctly. Therefore, we organize all lists in a map. 8101 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 8102 8103 // Helper function to fill the information map for the different supported 8104 // clauses. 8105 auto &&InfoGen = [&Info]( 8106 const ValueDecl *D, 8107 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 8108 OpenMPMapClauseKind MapType, 8109 ArrayRef<OpenMPMapModifierKind> MapModifiers, 8110 bool ReturnDevicePointer, bool IsImplicit) { 8111 const ValueDecl *VD = 8112 D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr; 8113 Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer, 8114 IsImplicit); 8115 }; 8116 8117 for (const auto *C : CurMapperDir->clauselists()) { 8118 const auto *MC = cast<OMPMapClause>(C); 8119 for (const auto L : MC->component_lists()) { 8120 InfoGen(L.first, L.second, MC->getMapType(), MC->getMapTypeModifiers(), 8121 /*ReturnDevicePointer=*/false, MC->isImplicit()); 8122 } 8123 } 8124 8125 for (const auto &M : Info) { 8126 // We need to know when we generate information for the first component 8127 // associated with a capture, because the mapping flags depend on it. 8128 bool IsFirstComponentList = true; 8129 8130 // Temporary versions of arrays 8131 MapBaseValuesArrayTy CurBasePointers; 8132 MapValuesArrayTy CurPointers; 8133 MapValuesArrayTy CurSizes; 8134 MapFlagsArrayTy CurTypes; 8135 StructRangeInfoTy PartialStruct; 8136 8137 for (const MapInfo &L : M.second) { 8138 assert(!L.Components.empty() && 8139 "Not expecting declaration with no component lists."); 8140 generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components, 8141 CurBasePointers, CurPointers, CurSizes, 8142 CurTypes, PartialStruct, 8143 IsFirstComponentList, L.IsImplicit); 8144 IsFirstComponentList = false; 8145 } 8146 8147 // If there is an entry in PartialStruct it means we have a struct with 8148 // individual members mapped. Emit an extra combined entry. 8149 if (PartialStruct.Base.isValid()) 8150 emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes, 8151 PartialStruct); 8152 8153 // We need to append the results of this capture to what we already have. 8154 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 8155 Pointers.append(CurPointers.begin(), CurPointers.end()); 8156 Sizes.append(CurSizes.begin(), CurSizes.end()); 8157 Types.append(CurTypes.begin(), CurTypes.end()); 8158 } 8159 } 8160 8161 /// Emit capture info for lambdas for variables captured by reference. 8162 void generateInfoForLambdaCaptures( 8163 const ValueDecl *VD, llvm::Value *Arg, MapBaseValuesArrayTy &BasePointers, 8164 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 8165 MapFlagsArrayTy &Types, 8166 llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const { 8167 const auto *RD = VD->getType() 8168 .getCanonicalType() 8169 .getNonReferenceType() 8170 ->getAsCXXRecordDecl(); 8171 if (!RD || !RD->isLambda()) 8172 return; 8173 Address VDAddr = Address(Arg, CGF.getContext().getDeclAlign(VD)); 8174 LValue VDLVal = CGF.MakeAddrLValue( 8175 VDAddr, VD->getType().getCanonicalType().getNonReferenceType()); 8176 llvm::DenseMap<const VarDecl *, FieldDecl *> Captures; 8177 FieldDecl *ThisCapture = nullptr; 8178 RD->getCaptureFields(Captures, ThisCapture); 8179 if (ThisCapture) { 8180 LValue ThisLVal = 8181 CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture); 8182 LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture); 8183 LambdaPointers.try_emplace(ThisLVal.getPointer(CGF), 8184 VDLVal.getPointer(CGF)); 8185 BasePointers.push_back(ThisLVal.getPointer(CGF)); 8186 Pointers.push_back(ThisLValVal.getPointer(CGF)); 8187 Sizes.push_back( 8188 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy), 8189 CGF.Int64Ty, /*isSigned=*/true)); 8190 Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8191 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 8192 } 8193 for (const LambdaCapture &LC : RD->captures()) { 8194 if (!LC.capturesVariable()) 8195 continue; 8196 const VarDecl *VD = LC.getCapturedVar(); 8197 if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType()) 8198 continue; 8199 auto It = Captures.find(VD); 8200 assert(It != Captures.end() && "Found lambda capture without field."); 8201 LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second); 8202 if (LC.getCaptureKind() == LCK_ByRef) { 8203 LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second); 8204 LambdaPointers.try_emplace(VarLVal.getPointer(CGF), 8205 VDLVal.getPointer(CGF)); 8206 BasePointers.push_back(VarLVal.getPointer(CGF)); 8207 Pointers.push_back(VarLValVal.getPointer(CGF)); 8208 Sizes.push_back(CGF.Builder.CreateIntCast( 8209 CGF.getTypeSize( 8210 VD->getType().getCanonicalType().getNonReferenceType()), 8211 CGF.Int64Ty, /*isSigned=*/true)); 8212 } else { 8213 RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation()); 8214 LambdaPointers.try_emplace(VarLVal.getPointer(CGF), 8215 VDLVal.getPointer(CGF)); 8216 BasePointers.push_back(VarLVal.getPointer(CGF)); 8217 Pointers.push_back(VarRVal.getScalarVal()); 8218 Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0)); 8219 } 8220 Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8221 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 8222 } 8223 } 8224 8225 /// Set correct indices for lambdas captures. 8226 void adjustMemberOfForLambdaCaptures( 8227 const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers, 8228 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 8229 MapFlagsArrayTy &Types) const { 8230 for (unsigned I = 0, E = Types.size(); I < E; ++I) { 8231 // Set correct member_of idx for all implicit lambda captures. 8232 if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8233 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT)) 8234 continue; 8235 llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]); 8236 assert(BasePtr && "Unable to find base lambda address."); 8237 int TgtIdx = -1; 8238 for (unsigned J = I; J > 0; --J) { 8239 unsigned Idx = J - 1; 8240 if (Pointers[Idx] != BasePtr) 8241 continue; 8242 TgtIdx = Idx; 8243 break; 8244 } 8245 assert(TgtIdx != -1 && "Unable to find parent lambda."); 8246 // All other current entries will be MEMBER_OF the combined entry 8247 // (except for PTR_AND_OBJ entries which do not have a placeholder value 8248 // 0xFFFF in the MEMBER_OF field). 8249 OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx); 8250 setCorrectMemberOfFlag(Types[I], MemberOfFlag); 8251 } 8252 } 8253 8254 /// Generate the base pointers, section pointers, sizes and map types 8255 /// associated to a given capture. 8256 void generateInfoForCapture(const CapturedStmt::Capture *Cap, 8257 llvm::Value *Arg, 8258 MapBaseValuesArrayTy &BasePointers, 8259 MapValuesArrayTy &Pointers, 8260 MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types, 8261 StructRangeInfoTy &PartialStruct) const { 8262 assert(!Cap->capturesVariableArrayType() && 8263 "Not expecting to generate map info for a variable array type!"); 8264 8265 // We need to know when we generating information for the first component 8266 const ValueDecl *VD = Cap->capturesThis() 8267 ? nullptr 8268 : Cap->getCapturedVar()->getCanonicalDecl(); 8269 8270 // If this declaration appears in a is_device_ptr clause we just have to 8271 // pass the pointer by value. If it is a reference to a declaration, we just 8272 // pass its value. 8273 if (DevPointersMap.count(VD)) { 8274 BasePointers.emplace_back(Arg, VD); 8275 Pointers.push_back(Arg); 8276 Sizes.push_back( 8277 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy), 8278 CGF.Int64Ty, /*isSigned=*/true)); 8279 Types.push_back(OMP_MAP_LITERAL | OMP_MAP_TARGET_PARAM); 8280 return; 8281 } 8282 8283 using MapData = 8284 std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef, 8285 OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool>; 8286 SmallVector<MapData, 4> DeclComponentLists; 8287 assert(CurDir.is<const OMPExecutableDirective *>() && 8288 "Expect a executable directive"); 8289 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8290 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) { 8291 for (const auto L : C->decl_component_lists(VD)) { 8292 assert(L.first == VD && 8293 "We got information for the wrong declaration??"); 8294 assert(!L.second.empty() && 8295 "Not expecting declaration with no component lists."); 8296 DeclComponentLists.emplace_back(L.second, C->getMapType(), 8297 C->getMapTypeModifiers(), 8298 C->isImplicit()); 8299 } 8300 } 8301 8302 // Find overlapping elements (including the offset from the base element). 8303 llvm::SmallDenseMap< 8304 const MapData *, 8305 llvm::SmallVector< 8306 OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>, 8307 4> 8308 OverlappedData; 8309 size_t Count = 0; 8310 for (const MapData &L : DeclComponentLists) { 8311 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8312 OpenMPMapClauseKind MapType; 8313 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8314 bool IsImplicit; 8315 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8316 ++Count; 8317 for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) { 8318 OMPClauseMappableExprCommon::MappableExprComponentListRef Components1; 8319 std::tie(Components1, MapType, MapModifiers, IsImplicit) = L1; 8320 auto CI = Components.rbegin(); 8321 auto CE = Components.rend(); 8322 auto SI = Components1.rbegin(); 8323 auto SE = Components1.rend(); 8324 for (; CI != CE && SI != SE; ++CI, ++SI) { 8325 if (CI->getAssociatedExpression()->getStmtClass() != 8326 SI->getAssociatedExpression()->getStmtClass()) 8327 break; 8328 // Are we dealing with different variables/fields? 8329 if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration()) 8330 break; 8331 } 8332 // Found overlapping if, at least for one component, reached the head of 8333 // the components list. 8334 if (CI == CE || SI == SE) { 8335 assert((CI != CE || SI != SE) && 8336 "Unexpected full match of the mapping components."); 8337 const MapData &BaseData = CI == CE ? L : L1; 8338 OMPClauseMappableExprCommon::MappableExprComponentListRef SubData = 8339 SI == SE ? Components : Components1; 8340 auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData); 8341 OverlappedElements.getSecond().push_back(SubData); 8342 } 8343 } 8344 } 8345 // Sort the overlapped elements for each item. 8346 llvm::SmallVector<const FieldDecl *, 4> Layout; 8347 if (!OverlappedData.empty()) { 8348 if (const auto *CRD = 8349 VD->getType().getCanonicalType()->getAsCXXRecordDecl()) 8350 getPlainLayout(CRD, Layout, /*AsBase=*/false); 8351 else { 8352 const auto *RD = VD->getType().getCanonicalType()->getAsRecordDecl(); 8353 Layout.append(RD->field_begin(), RD->field_end()); 8354 } 8355 } 8356 for (auto &Pair : OverlappedData) { 8357 llvm::sort( 8358 Pair.getSecond(), 8359 [&Layout]( 8360 OMPClauseMappableExprCommon::MappableExprComponentListRef First, 8361 OMPClauseMappableExprCommon::MappableExprComponentListRef 8362 Second) { 8363 auto CI = First.rbegin(); 8364 auto CE = First.rend(); 8365 auto SI = Second.rbegin(); 8366 auto SE = Second.rend(); 8367 for (; CI != CE && SI != SE; ++CI, ++SI) { 8368 if (CI->getAssociatedExpression()->getStmtClass() != 8369 SI->getAssociatedExpression()->getStmtClass()) 8370 break; 8371 // Are we dealing with different variables/fields? 8372 if (CI->getAssociatedDeclaration() != 8373 SI->getAssociatedDeclaration()) 8374 break; 8375 } 8376 8377 // Lists contain the same elements. 8378 if (CI == CE && SI == SE) 8379 return false; 8380 8381 // List with less elements is less than list with more elements. 8382 if (CI == CE || SI == SE) 8383 return CI == CE; 8384 8385 const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration()); 8386 const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration()); 8387 if (FD1->getParent() == FD2->getParent()) 8388 return FD1->getFieldIndex() < FD2->getFieldIndex(); 8389 const auto It = 8390 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) { 8391 return FD == FD1 || FD == FD2; 8392 }); 8393 return *It == FD1; 8394 }); 8395 } 8396 8397 // Associated with a capture, because the mapping flags depend on it. 8398 // Go through all of the elements with the overlapped elements. 8399 for (const auto &Pair : OverlappedData) { 8400 const MapData &L = *Pair.getFirst(); 8401 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8402 OpenMPMapClauseKind MapType; 8403 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8404 bool IsImplicit; 8405 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8406 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 8407 OverlappedComponents = Pair.getSecond(); 8408 bool IsFirstComponentList = true; 8409 generateInfoForComponentList(MapType, MapModifiers, Components, 8410 BasePointers, Pointers, Sizes, Types, 8411 PartialStruct, IsFirstComponentList, 8412 IsImplicit, OverlappedComponents); 8413 } 8414 // Go through other elements without overlapped elements. 8415 bool IsFirstComponentList = OverlappedData.empty(); 8416 for (const MapData &L : DeclComponentLists) { 8417 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8418 OpenMPMapClauseKind MapType; 8419 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8420 bool IsImplicit; 8421 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8422 auto It = OverlappedData.find(&L); 8423 if (It == OverlappedData.end()) 8424 generateInfoForComponentList(MapType, MapModifiers, Components, 8425 BasePointers, Pointers, Sizes, Types, 8426 PartialStruct, IsFirstComponentList, 8427 IsImplicit); 8428 IsFirstComponentList = false; 8429 } 8430 } 8431 8432 /// Generate the base pointers, section pointers, sizes and map types 8433 /// associated with the declare target link variables. 8434 void generateInfoForDeclareTargetLink(MapBaseValuesArrayTy &BasePointers, 8435 MapValuesArrayTy &Pointers, 8436 MapValuesArrayTy &Sizes, 8437 MapFlagsArrayTy &Types) const { 8438 assert(CurDir.is<const OMPExecutableDirective *>() && 8439 "Expect a executable directive"); 8440 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8441 // Map other list items in the map clause which are not captured variables 8442 // but "declare target link" global variables. 8443 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) { 8444 for (const auto L : C->component_lists()) { 8445 if (!L.first) 8446 continue; 8447 const auto *VD = dyn_cast<VarDecl>(L.first); 8448 if (!VD) 8449 continue; 8450 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 8451 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 8452 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() || 8453 !Res || *Res != OMPDeclareTargetDeclAttr::MT_Link) 8454 continue; 8455 StructRangeInfoTy PartialStruct; 8456 generateInfoForComponentList( 8457 C->getMapType(), C->getMapTypeModifiers(), L.second, BasePointers, 8458 Pointers, Sizes, Types, PartialStruct, 8459 /*IsFirstComponentList=*/true, C->isImplicit()); 8460 assert(!PartialStruct.Base.isValid() && 8461 "No partial structs for declare target link expected."); 8462 } 8463 } 8464 } 8465 8466 /// Generate the default map information for a given capture \a CI, 8467 /// record field declaration \a RI and captured value \a CV. 8468 void generateDefaultMapInfo(const CapturedStmt::Capture &CI, 8469 const FieldDecl &RI, llvm::Value *CV, 8470 MapBaseValuesArrayTy &CurBasePointers, 8471 MapValuesArrayTy &CurPointers, 8472 MapValuesArrayTy &CurSizes, 8473 MapFlagsArrayTy &CurMapTypes) const { 8474 bool IsImplicit = true; 8475 // Do the default mapping. 8476 if (CI.capturesThis()) { 8477 CurBasePointers.push_back(CV); 8478 CurPointers.push_back(CV); 8479 const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr()); 8480 CurSizes.push_back( 8481 CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()), 8482 CGF.Int64Ty, /*isSigned=*/true)); 8483 // Default map type. 8484 CurMapTypes.push_back(OMP_MAP_TO | OMP_MAP_FROM); 8485 } else if (CI.capturesVariableByCopy()) { 8486 CurBasePointers.push_back(CV); 8487 CurPointers.push_back(CV); 8488 if (!RI.getType()->isAnyPointerType()) { 8489 // We have to signal to the runtime captures passed by value that are 8490 // not pointers. 8491 CurMapTypes.push_back(OMP_MAP_LITERAL); 8492 CurSizes.push_back(CGF.Builder.CreateIntCast( 8493 CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true)); 8494 } else { 8495 // Pointers are implicitly mapped with a zero size and no flags 8496 // (other than first map that is added for all implicit maps). 8497 CurMapTypes.push_back(OMP_MAP_NONE); 8498 CurSizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty)); 8499 } 8500 const VarDecl *VD = CI.getCapturedVar(); 8501 auto I = FirstPrivateDecls.find(VD); 8502 if (I != FirstPrivateDecls.end()) 8503 IsImplicit = I->getSecond(); 8504 } else { 8505 assert(CI.capturesVariable() && "Expected captured reference."); 8506 const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr()); 8507 QualType ElementType = PtrTy->getPointeeType(); 8508 CurSizes.push_back(CGF.Builder.CreateIntCast( 8509 CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true)); 8510 // The default map type for a scalar/complex type is 'to' because by 8511 // default the value doesn't have to be retrieved. For an aggregate 8512 // type, the default is 'tofrom'. 8513 CurMapTypes.push_back(getMapModifiersForPrivateClauses(CI)); 8514 const VarDecl *VD = CI.getCapturedVar(); 8515 auto I = FirstPrivateDecls.find(VD); 8516 if (I != FirstPrivateDecls.end() && 8517 VD->getType().isConstant(CGF.getContext())) { 8518 llvm::Constant *Addr = 8519 CGF.CGM.getOpenMPRuntime().registerTargetFirstprivateCopy(CGF, VD); 8520 // Copy the value of the original variable to the new global copy. 8521 CGF.Builder.CreateMemCpy( 8522 CGF.MakeNaturalAlignAddrLValue(Addr, ElementType).getAddress(CGF), 8523 Address(CV, CGF.getContext().getTypeAlignInChars(ElementType)), 8524 CurSizes.back(), /*IsVolatile=*/false); 8525 // Use new global variable as the base pointers. 8526 CurBasePointers.push_back(Addr); 8527 CurPointers.push_back(Addr); 8528 } else { 8529 CurBasePointers.push_back(CV); 8530 if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) { 8531 Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue( 8532 CV, ElementType, CGF.getContext().getDeclAlign(VD), 8533 AlignmentSource::Decl)); 8534 CurPointers.push_back(PtrAddr.getPointer()); 8535 } else { 8536 CurPointers.push_back(CV); 8537 } 8538 } 8539 if (I != FirstPrivateDecls.end()) 8540 IsImplicit = I->getSecond(); 8541 } 8542 // Every default map produces a single argument which is a target parameter. 8543 CurMapTypes.back() |= OMP_MAP_TARGET_PARAM; 8544 8545 // Add flag stating this is an implicit map. 8546 if (IsImplicit) 8547 CurMapTypes.back() |= OMP_MAP_IMPLICIT; 8548 } 8549 }; 8550 } // anonymous namespace 8551 8552 /// Emit the arrays used to pass the captures and map information to the 8553 /// offloading runtime library. If there is no map or capture information, 8554 /// return nullptr by reference. 8555 static void 8556 emitOffloadingArrays(CodeGenFunction &CGF, 8557 MappableExprsHandler::MapBaseValuesArrayTy &BasePointers, 8558 MappableExprsHandler::MapValuesArrayTy &Pointers, 8559 MappableExprsHandler::MapValuesArrayTy &Sizes, 8560 MappableExprsHandler::MapFlagsArrayTy &MapTypes, 8561 CGOpenMPRuntime::TargetDataInfo &Info) { 8562 CodeGenModule &CGM = CGF.CGM; 8563 ASTContext &Ctx = CGF.getContext(); 8564 8565 // Reset the array information. 8566 Info.clearArrayInfo(); 8567 Info.NumberOfPtrs = BasePointers.size(); 8568 8569 if (Info.NumberOfPtrs) { 8570 // Detect if we have any capture size requiring runtime evaluation of the 8571 // size so that a constant array could be eventually used. 8572 bool hasRuntimeEvaluationCaptureSize = false; 8573 for (llvm::Value *S : Sizes) 8574 if (!isa<llvm::Constant>(S)) { 8575 hasRuntimeEvaluationCaptureSize = true; 8576 break; 8577 } 8578 8579 llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true); 8580 QualType PointerArrayType = Ctx.getConstantArrayType( 8581 Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal, 8582 /*IndexTypeQuals=*/0); 8583 8584 Info.BasePointersArray = 8585 CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer(); 8586 Info.PointersArray = 8587 CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer(); 8588 8589 // If we don't have any VLA types or other types that require runtime 8590 // evaluation, we can use a constant array for the map sizes, otherwise we 8591 // need to fill up the arrays as we do for the pointers. 8592 QualType Int64Ty = 8593 Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 8594 if (hasRuntimeEvaluationCaptureSize) { 8595 QualType SizeArrayType = Ctx.getConstantArrayType( 8596 Int64Ty, PointerNumAP, nullptr, ArrayType::Normal, 8597 /*IndexTypeQuals=*/0); 8598 Info.SizesArray = 8599 CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer(); 8600 } else { 8601 // We expect all the sizes to be constant, so we collect them to create 8602 // a constant array. 8603 SmallVector<llvm::Constant *, 16> ConstSizes; 8604 for (llvm::Value *S : Sizes) 8605 ConstSizes.push_back(cast<llvm::Constant>(S)); 8606 8607 auto *SizesArrayInit = llvm::ConstantArray::get( 8608 llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes); 8609 std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"}); 8610 auto *SizesArrayGbl = new llvm::GlobalVariable( 8611 CGM.getModule(), SizesArrayInit->getType(), 8612 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 8613 SizesArrayInit, Name); 8614 SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 8615 Info.SizesArray = SizesArrayGbl; 8616 } 8617 8618 // The map types are always constant so we don't need to generate code to 8619 // fill arrays. Instead, we create an array constant. 8620 SmallVector<uint64_t, 4> Mapping(MapTypes.size(), 0); 8621 llvm::copy(MapTypes, Mapping.begin()); 8622 llvm::Constant *MapTypesArrayInit = 8623 llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping); 8624 std::string MaptypesName = 8625 CGM.getOpenMPRuntime().getName({"offload_maptypes"}); 8626 auto *MapTypesArrayGbl = new llvm::GlobalVariable( 8627 CGM.getModule(), MapTypesArrayInit->getType(), 8628 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 8629 MapTypesArrayInit, MaptypesName); 8630 MapTypesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 8631 Info.MapTypesArray = MapTypesArrayGbl; 8632 8633 for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) { 8634 llvm::Value *BPVal = *BasePointers[I]; 8635 llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32( 8636 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8637 Info.BasePointersArray, 0, I); 8638 BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8639 BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0)); 8640 Address BPAddr(BP, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 8641 CGF.Builder.CreateStore(BPVal, BPAddr); 8642 8643 if (Info.requiresDevicePointerInfo()) 8644 if (const ValueDecl *DevVD = BasePointers[I].getDevicePtrDecl()) 8645 Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr); 8646 8647 llvm::Value *PVal = Pointers[I]; 8648 llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32( 8649 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8650 Info.PointersArray, 0, I); 8651 P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8652 P, PVal->getType()->getPointerTo(/*AddrSpace=*/0)); 8653 Address PAddr(P, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 8654 CGF.Builder.CreateStore(PVal, PAddr); 8655 8656 if (hasRuntimeEvaluationCaptureSize) { 8657 llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32( 8658 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 8659 Info.SizesArray, 8660 /*Idx0=*/0, 8661 /*Idx1=*/I); 8662 Address SAddr(S, Ctx.getTypeAlignInChars(Int64Ty)); 8663 CGF.Builder.CreateStore( 8664 CGF.Builder.CreateIntCast(Sizes[I], CGM.Int64Ty, /*isSigned=*/true), 8665 SAddr); 8666 } 8667 } 8668 } 8669 } 8670 8671 /// Emit the arguments to be passed to the runtime library based on the 8672 /// arrays of pointers, sizes and map types. 8673 static void emitOffloadingArraysArgument( 8674 CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg, 8675 llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg, 8676 llvm::Value *&MapTypesArrayArg, CGOpenMPRuntime::TargetDataInfo &Info) { 8677 CodeGenModule &CGM = CGF.CGM; 8678 if (Info.NumberOfPtrs) { 8679 BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8680 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8681 Info.BasePointersArray, 8682 /*Idx0=*/0, /*Idx1=*/0); 8683 PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8684 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8685 Info.PointersArray, 8686 /*Idx0=*/0, 8687 /*Idx1=*/0); 8688 SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8689 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray, 8690 /*Idx0=*/0, /*Idx1=*/0); 8691 MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8692 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 8693 Info.MapTypesArray, 8694 /*Idx0=*/0, 8695 /*Idx1=*/0); 8696 } else { 8697 BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 8698 PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 8699 SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 8700 MapTypesArrayArg = 8701 llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 8702 } 8703 } 8704 8705 /// Check for inner distribute directive. 8706 static const OMPExecutableDirective * 8707 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) { 8708 const auto *CS = D.getInnermostCapturedStmt(); 8709 const auto *Body = 8710 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 8711 const Stmt *ChildStmt = 8712 CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 8713 8714 if (const auto *NestedDir = 8715 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 8716 OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind(); 8717 switch (D.getDirectiveKind()) { 8718 case OMPD_target: 8719 if (isOpenMPDistributeDirective(DKind)) 8720 return NestedDir; 8721 if (DKind == OMPD_teams) { 8722 Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers( 8723 /*IgnoreCaptured=*/true); 8724 if (!Body) 8725 return nullptr; 8726 ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 8727 if (const auto *NND = 8728 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 8729 DKind = NND->getDirectiveKind(); 8730 if (isOpenMPDistributeDirective(DKind)) 8731 return NND; 8732 } 8733 } 8734 return nullptr; 8735 case OMPD_target_teams: 8736 if (isOpenMPDistributeDirective(DKind)) 8737 return NestedDir; 8738 return nullptr; 8739 case OMPD_target_parallel: 8740 case OMPD_target_simd: 8741 case OMPD_target_parallel_for: 8742 case OMPD_target_parallel_for_simd: 8743 return nullptr; 8744 case OMPD_target_teams_distribute: 8745 case OMPD_target_teams_distribute_simd: 8746 case OMPD_target_teams_distribute_parallel_for: 8747 case OMPD_target_teams_distribute_parallel_for_simd: 8748 case OMPD_parallel: 8749 case OMPD_for: 8750 case OMPD_parallel_for: 8751 case OMPD_parallel_master: 8752 case OMPD_parallel_sections: 8753 case OMPD_for_simd: 8754 case OMPD_parallel_for_simd: 8755 case OMPD_cancel: 8756 case OMPD_cancellation_point: 8757 case OMPD_ordered: 8758 case OMPD_threadprivate: 8759 case OMPD_allocate: 8760 case OMPD_task: 8761 case OMPD_simd: 8762 case OMPD_sections: 8763 case OMPD_section: 8764 case OMPD_single: 8765 case OMPD_master: 8766 case OMPD_critical: 8767 case OMPD_taskyield: 8768 case OMPD_barrier: 8769 case OMPD_taskwait: 8770 case OMPD_taskgroup: 8771 case OMPD_atomic: 8772 case OMPD_flush: 8773 case OMPD_teams: 8774 case OMPD_target_data: 8775 case OMPD_target_exit_data: 8776 case OMPD_target_enter_data: 8777 case OMPD_distribute: 8778 case OMPD_distribute_simd: 8779 case OMPD_distribute_parallel_for: 8780 case OMPD_distribute_parallel_for_simd: 8781 case OMPD_teams_distribute: 8782 case OMPD_teams_distribute_simd: 8783 case OMPD_teams_distribute_parallel_for: 8784 case OMPD_teams_distribute_parallel_for_simd: 8785 case OMPD_target_update: 8786 case OMPD_declare_simd: 8787 case OMPD_declare_variant: 8788 case OMPD_declare_target: 8789 case OMPD_end_declare_target: 8790 case OMPD_declare_reduction: 8791 case OMPD_declare_mapper: 8792 case OMPD_taskloop: 8793 case OMPD_taskloop_simd: 8794 case OMPD_master_taskloop: 8795 case OMPD_master_taskloop_simd: 8796 case OMPD_parallel_master_taskloop: 8797 case OMPD_parallel_master_taskloop_simd: 8798 case OMPD_requires: 8799 case OMPD_unknown: 8800 llvm_unreachable("Unexpected directive."); 8801 } 8802 } 8803 8804 return nullptr; 8805 } 8806 8807 /// Emit the user-defined mapper function. The code generation follows the 8808 /// pattern in the example below. 8809 /// \code 8810 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle, 8811 /// void *base, void *begin, 8812 /// int64_t size, int64_t type) { 8813 /// // Allocate space for an array section first. 8814 /// if (size > 1 && !maptype.IsDelete) 8815 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 8816 /// size*sizeof(Ty), clearToFrom(type)); 8817 /// // Map members. 8818 /// for (unsigned i = 0; i < size; i++) { 8819 /// // For each component specified by this mapper: 8820 /// for (auto c : all_components) { 8821 /// if (c.hasMapper()) 8822 /// (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size, 8823 /// c.arg_type); 8824 /// else 8825 /// __tgt_push_mapper_component(rt_mapper_handle, c.arg_base, 8826 /// c.arg_begin, c.arg_size, c.arg_type); 8827 /// } 8828 /// } 8829 /// // Delete the array section. 8830 /// if (size > 1 && maptype.IsDelete) 8831 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 8832 /// size*sizeof(Ty), clearToFrom(type)); 8833 /// } 8834 /// \endcode 8835 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D, 8836 CodeGenFunction *CGF) { 8837 if (UDMMap.count(D) > 0) 8838 return; 8839 ASTContext &C = CGM.getContext(); 8840 QualType Ty = D->getType(); 8841 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 8842 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 8843 auto *MapperVarDecl = 8844 cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl()); 8845 SourceLocation Loc = D->getLocation(); 8846 CharUnits ElementSize = C.getTypeSizeInChars(Ty); 8847 8848 // Prepare mapper function arguments and attributes. 8849 ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 8850 C.VoidPtrTy, ImplicitParamDecl::Other); 8851 ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 8852 ImplicitParamDecl::Other); 8853 ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 8854 C.VoidPtrTy, ImplicitParamDecl::Other); 8855 ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 8856 ImplicitParamDecl::Other); 8857 ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 8858 ImplicitParamDecl::Other); 8859 FunctionArgList Args; 8860 Args.push_back(&HandleArg); 8861 Args.push_back(&BaseArg); 8862 Args.push_back(&BeginArg); 8863 Args.push_back(&SizeArg); 8864 Args.push_back(&TypeArg); 8865 const CGFunctionInfo &FnInfo = 8866 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 8867 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 8868 SmallString<64> TyStr; 8869 llvm::raw_svector_ostream Out(TyStr); 8870 CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out); 8871 std::string Name = getName({"omp_mapper", TyStr, D->getName()}); 8872 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 8873 Name, &CGM.getModule()); 8874 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 8875 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 8876 // Start the mapper function code generation. 8877 CodeGenFunction MapperCGF(CGM); 8878 MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 8879 // Compute the starting and end addreses of array elements. 8880 llvm::Value *Size = MapperCGF.EmitLoadOfScalar( 8881 MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false, 8882 C.getPointerType(Int64Ty), Loc); 8883 llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast( 8884 MapperCGF.GetAddrOfLocalVar(&BeginArg).getPointer(), 8885 CGM.getTypes().ConvertTypeForMem(C.getPointerType(PtrTy))); 8886 llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(PtrBegin, Size); 8887 llvm::Value *MapType = MapperCGF.EmitLoadOfScalar( 8888 MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false, 8889 C.getPointerType(Int64Ty), Loc); 8890 // Prepare common arguments for array initiation and deletion. 8891 llvm::Value *Handle = MapperCGF.EmitLoadOfScalar( 8892 MapperCGF.GetAddrOfLocalVar(&HandleArg), 8893 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 8894 llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar( 8895 MapperCGF.GetAddrOfLocalVar(&BaseArg), 8896 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 8897 llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar( 8898 MapperCGF.GetAddrOfLocalVar(&BeginArg), 8899 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 8900 8901 // Emit array initiation if this is an array section and \p MapType indicates 8902 // that memory allocation is required. 8903 llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head"); 8904 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 8905 ElementSize, HeadBB, /*IsInit=*/true); 8906 8907 // Emit a for loop to iterate through SizeArg of elements and map all of them. 8908 8909 // Emit the loop header block. 8910 MapperCGF.EmitBlock(HeadBB); 8911 llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body"); 8912 llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done"); 8913 // Evaluate whether the initial condition is satisfied. 8914 llvm::Value *IsEmpty = 8915 MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty"); 8916 MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 8917 llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock(); 8918 8919 // Emit the loop body block. 8920 MapperCGF.EmitBlock(BodyBB); 8921 llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI( 8922 PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent"); 8923 PtrPHI->addIncoming(PtrBegin, EntryBB); 8924 Address PtrCurrent = 8925 Address(PtrPHI, MapperCGF.GetAddrOfLocalVar(&BeginArg) 8926 .getAlignment() 8927 .alignmentOfArrayElement(ElementSize)); 8928 // Privatize the declared variable of mapper to be the current array element. 8929 CodeGenFunction::OMPPrivateScope Scope(MapperCGF); 8930 Scope.addPrivate(MapperVarDecl, [&MapperCGF, PtrCurrent, PtrTy]() { 8931 return MapperCGF 8932 .EmitLoadOfPointerLValue(PtrCurrent, PtrTy->castAs<PointerType>()) 8933 .getAddress(MapperCGF); 8934 }); 8935 (void)Scope.Privatize(); 8936 8937 // Get map clause information. Fill up the arrays with all mapped variables. 8938 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 8939 MappableExprsHandler::MapValuesArrayTy Pointers; 8940 MappableExprsHandler::MapValuesArrayTy Sizes; 8941 MappableExprsHandler::MapFlagsArrayTy MapTypes; 8942 MappableExprsHandler MEHandler(*D, MapperCGF); 8943 MEHandler.generateAllInfoForMapper(BasePointers, Pointers, Sizes, MapTypes); 8944 8945 // Call the runtime API __tgt_mapper_num_components to get the number of 8946 // pre-existing components. 8947 llvm::Value *OffloadingArgs[] = {Handle}; 8948 llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall( 8949 createRuntimeFunction(OMPRTL__tgt_mapper_num_components), OffloadingArgs); 8950 llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl( 8951 PreviousSize, 8952 MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset())); 8953 8954 // Fill up the runtime mapper handle for all components. 8955 for (unsigned I = 0; I < BasePointers.size(); ++I) { 8956 llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast( 8957 *BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 8958 llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast( 8959 Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 8960 llvm::Value *CurSizeArg = Sizes[I]; 8961 8962 // Extract the MEMBER_OF field from the map type. 8963 llvm::BasicBlock *MemberBB = MapperCGF.createBasicBlock("omp.member"); 8964 MapperCGF.EmitBlock(MemberBB); 8965 llvm::Value *OriMapType = MapperCGF.Builder.getInt64(MapTypes[I]); 8966 llvm::Value *Member = MapperCGF.Builder.CreateAnd( 8967 OriMapType, 8968 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_MEMBER_OF)); 8969 llvm::BasicBlock *MemberCombineBB = 8970 MapperCGF.createBasicBlock("omp.member.combine"); 8971 llvm::BasicBlock *TypeBB = MapperCGF.createBasicBlock("omp.type"); 8972 llvm::Value *IsMember = MapperCGF.Builder.CreateIsNull(Member); 8973 MapperCGF.Builder.CreateCondBr(IsMember, TypeBB, MemberCombineBB); 8974 // Add the number of pre-existing components to the MEMBER_OF field if it 8975 // is valid. 8976 MapperCGF.EmitBlock(MemberCombineBB); 8977 llvm::Value *CombinedMember = 8978 MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize); 8979 // Do nothing if it is not a member of previous components. 8980 MapperCGF.EmitBlock(TypeBB); 8981 llvm::PHINode *MemberMapType = 8982 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.membermaptype"); 8983 MemberMapType->addIncoming(OriMapType, MemberBB); 8984 MemberMapType->addIncoming(CombinedMember, MemberCombineBB); 8985 8986 // Combine the map type inherited from user-defined mapper with that 8987 // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM 8988 // bits of the \a MapType, which is the input argument of the mapper 8989 // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM 8990 // bits of MemberMapType. 8991 // [OpenMP 5.0], 1.2.6. map-type decay. 8992 // | alloc | to | from | tofrom | release | delete 8993 // ---------------------------------------------------------- 8994 // alloc | alloc | alloc | alloc | alloc | release | delete 8995 // to | alloc | to | alloc | to | release | delete 8996 // from | alloc | alloc | from | from | release | delete 8997 // tofrom | alloc | to | from | tofrom | release | delete 8998 llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd( 8999 MapType, 9000 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO | 9001 MappableExprsHandler::OMP_MAP_FROM)); 9002 llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc"); 9003 llvm::BasicBlock *AllocElseBB = 9004 MapperCGF.createBasicBlock("omp.type.alloc.else"); 9005 llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to"); 9006 llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else"); 9007 llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from"); 9008 llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end"); 9009 llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom); 9010 MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB); 9011 // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM. 9012 MapperCGF.EmitBlock(AllocBB); 9013 llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd( 9014 MemberMapType, 9015 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 9016 MappableExprsHandler::OMP_MAP_FROM))); 9017 MapperCGF.Builder.CreateBr(EndBB); 9018 MapperCGF.EmitBlock(AllocElseBB); 9019 llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ( 9020 LeftToFrom, 9021 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO)); 9022 MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB); 9023 // In case of to, clear OMP_MAP_FROM. 9024 MapperCGF.EmitBlock(ToBB); 9025 llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd( 9026 MemberMapType, 9027 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM)); 9028 MapperCGF.Builder.CreateBr(EndBB); 9029 MapperCGF.EmitBlock(ToElseBB); 9030 llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ( 9031 LeftToFrom, 9032 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM)); 9033 MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB); 9034 // In case of from, clear OMP_MAP_TO. 9035 MapperCGF.EmitBlock(FromBB); 9036 llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd( 9037 MemberMapType, 9038 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO)); 9039 // In case of tofrom, do nothing. 9040 MapperCGF.EmitBlock(EndBB); 9041 llvm::PHINode *CurMapType = 9042 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype"); 9043 CurMapType->addIncoming(AllocMapType, AllocBB); 9044 CurMapType->addIncoming(ToMapType, ToBB); 9045 CurMapType->addIncoming(FromMapType, FromBB); 9046 CurMapType->addIncoming(MemberMapType, ToElseBB); 9047 9048 // TODO: call the corresponding mapper function if a user-defined mapper is 9049 // associated with this map clause. 9050 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 9051 // data structure. 9052 llvm::Value *OffloadingArgs[] = {Handle, CurBaseArg, CurBeginArg, 9053 CurSizeArg, CurMapType}; 9054 MapperCGF.EmitRuntimeCall( 9055 createRuntimeFunction(OMPRTL__tgt_push_mapper_component), 9056 OffloadingArgs); 9057 } 9058 9059 // Update the pointer to point to the next element that needs to be mapped, 9060 // and check whether we have mapped all elements. 9061 llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32( 9062 PtrPHI, /*Idx0=*/1, "omp.arraymap.next"); 9063 PtrPHI->addIncoming(PtrNext, BodyBB); 9064 llvm::Value *IsDone = 9065 MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone"); 9066 llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit"); 9067 MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB); 9068 9069 MapperCGF.EmitBlock(ExitBB); 9070 // Emit array deletion if this is an array section and \p MapType indicates 9071 // that deletion is required. 9072 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 9073 ElementSize, DoneBB, /*IsInit=*/false); 9074 9075 // Emit the function exit block. 9076 MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true); 9077 MapperCGF.FinishFunction(); 9078 UDMMap.try_emplace(D, Fn); 9079 if (CGF) { 9080 auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn); 9081 Decls.second.push_back(D); 9082 } 9083 } 9084 9085 /// Emit the array initialization or deletion portion for user-defined mapper 9086 /// code generation. First, it evaluates whether an array section is mapped and 9087 /// whether the \a MapType instructs to delete this section. If \a IsInit is 9088 /// true, and \a MapType indicates to not delete this array, array 9089 /// initialization code is generated. If \a IsInit is false, and \a MapType 9090 /// indicates to not this array, array deletion code is generated. 9091 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel( 9092 CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base, 9093 llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType, 9094 CharUnits ElementSize, llvm::BasicBlock *ExitBB, bool IsInit) { 9095 StringRef Prefix = IsInit ? ".init" : ".del"; 9096 9097 // Evaluate if this is an array section. 9098 llvm::BasicBlock *IsDeleteBB = 9099 MapperCGF.createBasicBlock(getName({"omp.array", Prefix, ".evaldelete"})); 9100 llvm::BasicBlock *BodyBB = 9101 MapperCGF.createBasicBlock(getName({"omp.array", Prefix})); 9102 llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGE( 9103 Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray"); 9104 MapperCGF.Builder.CreateCondBr(IsArray, IsDeleteBB, ExitBB); 9105 9106 // Evaluate if we are going to delete this section. 9107 MapperCGF.EmitBlock(IsDeleteBB); 9108 llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd( 9109 MapType, 9110 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE)); 9111 llvm::Value *DeleteCond; 9112 if (IsInit) { 9113 DeleteCond = MapperCGF.Builder.CreateIsNull( 9114 DeleteBit, getName({"omp.array", Prefix, ".delete"})); 9115 } else { 9116 DeleteCond = MapperCGF.Builder.CreateIsNotNull( 9117 DeleteBit, getName({"omp.array", Prefix, ".delete"})); 9118 } 9119 MapperCGF.Builder.CreateCondBr(DeleteCond, BodyBB, ExitBB); 9120 9121 MapperCGF.EmitBlock(BodyBB); 9122 // Get the array size by multiplying element size and element number (i.e., \p 9123 // Size). 9124 llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul( 9125 Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity())); 9126 // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves 9127 // memory allocation/deletion purpose only. 9128 llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd( 9129 MapType, 9130 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 9131 MappableExprsHandler::OMP_MAP_FROM))); 9132 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 9133 // data structure. 9134 llvm::Value *OffloadingArgs[] = {Handle, Base, Begin, ArraySize, MapTypeArg}; 9135 MapperCGF.EmitRuntimeCall( 9136 createRuntimeFunction(OMPRTL__tgt_push_mapper_component), OffloadingArgs); 9137 } 9138 9139 void CGOpenMPRuntime::emitTargetNumIterationsCall( 9140 CodeGenFunction &CGF, const OMPExecutableDirective &D, 9141 llvm::Value *DeviceID, 9142 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 9143 const OMPLoopDirective &D)> 9144 SizeEmitter) { 9145 OpenMPDirectiveKind Kind = D.getDirectiveKind(); 9146 const OMPExecutableDirective *TD = &D; 9147 // Get nested teams distribute kind directive, if any. 9148 if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind)) 9149 TD = getNestedDistributeDirective(CGM.getContext(), D); 9150 if (!TD) 9151 return; 9152 const auto *LD = cast<OMPLoopDirective>(TD); 9153 auto &&CodeGen = [LD, DeviceID, SizeEmitter, this](CodeGenFunction &CGF, 9154 PrePostActionTy &) { 9155 if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) { 9156 llvm::Value *Args[] = {DeviceID, NumIterations}; 9157 CGF.EmitRuntimeCall( 9158 createRuntimeFunction(OMPRTL__kmpc_push_target_tripcount), Args); 9159 } 9160 }; 9161 emitInlinedDirective(CGF, OMPD_unknown, CodeGen); 9162 } 9163 9164 void CGOpenMPRuntime::emitTargetCall( 9165 CodeGenFunction &CGF, const OMPExecutableDirective &D, 9166 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 9167 const Expr *Device, 9168 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 9169 const OMPLoopDirective &D)> 9170 SizeEmitter) { 9171 if (!CGF.HaveInsertPoint()) 9172 return; 9173 9174 assert(OutlinedFn && "Invalid outlined function!"); 9175 9176 const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>(); 9177 llvm::SmallVector<llvm::Value *, 16> CapturedVars; 9178 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 9179 auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF, 9180 PrePostActionTy &) { 9181 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9182 }; 9183 emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen); 9184 9185 CodeGenFunction::OMPTargetDataInfo InputInfo; 9186 llvm::Value *MapTypesArray = nullptr; 9187 // Fill up the pointer arrays and transfer execution to the device. 9188 auto &&ThenGen = [this, Device, OutlinedFn, OutlinedFnID, &D, &InputInfo, 9189 &MapTypesArray, &CS, RequiresOuterTask, &CapturedVars, 9190 SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) { 9191 // On top of the arrays that were filled up, the target offloading call 9192 // takes as arguments the device id as well as the host pointer. The host 9193 // pointer is used by the runtime library to identify the current target 9194 // region, so it only has to be unique and not necessarily point to 9195 // anything. It could be the pointer to the outlined function that 9196 // implements the target region, but we aren't using that so that the 9197 // compiler doesn't need to keep that, and could therefore inline the host 9198 // function if proven worthwhile during optimization. 9199 9200 // From this point on, we need to have an ID of the target region defined. 9201 assert(OutlinedFnID && "Invalid outlined function ID!"); 9202 9203 // Emit device ID if any. 9204 llvm::Value *DeviceID; 9205 if (Device) { 9206 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 9207 CGF.Int64Ty, /*isSigned=*/true); 9208 } else { 9209 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 9210 } 9211 9212 // Emit the number of elements in the offloading arrays. 9213 llvm::Value *PointerNum = 9214 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 9215 9216 // Return value of the runtime offloading call. 9217 llvm::Value *Return; 9218 9219 llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D); 9220 llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D); 9221 9222 // Emit tripcount for the target loop-based directive. 9223 emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter); 9224 9225 bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 9226 // The target region is an outlined function launched by the runtime 9227 // via calls __tgt_target() or __tgt_target_teams(). 9228 // 9229 // __tgt_target() launches a target region with one team and one thread, 9230 // executing a serial region. This master thread may in turn launch 9231 // more threads within its team upon encountering a parallel region, 9232 // however, no additional teams can be launched on the device. 9233 // 9234 // __tgt_target_teams() launches a target region with one or more teams, 9235 // each with one or more threads. This call is required for target 9236 // constructs such as: 9237 // 'target teams' 9238 // 'target' / 'teams' 9239 // 'target teams distribute parallel for' 9240 // 'target parallel' 9241 // and so on. 9242 // 9243 // Note that on the host and CPU targets, the runtime implementation of 9244 // these calls simply call the outlined function without forking threads. 9245 // The outlined functions themselves have runtime calls to 9246 // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by 9247 // the compiler in emitTeamsCall() and emitParallelCall(). 9248 // 9249 // In contrast, on the NVPTX target, the implementation of 9250 // __tgt_target_teams() launches a GPU kernel with the requested number 9251 // of teams and threads so no additional calls to the runtime are required. 9252 if (NumTeams) { 9253 // If we have NumTeams defined this means that we have an enclosed teams 9254 // region. Therefore we also expect to have NumThreads defined. These two 9255 // values should be defined in the presence of a teams directive, 9256 // regardless of having any clauses associated. If the user is using teams 9257 // but no clauses, these two values will be the default that should be 9258 // passed to the runtime library - a 32-bit integer with the value zero. 9259 assert(NumThreads && "Thread limit expression should be available along " 9260 "with number of teams."); 9261 llvm::Value *OffloadingArgs[] = {DeviceID, 9262 OutlinedFnID, 9263 PointerNum, 9264 InputInfo.BasePointersArray.getPointer(), 9265 InputInfo.PointersArray.getPointer(), 9266 InputInfo.SizesArray.getPointer(), 9267 MapTypesArray, 9268 NumTeams, 9269 NumThreads}; 9270 Return = CGF.EmitRuntimeCall( 9271 createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_teams_nowait 9272 : OMPRTL__tgt_target_teams), 9273 OffloadingArgs); 9274 } else { 9275 llvm::Value *OffloadingArgs[] = {DeviceID, 9276 OutlinedFnID, 9277 PointerNum, 9278 InputInfo.BasePointersArray.getPointer(), 9279 InputInfo.PointersArray.getPointer(), 9280 InputInfo.SizesArray.getPointer(), 9281 MapTypesArray}; 9282 Return = CGF.EmitRuntimeCall( 9283 createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_nowait 9284 : OMPRTL__tgt_target), 9285 OffloadingArgs); 9286 } 9287 9288 // Check the error code and execute the host version if required. 9289 llvm::BasicBlock *OffloadFailedBlock = 9290 CGF.createBasicBlock("omp_offload.failed"); 9291 llvm::BasicBlock *OffloadContBlock = 9292 CGF.createBasicBlock("omp_offload.cont"); 9293 llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return); 9294 CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock); 9295 9296 CGF.EmitBlock(OffloadFailedBlock); 9297 if (RequiresOuterTask) { 9298 CapturedVars.clear(); 9299 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9300 } 9301 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9302 CGF.EmitBranch(OffloadContBlock); 9303 9304 CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true); 9305 }; 9306 9307 // Notify that the host version must be executed. 9308 auto &&ElseGen = [this, &D, OutlinedFn, &CS, &CapturedVars, 9309 RequiresOuterTask](CodeGenFunction &CGF, 9310 PrePostActionTy &) { 9311 if (RequiresOuterTask) { 9312 CapturedVars.clear(); 9313 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9314 } 9315 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9316 }; 9317 9318 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray, 9319 &CapturedVars, RequiresOuterTask, 9320 &CS](CodeGenFunction &CGF, PrePostActionTy &) { 9321 // Fill up the arrays with all the captured variables. 9322 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 9323 MappableExprsHandler::MapValuesArrayTy Pointers; 9324 MappableExprsHandler::MapValuesArrayTy Sizes; 9325 MappableExprsHandler::MapFlagsArrayTy MapTypes; 9326 9327 // Get mappable expression information. 9328 MappableExprsHandler MEHandler(D, CGF); 9329 llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers; 9330 9331 auto RI = CS.getCapturedRecordDecl()->field_begin(); 9332 auto CV = CapturedVars.begin(); 9333 for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(), 9334 CE = CS.capture_end(); 9335 CI != CE; ++CI, ++RI, ++CV) { 9336 MappableExprsHandler::MapBaseValuesArrayTy CurBasePointers; 9337 MappableExprsHandler::MapValuesArrayTy CurPointers; 9338 MappableExprsHandler::MapValuesArrayTy CurSizes; 9339 MappableExprsHandler::MapFlagsArrayTy CurMapTypes; 9340 MappableExprsHandler::StructRangeInfoTy PartialStruct; 9341 9342 // VLA sizes are passed to the outlined region by copy and do not have map 9343 // information associated. 9344 if (CI->capturesVariableArrayType()) { 9345 CurBasePointers.push_back(*CV); 9346 CurPointers.push_back(*CV); 9347 CurSizes.push_back(CGF.Builder.CreateIntCast( 9348 CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true)); 9349 // Copy to the device as an argument. No need to retrieve it. 9350 CurMapTypes.push_back(MappableExprsHandler::OMP_MAP_LITERAL | 9351 MappableExprsHandler::OMP_MAP_TARGET_PARAM | 9352 MappableExprsHandler::OMP_MAP_IMPLICIT); 9353 } else { 9354 // If we have any information in the map clause, we use it, otherwise we 9355 // just do a default mapping. 9356 MEHandler.generateInfoForCapture(CI, *CV, CurBasePointers, CurPointers, 9357 CurSizes, CurMapTypes, PartialStruct); 9358 if (CurBasePointers.empty()) 9359 MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurBasePointers, 9360 CurPointers, CurSizes, CurMapTypes); 9361 // Generate correct mapping for variables captured by reference in 9362 // lambdas. 9363 if (CI->capturesVariable()) 9364 MEHandler.generateInfoForLambdaCaptures( 9365 CI->getCapturedVar(), *CV, CurBasePointers, CurPointers, CurSizes, 9366 CurMapTypes, LambdaPointers); 9367 } 9368 // We expect to have at least an element of information for this capture. 9369 assert(!CurBasePointers.empty() && 9370 "Non-existing map pointer for capture!"); 9371 assert(CurBasePointers.size() == CurPointers.size() && 9372 CurBasePointers.size() == CurSizes.size() && 9373 CurBasePointers.size() == CurMapTypes.size() && 9374 "Inconsistent map information sizes!"); 9375 9376 // If there is an entry in PartialStruct it means we have a struct with 9377 // individual members mapped. Emit an extra combined entry. 9378 if (PartialStruct.Base.isValid()) 9379 MEHandler.emitCombinedEntry(BasePointers, Pointers, Sizes, MapTypes, 9380 CurMapTypes, PartialStruct); 9381 9382 // We need to append the results of this capture to what we already have. 9383 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 9384 Pointers.append(CurPointers.begin(), CurPointers.end()); 9385 Sizes.append(CurSizes.begin(), CurSizes.end()); 9386 MapTypes.append(CurMapTypes.begin(), CurMapTypes.end()); 9387 } 9388 // Adjust MEMBER_OF flags for the lambdas captures. 9389 MEHandler.adjustMemberOfForLambdaCaptures(LambdaPointers, BasePointers, 9390 Pointers, MapTypes); 9391 // Map other list items in the map clause which are not captured variables 9392 // but "declare target link" global variables. 9393 MEHandler.generateInfoForDeclareTargetLink(BasePointers, Pointers, Sizes, 9394 MapTypes); 9395 9396 TargetDataInfo Info; 9397 // Fill up the arrays and create the arguments. 9398 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 9399 emitOffloadingArraysArgument(CGF, Info.BasePointersArray, 9400 Info.PointersArray, Info.SizesArray, 9401 Info.MapTypesArray, Info); 9402 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 9403 InputInfo.BasePointersArray = 9404 Address(Info.BasePointersArray, CGM.getPointerAlign()); 9405 InputInfo.PointersArray = 9406 Address(Info.PointersArray, CGM.getPointerAlign()); 9407 InputInfo.SizesArray = Address(Info.SizesArray, CGM.getPointerAlign()); 9408 MapTypesArray = Info.MapTypesArray; 9409 if (RequiresOuterTask) 9410 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 9411 else 9412 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 9413 }; 9414 9415 auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask]( 9416 CodeGenFunction &CGF, PrePostActionTy &) { 9417 if (RequiresOuterTask) { 9418 CodeGenFunction::OMPTargetDataInfo InputInfo; 9419 CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo); 9420 } else { 9421 emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen); 9422 } 9423 }; 9424 9425 // If we have a target function ID it means that we need to support 9426 // offloading, otherwise, just execute on the host. We need to execute on host 9427 // regardless of the conditional in the if clause if, e.g., the user do not 9428 // specify target triples. 9429 if (OutlinedFnID) { 9430 if (IfCond) { 9431 emitIfClause(CGF, IfCond, TargetThenGen, TargetElseGen); 9432 } else { 9433 RegionCodeGenTy ThenRCG(TargetThenGen); 9434 ThenRCG(CGF); 9435 } 9436 } else { 9437 RegionCodeGenTy ElseRCG(TargetElseGen); 9438 ElseRCG(CGF); 9439 } 9440 } 9441 9442 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S, 9443 StringRef ParentName) { 9444 if (!S) 9445 return; 9446 9447 // Codegen OMP target directives that offload compute to the device. 9448 bool RequiresDeviceCodegen = 9449 isa<OMPExecutableDirective>(S) && 9450 isOpenMPTargetExecutionDirective( 9451 cast<OMPExecutableDirective>(S)->getDirectiveKind()); 9452 9453 if (RequiresDeviceCodegen) { 9454 const auto &E = *cast<OMPExecutableDirective>(S); 9455 unsigned DeviceID; 9456 unsigned FileID; 9457 unsigned Line; 9458 getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID, 9459 FileID, Line); 9460 9461 // Is this a target region that should not be emitted as an entry point? If 9462 // so just signal we are done with this target region. 9463 if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID, 9464 ParentName, Line)) 9465 return; 9466 9467 switch (E.getDirectiveKind()) { 9468 case OMPD_target: 9469 CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName, 9470 cast<OMPTargetDirective>(E)); 9471 break; 9472 case OMPD_target_parallel: 9473 CodeGenFunction::EmitOMPTargetParallelDeviceFunction( 9474 CGM, ParentName, cast<OMPTargetParallelDirective>(E)); 9475 break; 9476 case OMPD_target_teams: 9477 CodeGenFunction::EmitOMPTargetTeamsDeviceFunction( 9478 CGM, ParentName, cast<OMPTargetTeamsDirective>(E)); 9479 break; 9480 case OMPD_target_teams_distribute: 9481 CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction( 9482 CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E)); 9483 break; 9484 case OMPD_target_teams_distribute_simd: 9485 CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction( 9486 CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E)); 9487 break; 9488 case OMPD_target_parallel_for: 9489 CodeGenFunction::EmitOMPTargetParallelForDeviceFunction( 9490 CGM, ParentName, cast<OMPTargetParallelForDirective>(E)); 9491 break; 9492 case OMPD_target_parallel_for_simd: 9493 CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction( 9494 CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E)); 9495 break; 9496 case OMPD_target_simd: 9497 CodeGenFunction::EmitOMPTargetSimdDeviceFunction( 9498 CGM, ParentName, cast<OMPTargetSimdDirective>(E)); 9499 break; 9500 case OMPD_target_teams_distribute_parallel_for: 9501 CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction( 9502 CGM, ParentName, 9503 cast<OMPTargetTeamsDistributeParallelForDirective>(E)); 9504 break; 9505 case OMPD_target_teams_distribute_parallel_for_simd: 9506 CodeGenFunction:: 9507 EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction( 9508 CGM, ParentName, 9509 cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E)); 9510 break; 9511 case OMPD_parallel: 9512 case OMPD_for: 9513 case OMPD_parallel_for: 9514 case OMPD_parallel_master: 9515 case OMPD_parallel_sections: 9516 case OMPD_for_simd: 9517 case OMPD_parallel_for_simd: 9518 case OMPD_cancel: 9519 case OMPD_cancellation_point: 9520 case OMPD_ordered: 9521 case OMPD_threadprivate: 9522 case OMPD_allocate: 9523 case OMPD_task: 9524 case OMPD_simd: 9525 case OMPD_sections: 9526 case OMPD_section: 9527 case OMPD_single: 9528 case OMPD_master: 9529 case OMPD_critical: 9530 case OMPD_taskyield: 9531 case OMPD_barrier: 9532 case OMPD_taskwait: 9533 case OMPD_taskgroup: 9534 case OMPD_atomic: 9535 case OMPD_flush: 9536 case OMPD_teams: 9537 case OMPD_target_data: 9538 case OMPD_target_exit_data: 9539 case OMPD_target_enter_data: 9540 case OMPD_distribute: 9541 case OMPD_distribute_simd: 9542 case OMPD_distribute_parallel_for: 9543 case OMPD_distribute_parallel_for_simd: 9544 case OMPD_teams_distribute: 9545 case OMPD_teams_distribute_simd: 9546 case OMPD_teams_distribute_parallel_for: 9547 case OMPD_teams_distribute_parallel_for_simd: 9548 case OMPD_target_update: 9549 case OMPD_declare_simd: 9550 case OMPD_declare_variant: 9551 case OMPD_declare_target: 9552 case OMPD_end_declare_target: 9553 case OMPD_declare_reduction: 9554 case OMPD_declare_mapper: 9555 case OMPD_taskloop: 9556 case OMPD_taskloop_simd: 9557 case OMPD_master_taskloop: 9558 case OMPD_master_taskloop_simd: 9559 case OMPD_parallel_master_taskloop: 9560 case OMPD_parallel_master_taskloop_simd: 9561 case OMPD_requires: 9562 case OMPD_unknown: 9563 llvm_unreachable("Unknown target directive for OpenMP device codegen."); 9564 } 9565 return; 9566 } 9567 9568 if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) { 9569 if (!E->hasAssociatedStmt() || !E->getAssociatedStmt()) 9570 return; 9571 9572 scanForTargetRegionsFunctions( 9573 E->getInnermostCapturedStmt()->getCapturedStmt(), ParentName); 9574 return; 9575 } 9576 9577 // If this is a lambda function, look into its body. 9578 if (const auto *L = dyn_cast<LambdaExpr>(S)) 9579 S = L->getBody(); 9580 9581 // Keep looking for target regions recursively. 9582 for (const Stmt *II : S->children()) 9583 scanForTargetRegionsFunctions(II, ParentName); 9584 } 9585 9586 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) { 9587 // If emitting code for the host, we do not process FD here. Instead we do 9588 // the normal code generation. 9589 if (!CGM.getLangOpts().OpenMPIsDevice) { 9590 if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl())) { 9591 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 9592 OMPDeclareTargetDeclAttr::getDeviceType(FD); 9593 // Do not emit device_type(nohost) functions for the host. 9594 if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_NoHost) 9595 return true; 9596 } 9597 return false; 9598 } 9599 9600 const ValueDecl *VD = cast<ValueDecl>(GD.getDecl()); 9601 // Try to detect target regions in the function. 9602 if (const auto *FD = dyn_cast<FunctionDecl>(VD)) { 9603 StringRef Name = CGM.getMangledName(GD); 9604 scanForTargetRegionsFunctions(FD->getBody(), Name); 9605 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 9606 OMPDeclareTargetDeclAttr::getDeviceType(FD); 9607 // Do not emit device_type(nohost) functions for the host. 9608 if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_Host) 9609 return true; 9610 } 9611 9612 // Do not to emit function if it is not marked as declare target. 9613 return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) && 9614 AlreadyEmittedTargetDecls.count(VD) == 0; 9615 } 9616 9617 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 9618 if (!CGM.getLangOpts().OpenMPIsDevice) 9619 return false; 9620 9621 // Check if there are Ctors/Dtors in this declaration and look for target 9622 // regions in it. We use the complete variant to produce the kernel name 9623 // mangling. 9624 QualType RDTy = cast<VarDecl>(GD.getDecl())->getType(); 9625 if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) { 9626 for (const CXXConstructorDecl *Ctor : RD->ctors()) { 9627 StringRef ParentName = 9628 CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete)); 9629 scanForTargetRegionsFunctions(Ctor->getBody(), ParentName); 9630 } 9631 if (const CXXDestructorDecl *Dtor = RD->getDestructor()) { 9632 StringRef ParentName = 9633 CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete)); 9634 scanForTargetRegionsFunctions(Dtor->getBody(), ParentName); 9635 } 9636 } 9637 9638 // Do not to emit variable if it is not marked as declare target. 9639 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 9640 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration( 9641 cast<VarDecl>(GD.getDecl())); 9642 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 9643 (*Res == OMPDeclareTargetDeclAttr::MT_To && 9644 HasRequiresUnifiedSharedMemory)) { 9645 DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl())); 9646 return true; 9647 } 9648 return false; 9649 } 9650 9651 llvm::Constant * 9652 CGOpenMPRuntime::registerTargetFirstprivateCopy(CodeGenFunction &CGF, 9653 const VarDecl *VD) { 9654 assert(VD->getType().isConstant(CGM.getContext()) && 9655 "Expected constant variable."); 9656 StringRef VarName; 9657 llvm::Constant *Addr; 9658 llvm::GlobalValue::LinkageTypes Linkage; 9659 QualType Ty = VD->getType(); 9660 SmallString<128> Buffer; 9661 { 9662 unsigned DeviceID; 9663 unsigned FileID; 9664 unsigned Line; 9665 getTargetEntryUniqueInfo(CGM.getContext(), VD->getLocation(), DeviceID, 9666 FileID, Line); 9667 llvm::raw_svector_ostream OS(Buffer); 9668 OS << "__omp_offloading_firstprivate_" << llvm::format("_%x", DeviceID) 9669 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 9670 VarName = OS.str(); 9671 } 9672 Linkage = llvm::GlobalValue::InternalLinkage; 9673 Addr = 9674 getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(Ty), VarName, 9675 getDefaultFirstprivateAddressSpace()); 9676 cast<llvm::GlobalValue>(Addr)->setLinkage(Linkage); 9677 CharUnits VarSize = CGM.getContext().getTypeSizeInChars(Ty); 9678 CGM.addCompilerUsedGlobal(cast<llvm::GlobalValue>(Addr)); 9679 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 9680 VarName, Addr, VarSize, 9681 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo, Linkage); 9682 return Addr; 9683 } 9684 9685 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD, 9686 llvm::Constant *Addr) { 9687 if (CGM.getLangOpts().OMPTargetTriples.empty() && 9688 !CGM.getLangOpts().OpenMPIsDevice) 9689 return; 9690 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 9691 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 9692 if (!Res) { 9693 if (CGM.getLangOpts().OpenMPIsDevice) { 9694 // Register non-target variables being emitted in device code (debug info 9695 // may cause this). 9696 StringRef VarName = CGM.getMangledName(VD); 9697 EmittedNonTargetVariables.try_emplace(VarName, Addr); 9698 } 9699 return; 9700 } 9701 // Register declare target variables. 9702 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags; 9703 StringRef VarName; 9704 CharUnits VarSize; 9705 llvm::GlobalValue::LinkageTypes Linkage; 9706 9707 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 9708 !HasRequiresUnifiedSharedMemory) { 9709 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 9710 VarName = CGM.getMangledName(VD); 9711 if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) { 9712 VarSize = CGM.getContext().getTypeSizeInChars(VD->getType()); 9713 assert(!VarSize.isZero() && "Expected non-zero size of the variable"); 9714 } else { 9715 VarSize = CharUnits::Zero(); 9716 } 9717 Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false); 9718 // Temp solution to prevent optimizations of the internal variables. 9719 if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) { 9720 std::string RefName = getName({VarName, "ref"}); 9721 if (!CGM.GetGlobalValue(RefName)) { 9722 llvm::Constant *AddrRef = 9723 getOrCreateInternalVariable(Addr->getType(), RefName); 9724 auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef); 9725 GVAddrRef->setConstant(/*Val=*/true); 9726 GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage); 9727 GVAddrRef->setInitializer(Addr); 9728 CGM.addCompilerUsedGlobal(GVAddrRef); 9729 } 9730 } 9731 } else { 9732 assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 9733 (*Res == OMPDeclareTargetDeclAttr::MT_To && 9734 HasRequiresUnifiedSharedMemory)) && 9735 "Declare target attribute must link or to with unified memory."); 9736 if (*Res == OMPDeclareTargetDeclAttr::MT_Link) 9737 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink; 9738 else 9739 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 9740 9741 if (CGM.getLangOpts().OpenMPIsDevice) { 9742 VarName = Addr->getName(); 9743 Addr = nullptr; 9744 } else { 9745 VarName = getAddrOfDeclareTargetVar(VD).getName(); 9746 Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer()); 9747 } 9748 VarSize = CGM.getPointerSize(); 9749 Linkage = llvm::GlobalValue::WeakAnyLinkage; 9750 } 9751 9752 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 9753 VarName, Addr, VarSize, Flags, Linkage); 9754 } 9755 9756 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) { 9757 if (isa<FunctionDecl>(GD.getDecl()) || 9758 isa<OMPDeclareReductionDecl>(GD.getDecl())) 9759 return emitTargetFunctions(GD); 9760 9761 return emitTargetGlobalVariable(GD); 9762 } 9763 9764 void CGOpenMPRuntime::emitDeferredTargetDecls() const { 9765 for (const VarDecl *VD : DeferredGlobalVariables) { 9766 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 9767 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 9768 if (!Res) 9769 continue; 9770 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 9771 !HasRequiresUnifiedSharedMemory) { 9772 CGM.EmitGlobal(VD); 9773 } else { 9774 assert((*Res == OMPDeclareTargetDeclAttr::MT_Link || 9775 (*Res == OMPDeclareTargetDeclAttr::MT_To && 9776 HasRequiresUnifiedSharedMemory)) && 9777 "Expected link clause or to clause with unified memory."); 9778 (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 9779 } 9780 } 9781 } 9782 9783 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas( 9784 CodeGenFunction &CGF, const OMPExecutableDirective &D) const { 9785 assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) && 9786 " Expected target-based directive."); 9787 } 9788 9789 void CGOpenMPRuntime::processRequiresDirective(const OMPRequiresDecl *D) { 9790 for (const OMPClause *Clause : D->clauselists()) { 9791 if (Clause->getClauseKind() == OMPC_unified_shared_memory) { 9792 HasRequiresUnifiedSharedMemory = true; 9793 } else if (const auto *AC = 9794 dyn_cast<OMPAtomicDefaultMemOrderClause>(Clause)) { 9795 switch (AC->getAtomicDefaultMemOrderKind()) { 9796 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel: 9797 RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease; 9798 break; 9799 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst: 9800 RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent; 9801 break; 9802 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed: 9803 RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic; 9804 break; 9805 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown: 9806 break; 9807 } 9808 } 9809 } 9810 } 9811 9812 llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const { 9813 return RequiresAtomicOrdering; 9814 } 9815 9816 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD, 9817 LangAS &AS) { 9818 if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>()) 9819 return false; 9820 const auto *A = VD->getAttr<OMPAllocateDeclAttr>(); 9821 switch(A->getAllocatorType()) { 9822 case OMPAllocateDeclAttr::OMPDefaultMemAlloc: 9823 // Not supported, fallback to the default mem space. 9824 case OMPAllocateDeclAttr::OMPLargeCapMemAlloc: 9825 case OMPAllocateDeclAttr::OMPCGroupMemAlloc: 9826 case OMPAllocateDeclAttr::OMPHighBWMemAlloc: 9827 case OMPAllocateDeclAttr::OMPLowLatMemAlloc: 9828 case OMPAllocateDeclAttr::OMPThreadMemAlloc: 9829 case OMPAllocateDeclAttr::OMPConstMemAlloc: 9830 case OMPAllocateDeclAttr::OMPPTeamMemAlloc: 9831 AS = LangAS::Default; 9832 return true; 9833 case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc: 9834 llvm_unreachable("Expected predefined allocator for the variables with the " 9835 "static storage."); 9836 } 9837 return false; 9838 } 9839 9840 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const { 9841 return HasRequiresUnifiedSharedMemory; 9842 } 9843 9844 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII( 9845 CodeGenModule &CGM) 9846 : CGM(CGM) { 9847 if (CGM.getLangOpts().OpenMPIsDevice) { 9848 SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal; 9849 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false; 9850 } 9851 } 9852 9853 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() { 9854 if (CGM.getLangOpts().OpenMPIsDevice) 9855 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal; 9856 } 9857 9858 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) { 9859 if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal) 9860 return true; 9861 9862 const auto *D = cast<FunctionDecl>(GD.getDecl()); 9863 // Do not to emit function if it is marked as declare target as it was already 9864 // emitted. 9865 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) { 9866 if (D->hasBody() && AlreadyEmittedTargetDecls.count(D) == 0) { 9867 if (auto *F = dyn_cast_or_null<llvm::Function>( 9868 CGM.GetGlobalValue(CGM.getMangledName(GD)))) 9869 return !F->isDeclaration(); 9870 return false; 9871 } 9872 return true; 9873 } 9874 9875 return !AlreadyEmittedTargetDecls.insert(D).second; 9876 } 9877 9878 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() { 9879 // If we don't have entries or if we are emitting code for the device, we 9880 // don't need to do anything. 9881 if (CGM.getLangOpts().OMPTargetTriples.empty() || 9882 CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice || 9883 (OffloadEntriesInfoManager.empty() && 9884 !HasEmittedDeclareTargetRegion && 9885 !HasEmittedTargetRegion)) 9886 return nullptr; 9887 9888 // Create and register the function that handles the requires directives. 9889 ASTContext &C = CGM.getContext(); 9890 9891 llvm::Function *RequiresRegFn; 9892 { 9893 CodeGenFunction CGF(CGM); 9894 const auto &FI = CGM.getTypes().arrangeNullaryFunction(); 9895 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 9896 std::string ReqName = getName({"omp_offloading", "requires_reg"}); 9897 RequiresRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, ReqName, FI); 9898 CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {}); 9899 OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE; 9900 // TODO: check for other requires clauses. 9901 // The requires directive takes effect only when a target region is 9902 // present in the compilation unit. Otherwise it is ignored and not 9903 // passed to the runtime. This avoids the runtime from throwing an error 9904 // for mismatching requires clauses across compilation units that don't 9905 // contain at least 1 target region. 9906 assert((HasEmittedTargetRegion || 9907 HasEmittedDeclareTargetRegion || 9908 !OffloadEntriesInfoManager.empty()) && 9909 "Target or declare target region expected."); 9910 if (HasRequiresUnifiedSharedMemory) 9911 Flags = OMP_REQ_UNIFIED_SHARED_MEMORY; 9912 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_requires), 9913 llvm::ConstantInt::get(CGM.Int64Ty, Flags)); 9914 CGF.FinishFunction(); 9915 } 9916 return RequiresRegFn; 9917 } 9918 9919 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF, 9920 const OMPExecutableDirective &D, 9921 SourceLocation Loc, 9922 llvm::Function *OutlinedFn, 9923 ArrayRef<llvm::Value *> CapturedVars) { 9924 if (!CGF.HaveInsertPoint()) 9925 return; 9926 9927 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 9928 CodeGenFunction::RunCleanupsScope Scope(CGF); 9929 9930 // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn); 9931 llvm::Value *Args[] = { 9932 RTLoc, 9933 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 9934 CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())}; 9935 llvm::SmallVector<llvm::Value *, 16> RealArgs; 9936 RealArgs.append(std::begin(Args), std::end(Args)); 9937 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 9938 9939 llvm::FunctionCallee RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_teams); 9940 CGF.EmitRuntimeCall(RTLFn, RealArgs); 9941 } 9942 9943 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 9944 const Expr *NumTeams, 9945 const Expr *ThreadLimit, 9946 SourceLocation Loc) { 9947 if (!CGF.HaveInsertPoint()) 9948 return; 9949 9950 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 9951 9952 llvm::Value *NumTeamsVal = 9953 NumTeams 9954 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams), 9955 CGF.CGM.Int32Ty, /* isSigned = */ true) 9956 : CGF.Builder.getInt32(0); 9957 9958 llvm::Value *ThreadLimitVal = 9959 ThreadLimit 9960 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit), 9961 CGF.CGM.Int32Ty, /* isSigned = */ true) 9962 : CGF.Builder.getInt32(0); 9963 9964 // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit) 9965 llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal, 9966 ThreadLimitVal}; 9967 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_teams), 9968 PushNumTeamsArgs); 9969 } 9970 9971 void CGOpenMPRuntime::emitTargetDataCalls( 9972 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 9973 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 9974 if (!CGF.HaveInsertPoint()) 9975 return; 9976 9977 // Action used to replace the default codegen action and turn privatization 9978 // off. 9979 PrePostActionTy NoPrivAction; 9980 9981 // Generate the code for the opening of the data environment. Capture all the 9982 // arguments of the runtime call by reference because they are used in the 9983 // closing of the region. 9984 auto &&BeginThenGen = [this, &D, Device, &Info, 9985 &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) { 9986 // Fill up the arrays with all the mapped variables. 9987 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 9988 MappableExprsHandler::MapValuesArrayTy Pointers; 9989 MappableExprsHandler::MapValuesArrayTy Sizes; 9990 MappableExprsHandler::MapFlagsArrayTy MapTypes; 9991 9992 // Get map clause information. 9993 MappableExprsHandler MCHandler(D, CGF); 9994 MCHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes); 9995 9996 // Fill up the arrays and create the arguments. 9997 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 9998 9999 llvm::Value *BasePointersArrayArg = nullptr; 10000 llvm::Value *PointersArrayArg = nullptr; 10001 llvm::Value *SizesArrayArg = nullptr; 10002 llvm::Value *MapTypesArrayArg = nullptr; 10003 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 10004 SizesArrayArg, MapTypesArrayArg, Info); 10005 10006 // Emit device ID if any. 10007 llvm::Value *DeviceID = nullptr; 10008 if (Device) { 10009 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10010 CGF.Int64Ty, /*isSigned=*/true); 10011 } else { 10012 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10013 } 10014 10015 // Emit the number of elements in the offloading arrays. 10016 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 10017 10018 llvm::Value *OffloadingArgs[] = { 10019 DeviceID, PointerNum, BasePointersArrayArg, 10020 PointersArrayArg, SizesArrayArg, MapTypesArrayArg}; 10021 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_begin), 10022 OffloadingArgs); 10023 10024 // If device pointer privatization is required, emit the body of the region 10025 // here. It will have to be duplicated: with and without privatization. 10026 if (!Info.CaptureDeviceAddrMap.empty()) 10027 CodeGen(CGF); 10028 }; 10029 10030 // Generate code for the closing of the data region. 10031 auto &&EndThenGen = [this, Device, &Info](CodeGenFunction &CGF, 10032 PrePostActionTy &) { 10033 assert(Info.isValid() && "Invalid data environment closing arguments."); 10034 10035 llvm::Value *BasePointersArrayArg = nullptr; 10036 llvm::Value *PointersArrayArg = nullptr; 10037 llvm::Value *SizesArrayArg = nullptr; 10038 llvm::Value *MapTypesArrayArg = nullptr; 10039 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 10040 SizesArrayArg, MapTypesArrayArg, Info); 10041 10042 // Emit device ID if any. 10043 llvm::Value *DeviceID = nullptr; 10044 if (Device) { 10045 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10046 CGF.Int64Ty, /*isSigned=*/true); 10047 } else { 10048 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10049 } 10050 10051 // Emit the number of elements in the offloading arrays. 10052 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 10053 10054 llvm::Value *OffloadingArgs[] = { 10055 DeviceID, PointerNum, BasePointersArrayArg, 10056 PointersArrayArg, SizesArrayArg, MapTypesArrayArg}; 10057 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_end), 10058 OffloadingArgs); 10059 }; 10060 10061 // If we need device pointer privatization, we need to emit the body of the 10062 // region with no privatization in the 'else' branch of the conditional. 10063 // Otherwise, we don't have to do anything. 10064 auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF, 10065 PrePostActionTy &) { 10066 if (!Info.CaptureDeviceAddrMap.empty()) { 10067 CodeGen.setAction(NoPrivAction); 10068 CodeGen(CGF); 10069 } 10070 }; 10071 10072 // We don't have to do anything to close the region if the if clause evaluates 10073 // to false. 10074 auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {}; 10075 10076 if (IfCond) { 10077 emitIfClause(CGF, IfCond, BeginThenGen, BeginElseGen); 10078 } else { 10079 RegionCodeGenTy RCG(BeginThenGen); 10080 RCG(CGF); 10081 } 10082 10083 // If we don't require privatization of device pointers, we emit the body in 10084 // between the runtime calls. This avoids duplicating the body code. 10085 if (Info.CaptureDeviceAddrMap.empty()) { 10086 CodeGen.setAction(NoPrivAction); 10087 CodeGen(CGF); 10088 } 10089 10090 if (IfCond) { 10091 emitIfClause(CGF, IfCond, EndThenGen, EndElseGen); 10092 } else { 10093 RegionCodeGenTy RCG(EndThenGen); 10094 RCG(CGF); 10095 } 10096 } 10097 10098 void CGOpenMPRuntime::emitTargetDataStandAloneCall( 10099 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 10100 const Expr *Device) { 10101 if (!CGF.HaveInsertPoint()) 10102 return; 10103 10104 assert((isa<OMPTargetEnterDataDirective>(D) || 10105 isa<OMPTargetExitDataDirective>(D) || 10106 isa<OMPTargetUpdateDirective>(D)) && 10107 "Expecting either target enter, exit data, or update directives."); 10108 10109 CodeGenFunction::OMPTargetDataInfo InputInfo; 10110 llvm::Value *MapTypesArray = nullptr; 10111 // Generate the code for the opening of the data environment. 10112 auto &&ThenGen = [this, &D, Device, &InputInfo, 10113 &MapTypesArray](CodeGenFunction &CGF, PrePostActionTy &) { 10114 // Emit device ID if any. 10115 llvm::Value *DeviceID = nullptr; 10116 if (Device) { 10117 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10118 CGF.Int64Ty, /*isSigned=*/true); 10119 } else { 10120 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10121 } 10122 10123 // Emit the number of elements in the offloading arrays. 10124 llvm::Constant *PointerNum = 10125 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 10126 10127 llvm::Value *OffloadingArgs[] = {DeviceID, 10128 PointerNum, 10129 InputInfo.BasePointersArray.getPointer(), 10130 InputInfo.PointersArray.getPointer(), 10131 InputInfo.SizesArray.getPointer(), 10132 MapTypesArray}; 10133 10134 // Select the right runtime function call for each expected standalone 10135 // directive. 10136 const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 10137 OpenMPRTLFunction RTLFn; 10138 switch (D.getDirectiveKind()) { 10139 case OMPD_target_enter_data: 10140 RTLFn = HasNowait ? OMPRTL__tgt_target_data_begin_nowait 10141 : OMPRTL__tgt_target_data_begin; 10142 break; 10143 case OMPD_target_exit_data: 10144 RTLFn = HasNowait ? OMPRTL__tgt_target_data_end_nowait 10145 : OMPRTL__tgt_target_data_end; 10146 break; 10147 case OMPD_target_update: 10148 RTLFn = HasNowait ? OMPRTL__tgt_target_data_update_nowait 10149 : OMPRTL__tgt_target_data_update; 10150 break; 10151 case OMPD_parallel: 10152 case OMPD_for: 10153 case OMPD_parallel_for: 10154 case OMPD_parallel_master: 10155 case OMPD_parallel_sections: 10156 case OMPD_for_simd: 10157 case OMPD_parallel_for_simd: 10158 case OMPD_cancel: 10159 case OMPD_cancellation_point: 10160 case OMPD_ordered: 10161 case OMPD_threadprivate: 10162 case OMPD_allocate: 10163 case OMPD_task: 10164 case OMPD_simd: 10165 case OMPD_sections: 10166 case OMPD_section: 10167 case OMPD_single: 10168 case OMPD_master: 10169 case OMPD_critical: 10170 case OMPD_taskyield: 10171 case OMPD_barrier: 10172 case OMPD_taskwait: 10173 case OMPD_taskgroup: 10174 case OMPD_atomic: 10175 case OMPD_flush: 10176 case OMPD_teams: 10177 case OMPD_target_data: 10178 case OMPD_distribute: 10179 case OMPD_distribute_simd: 10180 case OMPD_distribute_parallel_for: 10181 case OMPD_distribute_parallel_for_simd: 10182 case OMPD_teams_distribute: 10183 case OMPD_teams_distribute_simd: 10184 case OMPD_teams_distribute_parallel_for: 10185 case OMPD_teams_distribute_parallel_for_simd: 10186 case OMPD_declare_simd: 10187 case OMPD_declare_variant: 10188 case OMPD_declare_target: 10189 case OMPD_end_declare_target: 10190 case OMPD_declare_reduction: 10191 case OMPD_declare_mapper: 10192 case OMPD_taskloop: 10193 case OMPD_taskloop_simd: 10194 case OMPD_master_taskloop: 10195 case OMPD_master_taskloop_simd: 10196 case OMPD_parallel_master_taskloop: 10197 case OMPD_parallel_master_taskloop_simd: 10198 case OMPD_target: 10199 case OMPD_target_simd: 10200 case OMPD_target_teams_distribute: 10201 case OMPD_target_teams_distribute_simd: 10202 case OMPD_target_teams_distribute_parallel_for: 10203 case OMPD_target_teams_distribute_parallel_for_simd: 10204 case OMPD_target_teams: 10205 case OMPD_target_parallel: 10206 case OMPD_target_parallel_for: 10207 case OMPD_target_parallel_for_simd: 10208 case OMPD_requires: 10209 case OMPD_unknown: 10210 llvm_unreachable("Unexpected standalone target data directive."); 10211 break; 10212 } 10213 CGF.EmitRuntimeCall(createRuntimeFunction(RTLFn), OffloadingArgs); 10214 }; 10215 10216 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray]( 10217 CodeGenFunction &CGF, PrePostActionTy &) { 10218 // Fill up the arrays with all the mapped variables. 10219 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 10220 MappableExprsHandler::MapValuesArrayTy Pointers; 10221 MappableExprsHandler::MapValuesArrayTy Sizes; 10222 MappableExprsHandler::MapFlagsArrayTy MapTypes; 10223 10224 // Get map clause information. 10225 MappableExprsHandler MEHandler(D, CGF); 10226 MEHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes); 10227 10228 TargetDataInfo Info; 10229 // Fill up the arrays and create the arguments. 10230 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 10231 emitOffloadingArraysArgument(CGF, Info.BasePointersArray, 10232 Info.PointersArray, Info.SizesArray, 10233 Info.MapTypesArray, Info); 10234 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 10235 InputInfo.BasePointersArray = 10236 Address(Info.BasePointersArray, CGM.getPointerAlign()); 10237 InputInfo.PointersArray = 10238 Address(Info.PointersArray, CGM.getPointerAlign()); 10239 InputInfo.SizesArray = 10240 Address(Info.SizesArray, CGM.getPointerAlign()); 10241 MapTypesArray = Info.MapTypesArray; 10242 if (D.hasClausesOfKind<OMPDependClause>()) 10243 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 10244 else 10245 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 10246 }; 10247 10248 if (IfCond) { 10249 emitIfClause(CGF, IfCond, TargetThenGen, 10250 [](CodeGenFunction &CGF, PrePostActionTy &) {}); 10251 } else { 10252 RegionCodeGenTy ThenRCG(TargetThenGen); 10253 ThenRCG(CGF); 10254 } 10255 } 10256 10257 namespace { 10258 /// Kind of parameter in a function with 'declare simd' directive. 10259 enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector }; 10260 /// Attribute set of the parameter. 10261 struct ParamAttrTy { 10262 ParamKindTy Kind = Vector; 10263 llvm::APSInt StrideOrArg; 10264 llvm::APSInt Alignment; 10265 }; 10266 } // namespace 10267 10268 static unsigned evaluateCDTSize(const FunctionDecl *FD, 10269 ArrayRef<ParamAttrTy> ParamAttrs) { 10270 // Every vector variant of a SIMD-enabled function has a vector length (VLEN). 10271 // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument 10272 // of that clause. The VLEN value must be power of 2. 10273 // In other case the notion of the function`s "characteristic data type" (CDT) 10274 // is used to compute the vector length. 10275 // CDT is defined in the following order: 10276 // a) For non-void function, the CDT is the return type. 10277 // b) If the function has any non-uniform, non-linear parameters, then the 10278 // CDT is the type of the first such parameter. 10279 // c) If the CDT determined by a) or b) above is struct, union, or class 10280 // type which is pass-by-value (except for the type that maps to the 10281 // built-in complex data type), the characteristic data type is int. 10282 // d) If none of the above three cases is applicable, the CDT is int. 10283 // The VLEN is then determined based on the CDT and the size of vector 10284 // register of that ISA for which current vector version is generated. The 10285 // VLEN is computed using the formula below: 10286 // VLEN = sizeof(vector_register) / sizeof(CDT), 10287 // where vector register size specified in section 3.2.1 Registers and the 10288 // Stack Frame of original AMD64 ABI document. 10289 QualType RetType = FD->getReturnType(); 10290 if (RetType.isNull()) 10291 return 0; 10292 ASTContext &C = FD->getASTContext(); 10293 QualType CDT; 10294 if (!RetType.isNull() && !RetType->isVoidType()) { 10295 CDT = RetType; 10296 } else { 10297 unsigned Offset = 0; 10298 if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) { 10299 if (ParamAttrs[Offset].Kind == Vector) 10300 CDT = C.getPointerType(C.getRecordType(MD->getParent())); 10301 ++Offset; 10302 } 10303 if (CDT.isNull()) { 10304 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 10305 if (ParamAttrs[I + Offset].Kind == Vector) { 10306 CDT = FD->getParamDecl(I)->getType(); 10307 break; 10308 } 10309 } 10310 } 10311 } 10312 if (CDT.isNull()) 10313 CDT = C.IntTy; 10314 CDT = CDT->getCanonicalTypeUnqualified(); 10315 if (CDT->isRecordType() || CDT->isUnionType()) 10316 CDT = C.IntTy; 10317 return C.getTypeSize(CDT); 10318 } 10319 10320 static void 10321 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn, 10322 const llvm::APSInt &VLENVal, 10323 ArrayRef<ParamAttrTy> ParamAttrs, 10324 OMPDeclareSimdDeclAttr::BranchStateTy State) { 10325 struct ISADataTy { 10326 char ISA; 10327 unsigned VecRegSize; 10328 }; 10329 ISADataTy ISAData[] = { 10330 { 10331 'b', 128 10332 }, // SSE 10333 { 10334 'c', 256 10335 }, // AVX 10336 { 10337 'd', 256 10338 }, // AVX2 10339 { 10340 'e', 512 10341 }, // AVX512 10342 }; 10343 llvm::SmallVector<char, 2> Masked; 10344 switch (State) { 10345 case OMPDeclareSimdDeclAttr::BS_Undefined: 10346 Masked.push_back('N'); 10347 Masked.push_back('M'); 10348 break; 10349 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10350 Masked.push_back('N'); 10351 break; 10352 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10353 Masked.push_back('M'); 10354 break; 10355 } 10356 for (char Mask : Masked) { 10357 for (const ISADataTy &Data : ISAData) { 10358 SmallString<256> Buffer; 10359 llvm::raw_svector_ostream Out(Buffer); 10360 Out << "_ZGV" << Data.ISA << Mask; 10361 if (!VLENVal) { 10362 unsigned NumElts = evaluateCDTSize(FD, ParamAttrs); 10363 assert(NumElts && "Non-zero simdlen/cdtsize expected"); 10364 Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts); 10365 } else { 10366 Out << VLENVal; 10367 } 10368 for (const ParamAttrTy &ParamAttr : ParamAttrs) { 10369 switch (ParamAttr.Kind){ 10370 case LinearWithVarStride: 10371 Out << 's' << ParamAttr.StrideOrArg; 10372 break; 10373 case Linear: 10374 Out << 'l'; 10375 if (!!ParamAttr.StrideOrArg) 10376 Out << ParamAttr.StrideOrArg; 10377 break; 10378 case Uniform: 10379 Out << 'u'; 10380 break; 10381 case Vector: 10382 Out << 'v'; 10383 break; 10384 } 10385 if (!!ParamAttr.Alignment) 10386 Out << 'a' << ParamAttr.Alignment; 10387 } 10388 Out << '_' << Fn->getName(); 10389 Fn->addFnAttr(Out.str()); 10390 } 10391 } 10392 } 10393 10394 // This are the Functions that are needed to mangle the name of the 10395 // vector functions generated by the compiler, according to the rules 10396 // defined in the "Vector Function ABI specifications for AArch64", 10397 // available at 10398 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi. 10399 10400 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI. 10401 /// 10402 /// TODO: Need to implement the behavior for reference marked with a 10403 /// var or no linear modifiers (1.b in the section). For this, we 10404 /// need to extend ParamKindTy to support the linear modifiers. 10405 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) { 10406 QT = QT.getCanonicalType(); 10407 10408 if (QT->isVoidType()) 10409 return false; 10410 10411 if (Kind == ParamKindTy::Uniform) 10412 return false; 10413 10414 if (Kind == ParamKindTy::Linear) 10415 return false; 10416 10417 // TODO: Handle linear references with modifiers 10418 10419 if (Kind == ParamKindTy::LinearWithVarStride) 10420 return false; 10421 10422 return true; 10423 } 10424 10425 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI. 10426 static bool getAArch64PBV(QualType QT, ASTContext &C) { 10427 QT = QT.getCanonicalType(); 10428 unsigned Size = C.getTypeSize(QT); 10429 10430 // Only scalars and complex within 16 bytes wide set PVB to true. 10431 if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128) 10432 return false; 10433 10434 if (QT->isFloatingType()) 10435 return true; 10436 10437 if (QT->isIntegerType()) 10438 return true; 10439 10440 if (QT->isPointerType()) 10441 return true; 10442 10443 // TODO: Add support for complex types (section 3.1.2, item 2). 10444 10445 return false; 10446 } 10447 10448 /// Computes the lane size (LS) of a return type or of an input parameter, 10449 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI. 10450 /// TODO: Add support for references, section 3.2.1, item 1. 10451 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) { 10452 if (getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) { 10453 QualType PTy = QT.getCanonicalType()->getPointeeType(); 10454 if (getAArch64PBV(PTy, C)) 10455 return C.getTypeSize(PTy); 10456 } 10457 if (getAArch64PBV(QT, C)) 10458 return C.getTypeSize(QT); 10459 10460 return C.getTypeSize(C.getUIntPtrType()); 10461 } 10462 10463 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the 10464 // signature of the scalar function, as defined in 3.2.2 of the 10465 // AAVFABI. 10466 static std::tuple<unsigned, unsigned, bool> 10467 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) { 10468 QualType RetType = FD->getReturnType().getCanonicalType(); 10469 10470 ASTContext &C = FD->getASTContext(); 10471 10472 bool OutputBecomesInput = false; 10473 10474 llvm::SmallVector<unsigned, 8> Sizes; 10475 if (!RetType->isVoidType()) { 10476 Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C)); 10477 if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {})) 10478 OutputBecomesInput = true; 10479 } 10480 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 10481 QualType QT = FD->getParamDecl(I)->getType().getCanonicalType(); 10482 Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C)); 10483 } 10484 10485 assert(!Sizes.empty() && "Unable to determine NDS and WDS."); 10486 // The LS of a function parameter / return value can only be a power 10487 // of 2, starting from 8 bits, up to 128. 10488 assert(std::all_of(Sizes.begin(), Sizes.end(), 10489 [](unsigned Size) { 10490 return Size == 8 || Size == 16 || Size == 32 || 10491 Size == 64 || Size == 128; 10492 }) && 10493 "Invalid size"); 10494 10495 return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)), 10496 *std::max_element(std::begin(Sizes), std::end(Sizes)), 10497 OutputBecomesInput); 10498 } 10499 10500 /// Mangle the parameter part of the vector function name according to 10501 /// their OpenMP classification. The mangling function is defined in 10502 /// section 3.5 of the AAVFABI. 10503 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) { 10504 SmallString<256> Buffer; 10505 llvm::raw_svector_ostream Out(Buffer); 10506 for (const auto &ParamAttr : ParamAttrs) { 10507 switch (ParamAttr.Kind) { 10508 case LinearWithVarStride: 10509 Out << "ls" << ParamAttr.StrideOrArg; 10510 break; 10511 case Linear: 10512 Out << 'l'; 10513 // Don't print the step value if it is not present or if it is 10514 // equal to 1. 10515 if (!!ParamAttr.StrideOrArg && ParamAttr.StrideOrArg != 1) 10516 Out << ParamAttr.StrideOrArg; 10517 break; 10518 case Uniform: 10519 Out << 'u'; 10520 break; 10521 case Vector: 10522 Out << 'v'; 10523 break; 10524 } 10525 10526 if (!!ParamAttr.Alignment) 10527 Out << 'a' << ParamAttr.Alignment; 10528 } 10529 10530 return std::string(Out.str()); 10531 } 10532 10533 // Function used to add the attribute. The parameter `VLEN` is 10534 // templated to allow the use of "x" when targeting scalable functions 10535 // for SVE. 10536 template <typename T> 10537 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix, 10538 char ISA, StringRef ParSeq, 10539 StringRef MangledName, bool OutputBecomesInput, 10540 llvm::Function *Fn) { 10541 SmallString<256> Buffer; 10542 llvm::raw_svector_ostream Out(Buffer); 10543 Out << Prefix << ISA << LMask << VLEN; 10544 if (OutputBecomesInput) 10545 Out << "v"; 10546 Out << ParSeq << "_" << MangledName; 10547 Fn->addFnAttr(Out.str()); 10548 } 10549 10550 // Helper function to generate the Advanced SIMD names depending on 10551 // the value of the NDS when simdlen is not present. 10552 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask, 10553 StringRef Prefix, char ISA, 10554 StringRef ParSeq, StringRef MangledName, 10555 bool OutputBecomesInput, 10556 llvm::Function *Fn) { 10557 switch (NDS) { 10558 case 8: 10559 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 10560 OutputBecomesInput, Fn); 10561 addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName, 10562 OutputBecomesInput, Fn); 10563 break; 10564 case 16: 10565 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 10566 OutputBecomesInput, Fn); 10567 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 10568 OutputBecomesInput, Fn); 10569 break; 10570 case 32: 10571 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 10572 OutputBecomesInput, Fn); 10573 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 10574 OutputBecomesInput, Fn); 10575 break; 10576 case 64: 10577 case 128: 10578 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 10579 OutputBecomesInput, Fn); 10580 break; 10581 default: 10582 llvm_unreachable("Scalar type is too wide."); 10583 } 10584 } 10585 10586 /// Emit vector function attributes for AArch64, as defined in the AAVFABI. 10587 static void emitAArch64DeclareSimdFunction( 10588 CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN, 10589 ArrayRef<ParamAttrTy> ParamAttrs, 10590 OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName, 10591 char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) { 10592 10593 // Get basic data for building the vector signature. 10594 const auto Data = getNDSWDS(FD, ParamAttrs); 10595 const unsigned NDS = std::get<0>(Data); 10596 const unsigned WDS = std::get<1>(Data); 10597 const bool OutputBecomesInput = std::get<2>(Data); 10598 10599 // Check the values provided via `simdlen` by the user. 10600 // 1. A `simdlen(1)` doesn't produce vector signatures, 10601 if (UserVLEN == 1) { 10602 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10603 DiagnosticsEngine::Warning, 10604 "The clause simdlen(1) has no effect when targeting aarch64."); 10605 CGM.getDiags().Report(SLoc, DiagID); 10606 return; 10607 } 10608 10609 // 2. Section 3.3.1, item 1: user input must be a power of 2 for 10610 // Advanced SIMD output. 10611 if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) { 10612 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10613 DiagnosticsEngine::Warning, "The value specified in simdlen must be a " 10614 "power of 2 when targeting Advanced SIMD."); 10615 CGM.getDiags().Report(SLoc, DiagID); 10616 return; 10617 } 10618 10619 // 3. Section 3.4.1. SVE fixed lengh must obey the architectural 10620 // limits. 10621 if (ISA == 's' && UserVLEN != 0) { 10622 if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) { 10623 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10624 DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit " 10625 "lanes in the architectural constraints " 10626 "for SVE (min is 128-bit, max is " 10627 "2048-bit, by steps of 128-bit)"); 10628 CGM.getDiags().Report(SLoc, DiagID) << WDS; 10629 return; 10630 } 10631 } 10632 10633 // Sort out parameter sequence. 10634 const std::string ParSeq = mangleVectorParameters(ParamAttrs); 10635 StringRef Prefix = "_ZGV"; 10636 // Generate simdlen from user input (if any). 10637 if (UserVLEN) { 10638 if (ISA == 's') { 10639 // SVE generates only a masked function. 10640 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10641 OutputBecomesInput, Fn); 10642 } else { 10643 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 10644 // Advanced SIMD generates one or two functions, depending on 10645 // the `[not]inbranch` clause. 10646 switch (State) { 10647 case OMPDeclareSimdDeclAttr::BS_Undefined: 10648 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 10649 OutputBecomesInput, Fn); 10650 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10651 OutputBecomesInput, Fn); 10652 break; 10653 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10654 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 10655 OutputBecomesInput, Fn); 10656 break; 10657 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10658 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10659 OutputBecomesInput, Fn); 10660 break; 10661 } 10662 } 10663 } else { 10664 // If no user simdlen is provided, follow the AAVFABI rules for 10665 // generating the vector length. 10666 if (ISA == 's') { 10667 // SVE, section 3.4.1, item 1. 10668 addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName, 10669 OutputBecomesInput, Fn); 10670 } else { 10671 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 10672 // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or 10673 // two vector names depending on the use of the clause 10674 // `[not]inbranch`. 10675 switch (State) { 10676 case OMPDeclareSimdDeclAttr::BS_Undefined: 10677 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 10678 OutputBecomesInput, Fn); 10679 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 10680 OutputBecomesInput, Fn); 10681 break; 10682 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10683 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 10684 OutputBecomesInput, Fn); 10685 break; 10686 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10687 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 10688 OutputBecomesInput, Fn); 10689 break; 10690 } 10691 } 10692 } 10693 } 10694 10695 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD, 10696 llvm::Function *Fn) { 10697 ASTContext &C = CGM.getContext(); 10698 FD = FD->getMostRecentDecl(); 10699 // Map params to their positions in function decl. 10700 llvm::DenseMap<const Decl *, unsigned> ParamPositions; 10701 if (isa<CXXMethodDecl>(FD)) 10702 ParamPositions.try_emplace(FD, 0); 10703 unsigned ParamPos = ParamPositions.size(); 10704 for (const ParmVarDecl *P : FD->parameters()) { 10705 ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos); 10706 ++ParamPos; 10707 } 10708 while (FD) { 10709 for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) { 10710 llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size()); 10711 // Mark uniform parameters. 10712 for (const Expr *E : Attr->uniforms()) { 10713 E = E->IgnoreParenImpCasts(); 10714 unsigned Pos; 10715 if (isa<CXXThisExpr>(E)) { 10716 Pos = ParamPositions[FD]; 10717 } else { 10718 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 10719 ->getCanonicalDecl(); 10720 Pos = ParamPositions[PVD]; 10721 } 10722 ParamAttrs[Pos].Kind = Uniform; 10723 } 10724 // Get alignment info. 10725 auto NI = Attr->alignments_begin(); 10726 for (const Expr *E : Attr->aligneds()) { 10727 E = E->IgnoreParenImpCasts(); 10728 unsigned Pos; 10729 QualType ParmTy; 10730 if (isa<CXXThisExpr>(E)) { 10731 Pos = ParamPositions[FD]; 10732 ParmTy = E->getType(); 10733 } else { 10734 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 10735 ->getCanonicalDecl(); 10736 Pos = ParamPositions[PVD]; 10737 ParmTy = PVD->getType(); 10738 } 10739 ParamAttrs[Pos].Alignment = 10740 (*NI) 10741 ? (*NI)->EvaluateKnownConstInt(C) 10742 : llvm::APSInt::getUnsigned( 10743 C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy)) 10744 .getQuantity()); 10745 ++NI; 10746 } 10747 // Mark linear parameters. 10748 auto SI = Attr->steps_begin(); 10749 auto MI = Attr->modifiers_begin(); 10750 for (const Expr *E : Attr->linears()) { 10751 E = E->IgnoreParenImpCasts(); 10752 unsigned Pos; 10753 if (isa<CXXThisExpr>(E)) { 10754 Pos = ParamPositions[FD]; 10755 } else { 10756 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 10757 ->getCanonicalDecl(); 10758 Pos = ParamPositions[PVD]; 10759 } 10760 ParamAttrTy &ParamAttr = ParamAttrs[Pos]; 10761 ParamAttr.Kind = Linear; 10762 if (*SI) { 10763 Expr::EvalResult Result; 10764 if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) { 10765 if (const auto *DRE = 10766 cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) { 10767 if (const auto *StridePVD = cast<ParmVarDecl>(DRE->getDecl())) { 10768 ParamAttr.Kind = LinearWithVarStride; 10769 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned( 10770 ParamPositions[StridePVD->getCanonicalDecl()]); 10771 } 10772 } 10773 } else { 10774 ParamAttr.StrideOrArg = Result.Val.getInt(); 10775 } 10776 } 10777 ++SI; 10778 ++MI; 10779 } 10780 llvm::APSInt VLENVal; 10781 SourceLocation ExprLoc; 10782 const Expr *VLENExpr = Attr->getSimdlen(); 10783 if (VLENExpr) { 10784 VLENVal = VLENExpr->EvaluateKnownConstInt(C); 10785 ExprLoc = VLENExpr->getExprLoc(); 10786 } 10787 OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState(); 10788 if (CGM.getTriple().isX86()) { 10789 emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State); 10790 } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) { 10791 unsigned VLEN = VLENVal.getExtValue(); 10792 StringRef MangledName = Fn->getName(); 10793 if (CGM.getTarget().hasFeature("sve")) 10794 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 10795 MangledName, 's', 128, Fn, ExprLoc); 10796 if (CGM.getTarget().hasFeature("neon")) 10797 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 10798 MangledName, 'n', 128, Fn, ExprLoc); 10799 } 10800 } 10801 FD = FD->getPreviousDecl(); 10802 } 10803 } 10804 10805 namespace { 10806 /// Cleanup action for doacross support. 10807 class DoacrossCleanupTy final : public EHScopeStack::Cleanup { 10808 public: 10809 static const int DoacrossFinArgs = 2; 10810 10811 private: 10812 llvm::FunctionCallee RTLFn; 10813 llvm::Value *Args[DoacrossFinArgs]; 10814 10815 public: 10816 DoacrossCleanupTy(llvm::FunctionCallee RTLFn, 10817 ArrayRef<llvm::Value *> CallArgs) 10818 : RTLFn(RTLFn) { 10819 assert(CallArgs.size() == DoacrossFinArgs); 10820 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 10821 } 10822 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 10823 if (!CGF.HaveInsertPoint()) 10824 return; 10825 CGF.EmitRuntimeCall(RTLFn, Args); 10826 } 10827 }; 10828 } // namespace 10829 10830 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF, 10831 const OMPLoopDirective &D, 10832 ArrayRef<Expr *> NumIterations) { 10833 if (!CGF.HaveInsertPoint()) 10834 return; 10835 10836 ASTContext &C = CGM.getContext(); 10837 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 10838 RecordDecl *RD; 10839 if (KmpDimTy.isNull()) { 10840 // Build struct kmp_dim { // loop bounds info casted to kmp_int64 10841 // kmp_int64 lo; // lower 10842 // kmp_int64 up; // upper 10843 // kmp_int64 st; // stride 10844 // }; 10845 RD = C.buildImplicitRecord("kmp_dim"); 10846 RD->startDefinition(); 10847 addFieldToRecordDecl(C, RD, Int64Ty); 10848 addFieldToRecordDecl(C, RD, Int64Ty); 10849 addFieldToRecordDecl(C, RD, Int64Ty); 10850 RD->completeDefinition(); 10851 KmpDimTy = C.getRecordType(RD); 10852 } else { 10853 RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl()); 10854 } 10855 llvm::APInt Size(/*numBits=*/32, NumIterations.size()); 10856 QualType ArrayTy = 10857 C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0); 10858 10859 Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims"); 10860 CGF.EmitNullInitialization(DimsAddr, ArrayTy); 10861 enum { LowerFD = 0, UpperFD, StrideFD }; 10862 // Fill dims with data. 10863 for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) { 10864 LValue DimsLVal = CGF.MakeAddrLValue( 10865 CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy); 10866 // dims.upper = num_iterations; 10867 LValue UpperLVal = CGF.EmitLValueForField( 10868 DimsLVal, *std::next(RD->field_begin(), UpperFD)); 10869 llvm::Value *NumIterVal = 10870 CGF.EmitScalarConversion(CGF.EmitScalarExpr(NumIterations[I]), 10871 D.getNumIterations()->getType(), Int64Ty, 10872 D.getNumIterations()->getExprLoc()); 10873 CGF.EmitStoreOfScalar(NumIterVal, UpperLVal); 10874 // dims.stride = 1; 10875 LValue StrideLVal = CGF.EmitLValueForField( 10876 DimsLVal, *std::next(RD->field_begin(), StrideFD)); 10877 CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1), 10878 StrideLVal); 10879 } 10880 10881 // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, 10882 // kmp_int32 num_dims, struct kmp_dim * dims); 10883 llvm::Value *Args[] = { 10884 emitUpdateLocation(CGF, D.getBeginLoc()), 10885 getThreadID(CGF, D.getBeginLoc()), 10886 llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()), 10887 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 10888 CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(), 10889 CGM.VoidPtrTy)}; 10890 10891 llvm::FunctionCallee RTLFn = 10892 createRuntimeFunction(OMPRTL__kmpc_doacross_init); 10893 CGF.EmitRuntimeCall(RTLFn, Args); 10894 llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = { 10895 emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())}; 10896 llvm::FunctionCallee FiniRTLFn = 10897 createRuntimeFunction(OMPRTL__kmpc_doacross_fini); 10898 CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 10899 llvm::makeArrayRef(FiniArgs)); 10900 } 10901 10902 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 10903 const OMPDependClause *C) { 10904 QualType Int64Ty = 10905 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 10906 llvm::APInt Size(/*numBits=*/32, C->getNumLoops()); 10907 QualType ArrayTy = CGM.getContext().getConstantArrayType( 10908 Int64Ty, Size, nullptr, ArrayType::Normal, 0); 10909 Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr"); 10910 for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) { 10911 const Expr *CounterVal = C->getLoopData(I); 10912 assert(CounterVal); 10913 llvm::Value *CntVal = CGF.EmitScalarConversion( 10914 CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty, 10915 CounterVal->getExprLoc()); 10916 CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I), 10917 /*Volatile=*/false, Int64Ty); 10918 } 10919 llvm::Value *Args[] = { 10920 emitUpdateLocation(CGF, C->getBeginLoc()), 10921 getThreadID(CGF, C->getBeginLoc()), 10922 CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()}; 10923 llvm::FunctionCallee RTLFn; 10924 if (C->getDependencyKind() == OMPC_DEPEND_source) { 10925 RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_post); 10926 } else { 10927 assert(C->getDependencyKind() == OMPC_DEPEND_sink); 10928 RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_wait); 10929 } 10930 CGF.EmitRuntimeCall(RTLFn, Args); 10931 } 10932 10933 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc, 10934 llvm::FunctionCallee Callee, 10935 ArrayRef<llvm::Value *> Args) const { 10936 assert(Loc.isValid() && "Outlined function call location must be valid."); 10937 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 10938 10939 if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) { 10940 if (Fn->doesNotThrow()) { 10941 CGF.EmitNounwindRuntimeCall(Fn, Args); 10942 return; 10943 } 10944 } 10945 CGF.EmitRuntimeCall(Callee, Args); 10946 } 10947 10948 void CGOpenMPRuntime::emitOutlinedFunctionCall( 10949 CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn, 10950 ArrayRef<llvm::Value *> Args) const { 10951 emitCall(CGF, Loc, OutlinedFn, Args); 10952 } 10953 10954 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) { 10955 if (const auto *FD = dyn_cast<FunctionDecl>(D)) 10956 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD)) 10957 HasEmittedDeclareTargetRegion = true; 10958 } 10959 10960 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF, 10961 const VarDecl *NativeParam, 10962 const VarDecl *TargetParam) const { 10963 return CGF.GetAddrOfLocalVar(NativeParam); 10964 } 10965 10966 namespace { 10967 /// Cleanup action for allocate support. 10968 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup { 10969 public: 10970 static const int CleanupArgs = 3; 10971 10972 private: 10973 llvm::FunctionCallee RTLFn; 10974 llvm::Value *Args[CleanupArgs]; 10975 10976 public: 10977 OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn, 10978 ArrayRef<llvm::Value *> CallArgs) 10979 : RTLFn(RTLFn) { 10980 assert(CallArgs.size() == CleanupArgs && 10981 "Size of arguments does not match."); 10982 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 10983 } 10984 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 10985 if (!CGF.HaveInsertPoint()) 10986 return; 10987 CGF.EmitRuntimeCall(RTLFn, Args); 10988 } 10989 }; 10990 } // namespace 10991 10992 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF, 10993 const VarDecl *VD) { 10994 if (!VD) 10995 return Address::invalid(); 10996 const VarDecl *CVD = VD->getCanonicalDecl(); 10997 if (!CVD->hasAttr<OMPAllocateDeclAttr>()) 10998 return Address::invalid(); 10999 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>(); 11000 // Use the default allocation. 11001 if (AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc && 11002 !AA->getAllocator()) 11003 return Address::invalid(); 11004 llvm::Value *Size; 11005 CharUnits Align = CGM.getContext().getDeclAlign(CVD); 11006 if (CVD->getType()->isVariablyModifiedType()) { 11007 Size = CGF.getTypeSize(CVD->getType()); 11008 // Align the size: ((size + align - 1) / align) * align 11009 Size = CGF.Builder.CreateNUWAdd( 11010 Size, CGM.getSize(Align - CharUnits::fromQuantity(1))); 11011 Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align)); 11012 Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align)); 11013 } else { 11014 CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType()); 11015 Size = CGM.getSize(Sz.alignTo(Align)); 11016 } 11017 llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc()); 11018 assert(AA->getAllocator() && 11019 "Expected allocator expression for non-default allocator."); 11020 llvm::Value *Allocator = CGF.EmitScalarExpr(AA->getAllocator()); 11021 // According to the standard, the original allocator type is a enum (integer). 11022 // Convert to pointer type, if required. 11023 if (Allocator->getType()->isIntegerTy()) 11024 Allocator = CGF.Builder.CreateIntToPtr(Allocator, CGM.VoidPtrTy); 11025 else if (Allocator->getType()->isPointerTy()) 11026 Allocator = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Allocator, 11027 CGM.VoidPtrTy); 11028 llvm::Value *Args[] = {ThreadID, Size, Allocator}; 11029 11030 llvm::Value *Addr = 11031 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_alloc), Args, 11032 getName({CVD->getName(), ".void.addr"})); 11033 llvm::Value *FiniArgs[OMPAllocateCleanupTy::CleanupArgs] = {ThreadID, Addr, 11034 Allocator}; 11035 llvm::FunctionCallee FiniRTLFn = createRuntimeFunction(OMPRTL__kmpc_free); 11036 11037 CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 11038 llvm::makeArrayRef(FiniArgs)); 11039 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11040 Addr, 11041 CGF.ConvertTypeForMem(CGM.getContext().getPointerType(CVD->getType())), 11042 getName({CVD->getName(), ".addr"})); 11043 return Address(Addr, Align); 11044 } 11045 11046 namespace { 11047 using OMPContextSelectorData = 11048 OpenMPCtxSelectorData<ArrayRef<StringRef>, llvm::APSInt>; 11049 using CompleteOMPContextSelectorData = SmallVector<OMPContextSelectorData, 4>; 11050 } // anonymous namespace 11051 11052 /// Checks current context and returns true if it matches the context selector. 11053 template <OpenMPContextSelectorSetKind CtxSet, OpenMPContextSelectorKind Ctx, 11054 typename... Arguments> 11055 static bool checkContext(const OMPContextSelectorData &Data, 11056 Arguments... Params) { 11057 assert(Data.CtxSet != OMP_CTX_SET_unknown && Data.Ctx != OMP_CTX_unknown && 11058 "Unknown context selector or context selector set."); 11059 return false; 11060 } 11061 11062 /// Checks for implementation={vendor(<vendor>)} context selector. 11063 /// \returns true iff <vendor>="llvm", false otherwise. 11064 template <> 11065 bool checkContext<OMP_CTX_SET_implementation, OMP_CTX_vendor>( 11066 const OMPContextSelectorData &Data) { 11067 return llvm::all_of(Data.Names, 11068 [](StringRef S) { return !S.compare_lower("llvm"); }); 11069 } 11070 11071 /// Checks for device={kind(<kind>)} context selector. 11072 /// \returns true if <kind>="host" and compilation is for host. 11073 /// true if <kind>="nohost" and compilation is for device. 11074 /// true if <kind>="cpu" and compilation is for Arm, X86 or PPC CPU. 11075 /// true if <kind>="gpu" and compilation is for NVPTX or AMDGCN. 11076 /// false otherwise. 11077 template <> 11078 bool checkContext<OMP_CTX_SET_device, OMP_CTX_kind, CodeGenModule &>( 11079 const OMPContextSelectorData &Data, CodeGenModule &CGM) { 11080 for (StringRef Name : Data.Names) { 11081 if (!Name.compare_lower("host")) { 11082 if (CGM.getLangOpts().OpenMPIsDevice) 11083 return false; 11084 continue; 11085 } 11086 if (!Name.compare_lower("nohost")) { 11087 if (!CGM.getLangOpts().OpenMPIsDevice) 11088 return false; 11089 continue; 11090 } 11091 switch (CGM.getTriple().getArch()) { 11092 case llvm::Triple::arm: 11093 case llvm::Triple::armeb: 11094 case llvm::Triple::aarch64: 11095 case llvm::Triple::aarch64_be: 11096 case llvm::Triple::aarch64_32: 11097 case llvm::Triple::ppc: 11098 case llvm::Triple::ppc64: 11099 case llvm::Triple::ppc64le: 11100 case llvm::Triple::x86: 11101 case llvm::Triple::x86_64: 11102 if (Name.compare_lower("cpu")) 11103 return false; 11104 break; 11105 case llvm::Triple::amdgcn: 11106 case llvm::Triple::nvptx: 11107 case llvm::Triple::nvptx64: 11108 if (Name.compare_lower("gpu")) 11109 return false; 11110 break; 11111 case llvm::Triple::UnknownArch: 11112 case llvm::Triple::arc: 11113 case llvm::Triple::avr: 11114 case llvm::Triple::bpfel: 11115 case llvm::Triple::bpfeb: 11116 case llvm::Triple::hexagon: 11117 case llvm::Triple::mips: 11118 case llvm::Triple::mipsel: 11119 case llvm::Triple::mips64: 11120 case llvm::Triple::mips64el: 11121 case llvm::Triple::msp430: 11122 case llvm::Triple::r600: 11123 case llvm::Triple::riscv32: 11124 case llvm::Triple::riscv64: 11125 case llvm::Triple::sparc: 11126 case llvm::Triple::sparcv9: 11127 case llvm::Triple::sparcel: 11128 case llvm::Triple::systemz: 11129 case llvm::Triple::tce: 11130 case llvm::Triple::tcele: 11131 case llvm::Triple::thumb: 11132 case llvm::Triple::thumbeb: 11133 case llvm::Triple::xcore: 11134 case llvm::Triple::le32: 11135 case llvm::Triple::le64: 11136 case llvm::Triple::amdil: 11137 case llvm::Triple::amdil64: 11138 case llvm::Triple::hsail: 11139 case llvm::Triple::hsail64: 11140 case llvm::Triple::spir: 11141 case llvm::Triple::spir64: 11142 case llvm::Triple::kalimba: 11143 case llvm::Triple::shave: 11144 case llvm::Triple::lanai: 11145 case llvm::Triple::wasm32: 11146 case llvm::Triple::wasm64: 11147 case llvm::Triple::renderscript32: 11148 case llvm::Triple::renderscript64: 11149 case llvm::Triple::ve: 11150 return false; 11151 } 11152 } 11153 return true; 11154 } 11155 11156 static bool matchesContext(CodeGenModule &CGM, 11157 const CompleteOMPContextSelectorData &ContextData) { 11158 for (const OMPContextSelectorData &Data : ContextData) { 11159 switch (Data.Ctx) { 11160 case OMP_CTX_vendor: 11161 assert(Data.CtxSet == OMP_CTX_SET_implementation && 11162 "Expected implementation context selector set."); 11163 if (!checkContext<OMP_CTX_SET_implementation, OMP_CTX_vendor>(Data)) 11164 return false; 11165 break; 11166 case OMP_CTX_kind: 11167 assert(Data.CtxSet == OMP_CTX_SET_device && 11168 "Expected device context selector set."); 11169 if (!checkContext<OMP_CTX_SET_device, OMP_CTX_kind, CodeGenModule &>(Data, 11170 CGM)) 11171 return false; 11172 break; 11173 case OMP_CTX_unknown: 11174 llvm_unreachable("Unknown context selector kind."); 11175 } 11176 } 11177 return true; 11178 } 11179 11180 static CompleteOMPContextSelectorData 11181 translateAttrToContextSelectorData(ASTContext &C, 11182 const OMPDeclareVariantAttr *A) { 11183 CompleteOMPContextSelectorData Data; 11184 for (unsigned I = 0, E = A->scores_size(); I < E; ++I) { 11185 Data.emplace_back(); 11186 auto CtxSet = static_cast<OpenMPContextSelectorSetKind>( 11187 *std::next(A->ctxSelectorSets_begin(), I)); 11188 auto Ctx = static_cast<OpenMPContextSelectorKind>( 11189 *std::next(A->ctxSelectors_begin(), I)); 11190 Data.back().CtxSet = CtxSet; 11191 Data.back().Ctx = Ctx; 11192 const Expr *Score = *std::next(A->scores_begin(), I); 11193 Data.back().Score = Score->EvaluateKnownConstInt(C); 11194 switch (Ctx) { 11195 case OMP_CTX_vendor: 11196 assert(CtxSet == OMP_CTX_SET_implementation && 11197 "Expected implementation context selector set."); 11198 Data.back().Names = 11199 llvm::makeArrayRef(A->implVendors_begin(), A->implVendors_end()); 11200 break; 11201 case OMP_CTX_kind: 11202 assert(CtxSet == OMP_CTX_SET_device && 11203 "Expected device context selector set."); 11204 Data.back().Names = 11205 llvm::makeArrayRef(A->deviceKinds_begin(), A->deviceKinds_end()); 11206 break; 11207 case OMP_CTX_unknown: 11208 llvm_unreachable("Unknown context selector kind."); 11209 } 11210 } 11211 return Data; 11212 } 11213 11214 static bool isStrictSubset(const CompleteOMPContextSelectorData &LHS, 11215 const CompleteOMPContextSelectorData &RHS) { 11216 llvm::SmallDenseMap<std::pair<int, int>, llvm::StringSet<>, 4> RHSData; 11217 for (const OMPContextSelectorData &D : RHS) { 11218 auto &Pair = RHSData.FindAndConstruct(std::make_pair(D.CtxSet, D.Ctx)); 11219 Pair.getSecond().insert(D.Names.begin(), D.Names.end()); 11220 } 11221 bool AllSetsAreEqual = true; 11222 for (const OMPContextSelectorData &D : LHS) { 11223 auto It = RHSData.find(std::make_pair(D.CtxSet, D.Ctx)); 11224 if (It == RHSData.end()) 11225 return false; 11226 if (D.Names.size() > It->getSecond().size()) 11227 return false; 11228 if (llvm::set_union(It->getSecond(), D.Names)) 11229 return false; 11230 AllSetsAreEqual = 11231 AllSetsAreEqual && (D.Names.size() == It->getSecond().size()); 11232 } 11233 11234 return LHS.size() != RHS.size() || !AllSetsAreEqual; 11235 } 11236 11237 static bool greaterCtxScore(const CompleteOMPContextSelectorData &LHS, 11238 const CompleteOMPContextSelectorData &RHS) { 11239 // Score is calculated as sum of all scores + 1. 11240 llvm::APSInt LHSScore(llvm::APInt(64, 1), /*isUnsigned=*/false); 11241 bool RHSIsSubsetOfLHS = isStrictSubset(RHS, LHS); 11242 if (RHSIsSubsetOfLHS) { 11243 LHSScore = llvm::APSInt::get(0); 11244 } else { 11245 for (const OMPContextSelectorData &Data : LHS) { 11246 if (Data.Score.getBitWidth() > LHSScore.getBitWidth()) { 11247 LHSScore = LHSScore.extend(Data.Score.getBitWidth()) + Data.Score; 11248 } else if (Data.Score.getBitWidth() < LHSScore.getBitWidth()) { 11249 LHSScore += Data.Score.extend(LHSScore.getBitWidth()); 11250 } else { 11251 LHSScore += Data.Score; 11252 } 11253 } 11254 } 11255 llvm::APSInt RHSScore(llvm::APInt(64, 1), /*isUnsigned=*/false); 11256 if (!RHSIsSubsetOfLHS && isStrictSubset(LHS, RHS)) { 11257 RHSScore = llvm::APSInt::get(0); 11258 } else { 11259 for (const OMPContextSelectorData &Data : RHS) { 11260 if (Data.Score.getBitWidth() > RHSScore.getBitWidth()) { 11261 RHSScore = RHSScore.extend(Data.Score.getBitWidth()) + Data.Score; 11262 } else if (Data.Score.getBitWidth() < RHSScore.getBitWidth()) { 11263 RHSScore += Data.Score.extend(RHSScore.getBitWidth()); 11264 } else { 11265 RHSScore += Data.Score; 11266 } 11267 } 11268 } 11269 return llvm::APSInt::compareValues(LHSScore, RHSScore) >= 0; 11270 } 11271 11272 /// Finds the variant function that matches current context with its context 11273 /// selector. 11274 static const FunctionDecl *getDeclareVariantFunction(CodeGenModule &CGM, 11275 const FunctionDecl *FD) { 11276 if (!FD->hasAttrs() || !FD->hasAttr<OMPDeclareVariantAttr>()) 11277 return FD; 11278 // Iterate through all DeclareVariant attributes and check context selectors. 11279 const OMPDeclareVariantAttr *TopMostAttr = nullptr; 11280 CompleteOMPContextSelectorData TopMostData; 11281 for (const auto *A : FD->specific_attrs<OMPDeclareVariantAttr>()) { 11282 CompleteOMPContextSelectorData Data = 11283 translateAttrToContextSelectorData(CGM.getContext(), A); 11284 if (!matchesContext(CGM, Data)) 11285 continue; 11286 // If the attribute matches the context, find the attribute with the highest 11287 // score. 11288 if (!TopMostAttr || !greaterCtxScore(TopMostData, Data)) { 11289 TopMostAttr = A; 11290 TopMostData.swap(Data); 11291 } 11292 } 11293 if (!TopMostAttr) 11294 return FD; 11295 return cast<FunctionDecl>( 11296 cast<DeclRefExpr>(TopMostAttr->getVariantFuncRef()->IgnoreParenImpCasts()) 11297 ->getDecl()); 11298 } 11299 11300 bool CGOpenMPRuntime::emitDeclareVariant(GlobalDecl GD, bool IsForDefinition) { 11301 const auto *D = cast<FunctionDecl>(GD.getDecl()); 11302 // If the original function is defined already, use its definition. 11303 StringRef MangledName = CGM.getMangledName(GD); 11304 llvm::GlobalValue *Orig = CGM.GetGlobalValue(MangledName); 11305 if (Orig && !Orig->isDeclaration()) 11306 return false; 11307 const FunctionDecl *NewFD = getDeclareVariantFunction(CGM, D); 11308 // Emit original function if it does not have declare variant attribute or the 11309 // context does not match. 11310 if (NewFD == D) 11311 return false; 11312 GlobalDecl NewGD = GD.getWithDecl(NewFD); 11313 if (tryEmitDeclareVariant(NewGD, GD, Orig, IsForDefinition)) { 11314 DeferredVariantFunction.erase(D); 11315 return true; 11316 } 11317 DeferredVariantFunction.insert(std::make_pair(D, std::make_pair(NewGD, GD))); 11318 return true; 11319 } 11320 11321 CGOpenMPRuntime::NontemporalDeclsRAII::NontemporalDeclsRAII( 11322 CodeGenModule &CGM, const OMPLoopDirective &S) 11323 : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) { 11324 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11325 if (!NeedToPush) 11326 return; 11327 NontemporalDeclsSet &DS = 11328 CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back(); 11329 for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) { 11330 for (const Stmt *Ref : C->private_refs()) { 11331 const auto *SimpleRefExpr = cast<Expr>(Ref)->IgnoreParenImpCasts(); 11332 const ValueDecl *VD; 11333 if (const auto *DRE = dyn_cast<DeclRefExpr>(SimpleRefExpr)) { 11334 VD = DRE->getDecl(); 11335 } else { 11336 const auto *ME = cast<MemberExpr>(SimpleRefExpr); 11337 assert((ME->isImplicitCXXThis() || 11338 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) && 11339 "Expected member of current class."); 11340 VD = ME->getMemberDecl(); 11341 } 11342 DS.insert(VD); 11343 } 11344 } 11345 } 11346 11347 CGOpenMPRuntime::NontemporalDeclsRAII::~NontemporalDeclsRAII() { 11348 if (!NeedToPush) 11349 return; 11350 CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back(); 11351 } 11352 11353 bool CGOpenMPRuntime::isNontemporalDecl(const ValueDecl *VD) const { 11354 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11355 11356 return llvm::any_of( 11357 CGM.getOpenMPRuntime().NontemporalDeclsStack, 11358 [VD](const NontemporalDeclsSet &Set) { return Set.count(VD) > 0; }); 11359 } 11360 11361 void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis( 11362 const OMPExecutableDirective &S, 11363 llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled) 11364 const { 11365 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs; 11366 // Vars in target/task regions must be excluded completely. 11367 if (isOpenMPTargetExecutionDirective(S.getDirectiveKind()) || 11368 isOpenMPTaskingDirective(S.getDirectiveKind())) { 11369 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions; 11370 getOpenMPCaptureRegions(CaptureRegions, S.getDirectiveKind()); 11371 const CapturedStmt *CS = S.getCapturedStmt(CaptureRegions.front()); 11372 for (const CapturedStmt::Capture &Cap : CS->captures()) { 11373 if (Cap.capturesVariable() || Cap.capturesVariableByCopy()) 11374 NeedToCheckForLPCs.insert(Cap.getCapturedVar()); 11375 } 11376 } 11377 // Exclude vars in private clauses. 11378 for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) { 11379 for (const Expr *Ref : C->varlists()) { 11380 if (!Ref->getType()->isScalarType()) 11381 continue; 11382 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11383 if (!DRE) 11384 continue; 11385 NeedToCheckForLPCs.insert(DRE->getDecl()); 11386 } 11387 } 11388 for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) { 11389 for (const Expr *Ref : C->varlists()) { 11390 if (!Ref->getType()->isScalarType()) 11391 continue; 11392 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11393 if (!DRE) 11394 continue; 11395 NeedToCheckForLPCs.insert(DRE->getDecl()); 11396 } 11397 } 11398 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) { 11399 for (const Expr *Ref : C->varlists()) { 11400 if (!Ref->getType()->isScalarType()) 11401 continue; 11402 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11403 if (!DRE) 11404 continue; 11405 NeedToCheckForLPCs.insert(DRE->getDecl()); 11406 } 11407 } 11408 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) { 11409 for (const Expr *Ref : C->varlists()) { 11410 if (!Ref->getType()->isScalarType()) 11411 continue; 11412 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11413 if (!DRE) 11414 continue; 11415 NeedToCheckForLPCs.insert(DRE->getDecl()); 11416 } 11417 } 11418 for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) { 11419 for (const Expr *Ref : C->varlists()) { 11420 if (!Ref->getType()->isScalarType()) 11421 continue; 11422 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11423 if (!DRE) 11424 continue; 11425 NeedToCheckForLPCs.insert(DRE->getDecl()); 11426 } 11427 } 11428 for (const Decl *VD : NeedToCheckForLPCs) { 11429 for (const LastprivateConditionalData &Data : 11430 llvm::reverse(CGM.getOpenMPRuntime().LastprivateConditionalStack)) { 11431 if (Data.DeclToUniqueName.count(VD) > 0) { 11432 if (!Data.Disabled) 11433 NeedToAddForLPCsAsDisabled.insert(VD); 11434 break; 11435 } 11436 } 11437 } 11438 } 11439 11440 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII( 11441 CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal) 11442 : CGM(CGF.CGM), 11443 Action((CGM.getLangOpts().OpenMP >= 50 && 11444 llvm::any_of(S.getClausesOfKind<OMPLastprivateClause>(), 11445 [](const OMPLastprivateClause *C) { 11446 return C->getKind() == 11447 OMPC_LASTPRIVATE_conditional; 11448 })) 11449 ? ActionToDo::PushAsLastprivateConditional 11450 : ActionToDo::DoNotPush) { 11451 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11452 if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush) 11453 return; 11454 assert(Action == ActionToDo::PushAsLastprivateConditional && 11455 "Expected a push action."); 11456 LastprivateConditionalData &Data = 11457 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back(); 11458 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) { 11459 if (C->getKind() != OMPC_LASTPRIVATE_conditional) 11460 continue; 11461 11462 for (const Expr *Ref : C->varlists()) { 11463 Data.DeclToUniqueName.insert(std::make_pair( 11464 cast<DeclRefExpr>(Ref->IgnoreParenImpCasts())->getDecl(), 11465 SmallString<16>(generateUniqueName(CGM, "pl_cond", Ref)))); 11466 } 11467 } 11468 Data.IVLVal = IVLVal; 11469 Data.Fn = CGF.CurFn; 11470 } 11471 11472 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII( 11473 CodeGenFunction &CGF, const OMPExecutableDirective &S) 11474 : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) { 11475 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11476 if (CGM.getLangOpts().OpenMP < 50) 11477 return; 11478 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled; 11479 tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled); 11480 if (!NeedToAddForLPCsAsDisabled.empty()) { 11481 Action = ActionToDo::DisableLastprivateConditional; 11482 LastprivateConditionalData &Data = 11483 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back(); 11484 for (const Decl *VD : NeedToAddForLPCsAsDisabled) 11485 Data.DeclToUniqueName.insert(std::make_pair(VD, SmallString<16>())); 11486 Data.Fn = CGF.CurFn; 11487 Data.Disabled = true; 11488 } 11489 } 11490 11491 CGOpenMPRuntime::LastprivateConditionalRAII 11492 CGOpenMPRuntime::LastprivateConditionalRAII::disable( 11493 CodeGenFunction &CGF, const OMPExecutableDirective &S) { 11494 return LastprivateConditionalRAII(CGF, S); 11495 } 11496 11497 CGOpenMPRuntime::LastprivateConditionalRAII::~LastprivateConditionalRAII() { 11498 if (CGM.getLangOpts().OpenMP < 50) 11499 return; 11500 if (Action == ActionToDo::DisableLastprivateConditional) { 11501 assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled && 11502 "Expected list of disabled private vars."); 11503 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back(); 11504 } 11505 if (Action == ActionToDo::PushAsLastprivateConditional) { 11506 assert( 11507 !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled && 11508 "Expected list of lastprivate conditional vars."); 11509 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back(); 11510 } 11511 } 11512 11513 Address CGOpenMPRuntime::emitLastprivateConditionalInit(CodeGenFunction &CGF, 11514 const VarDecl *VD) { 11515 ASTContext &C = CGM.getContext(); 11516 auto I = LastprivateConditionalToTypes.find(CGF.CurFn); 11517 if (I == LastprivateConditionalToTypes.end()) 11518 I = LastprivateConditionalToTypes.try_emplace(CGF.CurFn).first; 11519 QualType NewType; 11520 const FieldDecl *VDField; 11521 const FieldDecl *FiredField; 11522 LValue BaseLVal; 11523 auto VI = I->getSecond().find(VD); 11524 if (VI == I->getSecond().end()) { 11525 RecordDecl *RD = C.buildImplicitRecord("lasprivate.conditional"); 11526 RD->startDefinition(); 11527 VDField = addFieldToRecordDecl(C, RD, VD->getType().getNonReferenceType()); 11528 FiredField = addFieldToRecordDecl(C, RD, C.CharTy); 11529 RD->completeDefinition(); 11530 NewType = C.getRecordType(RD); 11531 Address Addr = CGF.CreateMemTemp(NewType, C.getDeclAlign(VD), VD->getName()); 11532 BaseLVal = CGF.MakeAddrLValue(Addr, NewType, AlignmentSource::Decl); 11533 I->getSecond().try_emplace(VD, NewType, VDField, FiredField, BaseLVal); 11534 } else { 11535 NewType = std::get<0>(VI->getSecond()); 11536 VDField = std::get<1>(VI->getSecond()); 11537 FiredField = std::get<2>(VI->getSecond()); 11538 BaseLVal = std::get<3>(VI->getSecond()); 11539 } 11540 LValue FiredLVal = 11541 CGF.EmitLValueForField(BaseLVal, FiredField); 11542 CGF.EmitStoreOfScalar( 11543 llvm::ConstantInt::getNullValue(CGF.ConvertTypeForMem(C.CharTy)), 11544 FiredLVal); 11545 return CGF.EmitLValueForField(BaseLVal, VDField).getAddress(CGF); 11546 } 11547 11548 namespace { 11549 /// Checks if the lastprivate conditional variable is referenced in LHS. 11550 class LastprivateConditionalRefChecker final 11551 : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> { 11552 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM; 11553 const Expr *FoundE = nullptr; 11554 const Decl *FoundD = nullptr; 11555 StringRef UniqueDeclName; 11556 LValue IVLVal; 11557 llvm::Function *FoundFn = nullptr; 11558 SourceLocation Loc; 11559 11560 public: 11561 bool VisitDeclRefExpr(const DeclRefExpr *E) { 11562 for (const CGOpenMPRuntime::LastprivateConditionalData &D : 11563 llvm::reverse(LPM)) { 11564 auto It = D.DeclToUniqueName.find(E->getDecl()); 11565 if (It == D.DeclToUniqueName.end()) 11566 continue; 11567 if (D.Disabled) 11568 return false; 11569 FoundE = E; 11570 FoundD = E->getDecl()->getCanonicalDecl(); 11571 UniqueDeclName = It->second; 11572 IVLVal = D.IVLVal; 11573 FoundFn = D.Fn; 11574 break; 11575 } 11576 return FoundE == E; 11577 } 11578 bool VisitMemberExpr(const MemberExpr *E) { 11579 if (!CodeGenFunction::IsWrappedCXXThis(E->getBase())) 11580 return false; 11581 for (const CGOpenMPRuntime::LastprivateConditionalData &D : 11582 llvm::reverse(LPM)) { 11583 auto It = D.DeclToUniqueName.find(E->getMemberDecl()); 11584 if (It == D.DeclToUniqueName.end()) 11585 continue; 11586 if (D.Disabled) 11587 return false; 11588 FoundE = E; 11589 FoundD = E->getMemberDecl()->getCanonicalDecl(); 11590 UniqueDeclName = It->second; 11591 IVLVal = D.IVLVal; 11592 FoundFn = D.Fn; 11593 break; 11594 } 11595 return FoundE == E; 11596 } 11597 bool VisitStmt(const Stmt *S) { 11598 for (const Stmt *Child : S->children()) { 11599 if (!Child) 11600 continue; 11601 if (const auto *E = dyn_cast<Expr>(Child)) 11602 if (!E->isGLValue()) 11603 continue; 11604 if (Visit(Child)) 11605 return true; 11606 } 11607 return false; 11608 } 11609 explicit LastprivateConditionalRefChecker( 11610 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM) 11611 : LPM(LPM) {} 11612 std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *> 11613 getFoundData() const { 11614 return std::make_tuple(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn); 11615 } 11616 }; 11617 } // namespace 11618 11619 void CGOpenMPRuntime::emitLastprivateConditionalUpdate(CodeGenFunction &CGF, 11620 LValue IVLVal, 11621 StringRef UniqueDeclName, 11622 LValue LVal, 11623 SourceLocation Loc) { 11624 // Last updated loop counter for the lastprivate conditional var. 11625 // int<xx> last_iv = 0; 11626 llvm::Type *LLIVTy = CGF.ConvertTypeForMem(IVLVal.getType()); 11627 llvm::Constant *LastIV = 11628 getOrCreateInternalVariable(LLIVTy, getName({UniqueDeclName, "iv"})); 11629 cast<llvm::GlobalVariable>(LastIV)->setAlignment( 11630 IVLVal.getAlignment().getAsAlign()); 11631 LValue LastIVLVal = CGF.MakeNaturalAlignAddrLValue(LastIV, IVLVal.getType()); 11632 11633 // Last value of the lastprivate conditional. 11634 // decltype(priv_a) last_a; 11635 llvm::Constant *Last = getOrCreateInternalVariable( 11636 CGF.ConvertTypeForMem(LVal.getType()), UniqueDeclName); 11637 cast<llvm::GlobalVariable>(Last)->setAlignment( 11638 LVal.getAlignment().getAsAlign()); 11639 LValue LastLVal = 11640 CGF.MakeAddrLValue(Last, LVal.getType(), LVal.getAlignment()); 11641 11642 // Global loop counter. Required to handle inner parallel-for regions. 11643 // iv 11644 llvm::Value *IVVal = CGF.EmitLoadOfScalar(IVLVal, Loc); 11645 11646 // #pragma omp critical(a) 11647 // if (last_iv <= iv) { 11648 // last_iv = iv; 11649 // last_a = priv_a; 11650 // } 11651 auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal, 11652 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 11653 Action.Enter(CGF); 11654 llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(LastIVLVal, Loc); 11655 // (last_iv <= iv) ? Check if the variable is updated and store new 11656 // value in global var. 11657 llvm::Value *CmpRes; 11658 if (IVLVal.getType()->isSignedIntegerType()) { 11659 CmpRes = CGF.Builder.CreateICmpSLE(LastIVVal, IVVal); 11660 } else { 11661 assert(IVLVal.getType()->isUnsignedIntegerType() && 11662 "Loop iteration variable must be integer."); 11663 CmpRes = CGF.Builder.CreateICmpULE(LastIVVal, IVVal); 11664 } 11665 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lp_cond_then"); 11666 llvm::BasicBlock *ExitBB = CGF.createBasicBlock("lp_cond_exit"); 11667 CGF.Builder.CreateCondBr(CmpRes, ThenBB, ExitBB); 11668 // { 11669 CGF.EmitBlock(ThenBB); 11670 11671 // last_iv = iv; 11672 CGF.EmitStoreOfScalar(IVVal, LastIVLVal); 11673 11674 // last_a = priv_a; 11675 switch (CGF.getEvaluationKind(LVal.getType())) { 11676 case TEK_Scalar: { 11677 llvm::Value *PrivVal = CGF.EmitLoadOfScalar(LVal, Loc); 11678 CGF.EmitStoreOfScalar(PrivVal, LastLVal); 11679 break; 11680 } 11681 case TEK_Complex: { 11682 CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(LVal, Loc); 11683 CGF.EmitStoreOfComplex(PrivVal, LastLVal, /*isInit=*/false); 11684 break; 11685 } 11686 case TEK_Aggregate: 11687 llvm_unreachable( 11688 "Aggregates are not supported in lastprivate conditional."); 11689 } 11690 // } 11691 CGF.EmitBranch(ExitBB); 11692 // There is no need to emit line number for unconditional branch. 11693 (void)ApplyDebugLocation::CreateEmpty(CGF); 11694 CGF.EmitBlock(ExitBB, /*IsFinished=*/true); 11695 }; 11696 11697 if (CGM.getLangOpts().OpenMPSimd) { 11698 // Do not emit as a critical region as no parallel region could be emitted. 11699 RegionCodeGenTy ThenRCG(CodeGen); 11700 ThenRCG(CGF); 11701 } else { 11702 emitCriticalRegion(CGF, UniqueDeclName, CodeGen, Loc); 11703 } 11704 } 11705 11706 void CGOpenMPRuntime::checkAndEmitLastprivateConditional(CodeGenFunction &CGF, 11707 const Expr *LHS) { 11708 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty()) 11709 return; 11710 LastprivateConditionalRefChecker Checker(LastprivateConditionalStack); 11711 if (!Checker.Visit(LHS)) 11712 return; 11713 const Expr *FoundE; 11714 const Decl *FoundD; 11715 StringRef UniqueDeclName; 11716 LValue IVLVal; 11717 llvm::Function *FoundFn; 11718 std::tie(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn) = 11719 Checker.getFoundData(); 11720 if (FoundFn != CGF.CurFn) { 11721 // Special codegen for inner parallel regions. 11722 // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1; 11723 auto It = LastprivateConditionalToTypes[FoundFn].find(FoundD); 11724 assert(It != LastprivateConditionalToTypes[FoundFn].end() && 11725 "Lastprivate conditional is not found in outer region."); 11726 QualType StructTy = std::get<0>(It->getSecond()); 11727 const FieldDecl* FiredDecl = std::get<2>(It->getSecond()); 11728 LValue PrivLVal = CGF.EmitLValue(FoundE); 11729 Address StructAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11730 PrivLVal.getAddress(CGF), 11731 CGF.ConvertTypeForMem(CGF.getContext().getPointerType(StructTy))); 11732 LValue BaseLVal = 11733 CGF.MakeAddrLValue(StructAddr, StructTy, AlignmentSource::Decl); 11734 LValue FiredLVal = CGF.EmitLValueForField(BaseLVal, FiredDecl); 11735 CGF.EmitAtomicStore(RValue::get(llvm::ConstantInt::get( 11736 CGF.ConvertTypeForMem(FiredDecl->getType()), 1)), 11737 FiredLVal, llvm::AtomicOrdering::Unordered, 11738 /*IsVolatile=*/true, /*isInit=*/false); 11739 return; 11740 } 11741 11742 // Private address of the lastprivate conditional in the current context. 11743 // priv_a 11744 LValue LVal = CGF.EmitLValue(FoundE); 11745 emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal, 11746 FoundE->getExprLoc()); 11747 } 11748 11749 void CGOpenMPRuntime::checkAndEmitSharedLastprivateConditional( 11750 CodeGenFunction &CGF, const OMPExecutableDirective &D, 11751 const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) { 11752 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty()) 11753 return; 11754 auto Range = llvm::reverse(LastprivateConditionalStack); 11755 auto It = llvm::find_if( 11756 Range, [](const LastprivateConditionalData &D) { return !D.Disabled; }); 11757 if (It == Range.end() || It->Fn != CGF.CurFn) 11758 return; 11759 auto LPCI = LastprivateConditionalToTypes.find(It->Fn); 11760 assert(LPCI != LastprivateConditionalToTypes.end() && 11761 "Lastprivates must be registered already."); 11762 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions; 11763 getOpenMPCaptureRegions(CaptureRegions, D.getDirectiveKind()); 11764 const CapturedStmt *CS = D.getCapturedStmt(CaptureRegions.back()); 11765 for (const auto &Pair : It->DeclToUniqueName) { 11766 const auto *VD = cast<VarDecl>(Pair.first->getCanonicalDecl()); 11767 if (!CS->capturesVariable(VD) || IgnoredDecls.count(VD) > 0) 11768 continue; 11769 auto I = LPCI->getSecond().find(Pair.first); 11770 assert(I != LPCI->getSecond().end() && 11771 "Lastprivate must be rehistered already."); 11772 // bool Cmp = priv_a.Fired != 0; 11773 LValue BaseLVal = std::get<3>(I->getSecond()); 11774 LValue FiredLVal = 11775 CGF.EmitLValueForField(BaseLVal, std::get<2>(I->getSecond())); 11776 llvm::Value *Res = CGF.EmitLoadOfScalar(FiredLVal, D.getBeginLoc()); 11777 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Res); 11778 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lpc.then"); 11779 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("lpc.done"); 11780 // if (Cmp) { 11781 CGF.Builder.CreateCondBr(Cmp, ThenBB, DoneBB); 11782 CGF.EmitBlock(ThenBB); 11783 Address Addr = CGF.GetAddrOfLocalVar(VD); 11784 LValue LVal; 11785 if (VD->getType()->isReferenceType()) 11786 LVal = CGF.EmitLoadOfReferenceLValue(Addr, VD->getType(), 11787 AlignmentSource::Decl); 11788 else 11789 LVal = CGF.MakeAddrLValue(Addr, VD->getType().getNonReferenceType(), 11790 AlignmentSource::Decl); 11791 emitLastprivateConditionalUpdate(CGF, It->IVLVal, Pair.second, LVal, 11792 D.getBeginLoc()); 11793 auto AL = ApplyDebugLocation::CreateArtificial(CGF); 11794 CGF.EmitBlock(DoneBB, /*IsFinal=*/true); 11795 // } 11796 } 11797 } 11798 11799 void CGOpenMPRuntime::emitLastprivateConditionalFinalUpdate( 11800 CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD, 11801 SourceLocation Loc) { 11802 if (CGF.getLangOpts().OpenMP < 50) 11803 return; 11804 auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(VD); 11805 assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() && 11806 "Unknown lastprivate conditional variable."); 11807 StringRef UniqueName = It->second; 11808 llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(UniqueName); 11809 // The variable was not updated in the region - exit. 11810 if (!GV) 11811 return; 11812 LValue LPLVal = CGF.MakeAddrLValue( 11813 GV, PrivLVal.getType().getNonReferenceType(), PrivLVal.getAlignment()); 11814 llvm::Value *Res = CGF.EmitLoadOfScalar(LPLVal, Loc); 11815 CGF.EmitStoreOfScalar(Res, PrivLVal); 11816 } 11817 11818 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction( 11819 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11820 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 11821 llvm_unreachable("Not supported in SIMD-only mode"); 11822 } 11823 11824 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction( 11825 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11826 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 11827 llvm_unreachable("Not supported in SIMD-only mode"); 11828 } 11829 11830 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction( 11831 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11832 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 11833 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 11834 bool Tied, unsigned &NumberOfParts) { 11835 llvm_unreachable("Not supported in SIMD-only mode"); 11836 } 11837 11838 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF, 11839 SourceLocation Loc, 11840 llvm::Function *OutlinedFn, 11841 ArrayRef<llvm::Value *> CapturedVars, 11842 const Expr *IfCond) { 11843 llvm_unreachable("Not supported in SIMD-only mode"); 11844 } 11845 11846 void CGOpenMPSIMDRuntime::emitCriticalRegion( 11847 CodeGenFunction &CGF, StringRef CriticalName, 11848 const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc, 11849 const Expr *Hint) { 11850 llvm_unreachable("Not supported in SIMD-only mode"); 11851 } 11852 11853 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF, 11854 const RegionCodeGenTy &MasterOpGen, 11855 SourceLocation Loc) { 11856 llvm_unreachable("Not supported in SIMD-only mode"); 11857 } 11858 11859 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 11860 SourceLocation Loc) { 11861 llvm_unreachable("Not supported in SIMD-only mode"); 11862 } 11863 11864 void CGOpenMPSIMDRuntime::emitTaskgroupRegion( 11865 CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen, 11866 SourceLocation Loc) { 11867 llvm_unreachable("Not supported in SIMD-only mode"); 11868 } 11869 11870 void CGOpenMPSIMDRuntime::emitSingleRegion( 11871 CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen, 11872 SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars, 11873 ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs, 11874 ArrayRef<const Expr *> AssignmentOps) { 11875 llvm_unreachable("Not supported in SIMD-only mode"); 11876 } 11877 11878 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF, 11879 const RegionCodeGenTy &OrderedOpGen, 11880 SourceLocation Loc, 11881 bool IsThreads) { 11882 llvm_unreachable("Not supported in SIMD-only mode"); 11883 } 11884 11885 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF, 11886 SourceLocation Loc, 11887 OpenMPDirectiveKind Kind, 11888 bool EmitChecks, 11889 bool ForceSimpleCall) { 11890 llvm_unreachable("Not supported in SIMD-only mode"); 11891 } 11892 11893 void CGOpenMPSIMDRuntime::emitForDispatchInit( 11894 CodeGenFunction &CGF, SourceLocation Loc, 11895 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 11896 bool Ordered, const DispatchRTInput &DispatchValues) { 11897 llvm_unreachable("Not supported in SIMD-only mode"); 11898 } 11899 11900 void CGOpenMPSIMDRuntime::emitForStaticInit( 11901 CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind, 11902 const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) { 11903 llvm_unreachable("Not supported in SIMD-only mode"); 11904 } 11905 11906 void CGOpenMPSIMDRuntime::emitDistributeStaticInit( 11907 CodeGenFunction &CGF, SourceLocation Loc, 11908 OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) { 11909 llvm_unreachable("Not supported in SIMD-only mode"); 11910 } 11911 11912 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 11913 SourceLocation Loc, 11914 unsigned IVSize, 11915 bool IVSigned) { 11916 llvm_unreachable("Not supported in SIMD-only mode"); 11917 } 11918 11919 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF, 11920 SourceLocation Loc, 11921 OpenMPDirectiveKind DKind) { 11922 llvm_unreachable("Not supported in SIMD-only mode"); 11923 } 11924 11925 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF, 11926 SourceLocation Loc, 11927 unsigned IVSize, bool IVSigned, 11928 Address IL, Address LB, 11929 Address UB, Address ST) { 11930 llvm_unreachable("Not supported in SIMD-only mode"); 11931 } 11932 11933 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 11934 llvm::Value *NumThreads, 11935 SourceLocation Loc) { 11936 llvm_unreachable("Not supported in SIMD-only mode"); 11937 } 11938 11939 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF, 11940 ProcBindKind ProcBind, 11941 SourceLocation Loc) { 11942 llvm_unreachable("Not supported in SIMD-only mode"); 11943 } 11944 11945 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 11946 const VarDecl *VD, 11947 Address VDAddr, 11948 SourceLocation Loc) { 11949 llvm_unreachable("Not supported in SIMD-only mode"); 11950 } 11951 11952 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition( 11953 const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit, 11954 CodeGenFunction *CGF) { 11955 llvm_unreachable("Not supported in SIMD-only mode"); 11956 } 11957 11958 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate( 11959 CodeGenFunction &CGF, QualType VarType, StringRef Name) { 11960 llvm_unreachable("Not supported in SIMD-only mode"); 11961 } 11962 11963 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF, 11964 ArrayRef<const Expr *> Vars, 11965 SourceLocation Loc, 11966 llvm::AtomicOrdering AO) { 11967 llvm_unreachable("Not supported in SIMD-only mode"); 11968 } 11969 11970 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 11971 const OMPExecutableDirective &D, 11972 llvm::Function *TaskFunction, 11973 QualType SharedsTy, Address Shareds, 11974 const Expr *IfCond, 11975 const OMPTaskDataTy &Data) { 11976 llvm_unreachable("Not supported in SIMD-only mode"); 11977 } 11978 11979 void CGOpenMPSIMDRuntime::emitTaskLoopCall( 11980 CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D, 11981 llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, 11982 const Expr *IfCond, const OMPTaskDataTy &Data) { 11983 llvm_unreachable("Not supported in SIMD-only mode"); 11984 } 11985 11986 void CGOpenMPSIMDRuntime::emitReduction( 11987 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates, 11988 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 11989 ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) { 11990 assert(Options.SimpleReduction && "Only simple reduction is expected."); 11991 CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs, 11992 ReductionOps, Options); 11993 } 11994 11995 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit( 11996 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 11997 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 11998 llvm_unreachable("Not supported in SIMD-only mode"); 11999 } 12000 12001 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 12002 SourceLocation Loc, 12003 ReductionCodeGen &RCG, 12004 unsigned N) { 12005 llvm_unreachable("Not supported in SIMD-only mode"); 12006 } 12007 12008 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF, 12009 SourceLocation Loc, 12010 llvm::Value *ReductionsPtr, 12011 LValue SharedLVal) { 12012 llvm_unreachable("Not supported in SIMD-only mode"); 12013 } 12014 12015 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 12016 SourceLocation Loc) { 12017 llvm_unreachable("Not supported in SIMD-only mode"); 12018 } 12019 12020 void CGOpenMPSIMDRuntime::emitCancellationPointCall( 12021 CodeGenFunction &CGF, SourceLocation Loc, 12022 OpenMPDirectiveKind CancelRegion) { 12023 llvm_unreachable("Not supported in SIMD-only mode"); 12024 } 12025 12026 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF, 12027 SourceLocation Loc, const Expr *IfCond, 12028 OpenMPDirectiveKind CancelRegion) { 12029 llvm_unreachable("Not supported in SIMD-only mode"); 12030 } 12031 12032 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction( 12033 const OMPExecutableDirective &D, StringRef ParentName, 12034 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 12035 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 12036 llvm_unreachable("Not supported in SIMD-only mode"); 12037 } 12038 12039 void CGOpenMPSIMDRuntime::emitTargetCall( 12040 CodeGenFunction &CGF, const OMPExecutableDirective &D, 12041 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 12042 const Expr *Device, 12043 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 12044 const OMPLoopDirective &D)> 12045 SizeEmitter) { 12046 llvm_unreachable("Not supported in SIMD-only mode"); 12047 } 12048 12049 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) { 12050 llvm_unreachable("Not supported in SIMD-only mode"); 12051 } 12052 12053 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 12054 llvm_unreachable("Not supported in SIMD-only mode"); 12055 } 12056 12057 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) { 12058 return false; 12059 } 12060 12061 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF, 12062 const OMPExecutableDirective &D, 12063 SourceLocation Loc, 12064 llvm::Function *OutlinedFn, 12065 ArrayRef<llvm::Value *> CapturedVars) { 12066 llvm_unreachable("Not supported in SIMD-only mode"); 12067 } 12068 12069 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 12070 const Expr *NumTeams, 12071 const Expr *ThreadLimit, 12072 SourceLocation Loc) { 12073 llvm_unreachable("Not supported in SIMD-only mode"); 12074 } 12075 12076 void CGOpenMPSIMDRuntime::emitTargetDataCalls( 12077 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 12078 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 12079 llvm_unreachable("Not supported in SIMD-only mode"); 12080 } 12081 12082 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall( 12083 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 12084 const Expr *Device) { 12085 llvm_unreachable("Not supported in SIMD-only mode"); 12086 } 12087 12088 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF, 12089 const OMPLoopDirective &D, 12090 ArrayRef<Expr *> NumIterations) { 12091 llvm_unreachable("Not supported in SIMD-only mode"); 12092 } 12093 12094 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 12095 const OMPDependClause *C) { 12096 llvm_unreachable("Not supported in SIMD-only mode"); 12097 } 12098 12099 const VarDecl * 12100 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD, 12101 const VarDecl *NativeParam) const { 12102 llvm_unreachable("Not supported in SIMD-only mode"); 12103 } 12104 12105 Address 12106 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF, 12107 const VarDecl *NativeParam, 12108 const VarDecl *TargetParam) const { 12109 llvm_unreachable("Not supported in SIMD-only mode"); 12110 } 12111