1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This provides a class for OpenMP runtime code generation. 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "CGOpenMPRuntime.h" 14 #include "CGCXXABI.h" 15 #include "CGCleanup.h" 16 #include "CGRecordLayout.h" 17 #include "CodeGenFunction.h" 18 #include "clang/AST/Attr.h" 19 #include "clang/AST/Decl.h" 20 #include "clang/AST/OpenMPClause.h" 21 #include "clang/AST/StmtOpenMP.h" 22 #include "clang/AST/StmtVisitor.h" 23 #include "clang/Basic/BitmaskEnum.h" 24 #include "clang/Basic/FileManager.h" 25 #include "clang/Basic/OpenMPKinds.h" 26 #include "clang/Basic/SourceManager.h" 27 #include "clang/CodeGen/ConstantInitBuilder.h" 28 #include "llvm/ADT/ArrayRef.h" 29 #include "llvm/ADT/SetOperations.h" 30 #include "llvm/ADT/StringExtras.h" 31 #include "llvm/Bitcode/BitcodeReader.h" 32 #include "llvm/Frontend/OpenMP/OMPIRBuilder.h" 33 #include "llvm/IR/Constants.h" 34 #include "llvm/IR/DerivedTypes.h" 35 #include "llvm/IR/GlobalValue.h" 36 #include "llvm/IR/Value.h" 37 #include "llvm/Support/AtomicOrdering.h" 38 #include "llvm/Support/Format.h" 39 #include "llvm/Support/raw_ostream.h" 40 #include <cassert> 41 42 using namespace clang; 43 using namespace CodeGen; 44 using namespace llvm::omp; 45 46 namespace { 47 /// Base class for handling code generation inside OpenMP regions. 48 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo { 49 public: 50 /// Kinds of OpenMP regions used in codegen. 51 enum CGOpenMPRegionKind { 52 /// Region with outlined function for standalone 'parallel' 53 /// directive. 54 ParallelOutlinedRegion, 55 /// Region with outlined function for standalone 'task' directive. 56 TaskOutlinedRegion, 57 /// Region for constructs that do not require function outlining, 58 /// like 'for', 'sections', 'atomic' etc. directives. 59 InlinedRegion, 60 /// Region with outlined function for standalone 'target' directive. 61 TargetRegion, 62 }; 63 64 CGOpenMPRegionInfo(const CapturedStmt &CS, 65 const CGOpenMPRegionKind RegionKind, 66 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 67 bool HasCancel) 68 : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind), 69 CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {} 70 71 CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind, 72 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 73 bool HasCancel) 74 : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen), 75 Kind(Kind), HasCancel(HasCancel) {} 76 77 /// Get a variable or parameter for storing global thread id 78 /// inside OpenMP construct. 79 virtual const VarDecl *getThreadIDVariable() const = 0; 80 81 /// Emit the captured statement body. 82 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override; 83 84 /// Get an LValue for the current ThreadID variable. 85 /// \return LValue for thread id variable. This LValue always has type int32*. 86 virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF); 87 88 virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {} 89 90 CGOpenMPRegionKind getRegionKind() const { return RegionKind; } 91 92 OpenMPDirectiveKind getDirectiveKind() const { return Kind; } 93 94 bool hasCancel() const { return HasCancel; } 95 96 static bool classof(const CGCapturedStmtInfo *Info) { 97 return Info->getKind() == CR_OpenMP; 98 } 99 100 ~CGOpenMPRegionInfo() override = default; 101 102 protected: 103 CGOpenMPRegionKind RegionKind; 104 RegionCodeGenTy CodeGen; 105 OpenMPDirectiveKind Kind; 106 bool HasCancel; 107 }; 108 109 /// API for captured statement code generation in OpenMP constructs. 110 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo { 111 public: 112 CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar, 113 const RegionCodeGenTy &CodeGen, 114 OpenMPDirectiveKind Kind, bool HasCancel, 115 StringRef HelperName) 116 : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind, 117 HasCancel), 118 ThreadIDVar(ThreadIDVar), HelperName(HelperName) { 119 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 120 } 121 122 /// Get a variable or parameter for storing global thread id 123 /// inside OpenMP construct. 124 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 125 126 /// Get the name of the capture helper. 127 StringRef getHelperName() const override { return HelperName; } 128 129 static bool classof(const CGCapturedStmtInfo *Info) { 130 return CGOpenMPRegionInfo::classof(Info) && 131 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 132 ParallelOutlinedRegion; 133 } 134 135 private: 136 /// A variable or parameter storing global thread id for OpenMP 137 /// constructs. 138 const VarDecl *ThreadIDVar; 139 StringRef HelperName; 140 }; 141 142 /// API for captured statement code generation in OpenMP constructs. 143 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo { 144 public: 145 class UntiedTaskActionTy final : public PrePostActionTy { 146 bool Untied; 147 const VarDecl *PartIDVar; 148 const RegionCodeGenTy UntiedCodeGen; 149 llvm::SwitchInst *UntiedSwitch = nullptr; 150 151 public: 152 UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar, 153 const RegionCodeGenTy &UntiedCodeGen) 154 : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {} 155 void Enter(CodeGenFunction &CGF) override { 156 if (Untied) { 157 // Emit task switching point. 158 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 159 CGF.GetAddrOfLocalVar(PartIDVar), 160 PartIDVar->getType()->castAs<PointerType>()); 161 llvm::Value *Res = 162 CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation()); 163 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done."); 164 UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB); 165 CGF.EmitBlock(DoneBB); 166 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 167 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 168 UntiedSwitch->addCase(CGF.Builder.getInt32(0), 169 CGF.Builder.GetInsertBlock()); 170 emitUntiedSwitch(CGF); 171 } 172 } 173 void emitUntiedSwitch(CodeGenFunction &CGF) const { 174 if (Untied) { 175 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 176 CGF.GetAddrOfLocalVar(PartIDVar), 177 PartIDVar->getType()->castAs<PointerType>()); 178 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 179 PartIdLVal); 180 UntiedCodeGen(CGF); 181 CodeGenFunction::JumpDest CurPoint = 182 CGF.getJumpDestInCurrentScope(".untied.next."); 183 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 184 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 185 UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 186 CGF.Builder.GetInsertBlock()); 187 CGF.EmitBranchThroughCleanup(CurPoint); 188 CGF.EmitBlock(CurPoint.getBlock()); 189 } 190 } 191 unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); } 192 }; 193 CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS, 194 const VarDecl *ThreadIDVar, 195 const RegionCodeGenTy &CodeGen, 196 OpenMPDirectiveKind Kind, bool HasCancel, 197 const UntiedTaskActionTy &Action) 198 : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel), 199 ThreadIDVar(ThreadIDVar), Action(Action) { 200 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 201 } 202 203 /// Get a variable or parameter for storing global thread id 204 /// inside OpenMP construct. 205 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 206 207 /// Get an LValue for the current ThreadID variable. 208 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override; 209 210 /// Get the name of the capture helper. 211 StringRef getHelperName() const override { return ".omp_outlined."; } 212 213 void emitUntiedSwitch(CodeGenFunction &CGF) override { 214 Action.emitUntiedSwitch(CGF); 215 } 216 217 static bool classof(const CGCapturedStmtInfo *Info) { 218 return CGOpenMPRegionInfo::classof(Info) && 219 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 220 TaskOutlinedRegion; 221 } 222 223 private: 224 /// A variable or parameter storing global thread id for OpenMP 225 /// constructs. 226 const VarDecl *ThreadIDVar; 227 /// Action for emitting code for untied tasks. 228 const UntiedTaskActionTy &Action; 229 }; 230 231 /// API for inlined captured statement code generation in OpenMP 232 /// constructs. 233 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo { 234 public: 235 CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI, 236 const RegionCodeGenTy &CodeGen, 237 OpenMPDirectiveKind Kind, bool HasCancel) 238 : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel), 239 OldCSI(OldCSI), 240 OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {} 241 242 // Retrieve the value of the context parameter. 243 llvm::Value *getContextValue() const override { 244 if (OuterRegionInfo) 245 return OuterRegionInfo->getContextValue(); 246 llvm_unreachable("No context value for inlined OpenMP region"); 247 } 248 249 void setContextValue(llvm::Value *V) override { 250 if (OuterRegionInfo) { 251 OuterRegionInfo->setContextValue(V); 252 return; 253 } 254 llvm_unreachable("No context value for inlined OpenMP region"); 255 } 256 257 /// Lookup the captured field decl for a variable. 258 const FieldDecl *lookup(const VarDecl *VD) const override { 259 if (OuterRegionInfo) 260 return OuterRegionInfo->lookup(VD); 261 // If there is no outer outlined region,no need to lookup in a list of 262 // captured variables, we can use the original one. 263 return nullptr; 264 } 265 266 FieldDecl *getThisFieldDecl() const override { 267 if (OuterRegionInfo) 268 return OuterRegionInfo->getThisFieldDecl(); 269 return nullptr; 270 } 271 272 /// Get a variable or parameter for storing global thread id 273 /// inside OpenMP construct. 274 const VarDecl *getThreadIDVariable() const override { 275 if (OuterRegionInfo) 276 return OuterRegionInfo->getThreadIDVariable(); 277 return nullptr; 278 } 279 280 /// Get an LValue for the current ThreadID variable. 281 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override { 282 if (OuterRegionInfo) 283 return OuterRegionInfo->getThreadIDVariableLValue(CGF); 284 llvm_unreachable("No LValue for inlined OpenMP construct"); 285 } 286 287 /// Get the name of the capture helper. 288 StringRef getHelperName() const override { 289 if (auto *OuterRegionInfo = getOldCSI()) 290 return OuterRegionInfo->getHelperName(); 291 llvm_unreachable("No helper name for inlined OpenMP construct"); 292 } 293 294 void emitUntiedSwitch(CodeGenFunction &CGF) override { 295 if (OuterRegionInfo) 296 OuterRegionInfo->emitUntiedSwitch(CGF); 297 } 298 299 CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; } 300 301 static bool classof(const CGCapturedStmtInfo *Info) { 302 return CGOpenMPRegionInfo::classof(Info) && 303 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion; 304 } 305 306 ~CGOpenMPInlinedRegionInfo() override = default; 307 308 private: 309 /// CodeGen info about outer OpenMP region. 310 CodeGenFunction::CGCapturedStmtInfo *OldCSI; 311 CGOpenMPRegionInfo *OuterRegionInfo; 312 }; 313 314 /// API for captured statement code generation in OpenMP target 315 /// constructs. For this captures, implicit parameters are used instead of the 316 /// captured fields. The name of the target region has to be unique in a given 317 /// application so it is provided by the client, because only the client has 318 /// the information to generate that. 319 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo { 320 public: 321 CGOpenMPTargetRegionInfo(const CapturedStmt &CS, 322 const RegionCodeGenTy &CodeGen, StringRef HelperName) 323 : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target, 324 /*HasCancel=*/false), 325 HelperName(HelperName) {} 326 327 /// This is unused for target regions because each starts executing 328 /// with a single thread. 329 const VarDecl *getThreadIDVariable() const override { return nullptr; } 330 331 /// Get the name of the capture helper. 332 StringRef getHelperName() const override { return HelperName; } 333 334 static bool classof(const CGCapturedStmtInfo *Info) { 335 return CGOpenMPRegionInfo::classof(Info) && 336 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion; 337 } 338 339 private: 340 StringRef HelperName; 341 }; 342 343 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) { 344 llvm_unreachable("No codegen for expressions"); 345 } 346 /// API for generation of expressions captured in a innermost OpenMP 347 /// region. 348 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo { 349 public: 350 CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS) 351 : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen, 352 OMPD_unknown, 353 /*HasCancel=*/false), 354 PrivScope(CGF) { 355 // Make sure the globals captured in the provided statement are local by 356 // using the privatization logic. We assume the same variable is not 357 // captured more than once. 358 for (const auto &C : CS.captures()) { 359 if (!C.capturesVariable() && !C.capturesVariableByCopy()) 360 continue; 361 362 const VarDecl *VD = C.getCapturedVar(); 363 if (VD->isLocalVarDeclOrParm()) 364 continue; 365 366 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD), 367 /*RefersToEnclosingVariableOrCapture=*/false, 368 VD->getType().getNonReferenceType(), VK_LValue, 369 C.getLocation()); 370 PrivScope.addPrivate( 371 VD, [&CGF, &DRE]() { return CGF.EmitLValue(&DRE).getAddress(CGF); }); 372 } 373 (void)PrivScope.Privatize(); 374 } 375 376 /// Lookup the captured field decl for a variable. 377 const FieldDecl *lookup(const VarDecl *VD) const override { 378 if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD)) 379 return FD; 380 return nullptr; 381 } 382 383 /// Emit the captured statement body. 384 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override { 385 llvm_unreachable("No body for expressions"); 386 } 387 388 /// Get a variable or parameter for storing global thread id 389 /// inside OpenMP construct. 390 const VarDecl *getThreadIDVariable() const override { 391 llvm_unreachable("No thread id for expressions"); 392 } 393 394 /// Get the name of the capture helper. 395 StringRef getHelperName() const override { 396 llvm_unreachable("No helper name for expressions"); 397 } 398 399 static bool classof(const CGCapturedStmtInfo *Info) { return false; } 400 401 private: 402 /// Private scope to capture global variables. 403 CodeGenFunction::OMPPrivateScope PrivScope; 404 }; 405 406 /// RAII for emitting code of OpenMP constructs. 407 class InlinedOpenMPRegionRAII { 408 CodeGenFunction &CGF; 409 llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields; 410 FieldDecl *LambdaThisCaptureField = nullptr; 411 const CodeGen::CGBlockInfo *BlockInfo = nullptr; 412 413 public: 414 /// Constructs region for combined constructs. 415 /// \param CodeGen Code generation sequence for combined directives. Includes 416 /// a list of functions used for code generation of implicitly inlined 417 /// regions. 418 InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen, 419 OpenMPDirectiveKind Kind, bool HasCancel) 420 : CGF(CGF) { 421 // Start emission for the construct. 422 CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo( 423 CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel); 424 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 425 LambdaThisCaptureField = CGF.LambdaThisCaptureField; 426 CGF.LambdaThisCaptureField = nullptr; 427 BlockInfo = CGF.BlockInfo; 428 CGF.BlockInfo = nullptr; 429 } 430 431 ~InlinedOpenMPRegionRAII() { 432 // Restore original CapturedStmtInfo only if we're done with code emission. 433 auto *OldCSI = 434 cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI(); 435 delete CGF.CapturedStmtInfo; 436 CGF.CapturedStmtInfo = OldCSI; 437 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 438 CGF.LambdaThisCaptureField = LambdaThisCaptureField; 439 CGF.BlockInfo = BlockInfo; 440 } 441 }; 442 443 /// Values for bit flags used in the ident_t to describe the fields. 444 /// All enumeric elements are named and described in accordance with the code 445 /// from https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 446 enum OpenMPLocationFlags : unsigned { 447 /// Use trampoline for internal microtask. 448 OMP_IDENT_IMD = 0x01, 449 /// Use c-style ident structure. 450 OMP_IDENT_KMPC = 0x02, 451 /// Atomic reduction option for kmpc_reduce. 452 OMP_ATOMIC_REDUCE = 0x10, 453 /// Explicit 'barrier' directive. 454 OMP_IDENT_BARRIER_EXPL = 0x20, 455 /// Implicit barrier in code. 456 OMP_IDENT_BARRIER_IMPL = 0x40, 457 /// Implicit barrier in 'for' directive. 458 OMP_IDENT_BARRIER_IMPL_FOR = 0x40, 459 /// Implicit barrier in 'sections' directive. 460 OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0, 461 /// Implicit barrier in 'single' directive. 462 OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140, 463 /// Call of __kmp_for_static_init for static loop. 464 OMP_IDENT_WORK_LOOP = 0x200, 465 /// Call of __kmp_for_static_init for sections. 466 OMP_IDENT_WORK_SECTIONS = 0x400, 467 /// Call of __kmp_for_static_init for distribute. 468 OMP_IDENT_WORK_DISTRIBUTE = 0x800, 469 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE) 470 }; 471 472 namespace { 473 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 474 /// Values for bit flags for marking which requires clauses have been used. 475 enum OpenMPOffloadingRequiresDirFlags : int64_t { 476 /// flag undefined. 477 OMP_REQ_UNDEFINED = 0x000, 478 /// no requires clause present. 479 OMP_REQ_NONE = 0x001, 480 /// reverse_offload clause. 481 OMP_REQ_REVERSE_OFFLOAD = 0x002, 482 /// unified_address clause. 483 OMP_REQ_UNIFIED_ADDRESS = 0x004, 484 /// unified_shared_memory clause. 485 OMP_REQ_UNIFIED_SHARED_MEMORY = 0x008, 486 /// dynamic_allocators clause. 487 OMP_REQ_DYNAMIC_ALLOCATORS = 0x010, 488 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS) 489 }; 490 491 enum OpenMPOffloadingReservedDeviceIDs { 492 /// Device ID if the device was not defined, runtime should get it 493 /// from environment variables in the spec. 494 OMP_DEVICEID_UNDEF = -1, 495 }; 496 } // anonymous namespace 497 498 /// Describes ident structure that describes a source location. 499 /// All descriptions are taken from 500 /// https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 501 /// Original structure: 502 /// typedef struct ident { 503 /// kmp_int32 reserved_1; /**< might be used in Fortran; 504 /// see above */ 505 /// kmp_int32 flags; /**< also f.flags; KMP_IDENT_xxx flags; 506 /// KMP_IDENT_KMPC identifies this union 507 /// member */ 508 /// kmp_int32 reserved_2; /**< not really used in Fortran any more; 509 /// see above */ 510 ///#if USE_ITT_BUILD 511 /// /* but currently used for storing 512 /// region-specific ITT */ 513 /// /* contextual information. */ 514 ///#endif /* USE_ITT_BUILD */ 515 /// kmp_int32 reserved_3; /**< source[4] in Fortran, do not use for 516 /// C++ */ 517 /// char const *psource; /**< String describing the source location. 518 /// The string is composed of semi-colon separated 519 // fields which describe the source file, 520 /// the function and a pair of line numbers that 521 /// delimit the construct. 522 /// */ 523 /// } ident_t; 524 enum IdentFieldIndex { 525 /// might be used in Fortran 526 IdentField_Reserved_1, 527 /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member. 528 IdentField_Flags, 529 /// Not really used in Fortran any more 530 IdentField_Reserved_2, 531 /// Source[4] in Fortran, do not use for C++ 532 IdentField_Reserved_3, 533 /// String describing the source location. The string is composed of 534 /// semi-colon separated fields which describe the source file, the function 535 /// and a pair of line numbers that delimit the construct. 536 IdentField_PSource 537 }; 538 539 /// Schedule types for 'omp for' loops (these enumerators are taken from 540 /// the enum sched_type in kmp.h). 541 enum OpenMPSchedType { 542 /// Lower bound for default (unordered) versions. 543 OMP_sch_lower = 32, 544 OMP_sch_static_chunked = 33, 545 OMP_sch_static = 34, 546 OMP_sch_dynamic_chunked = 35, 547 OMP_sch_guided_chunked = 36, 548 OMP_sch_runtime = 37, 549 OMP_sch_auto = 38, 550 /// static with chunk adjustment (e.g., simd) 551 OMP_sch_static_balanced_chunked = 45, 552 /// Lower bound for 'ordered' versions. 553 OMP_ord_lower = 64, 554 OMP_ord_static_chunked = 65, 555 OMP_ord_static = 66, 556 OMP_ord_dynamic_chunked = 67, 557 OMP_ord_guided_chunked = 68, 558 OMP_ord_runtime = 69, 559 OMP_ord_auto = 70, 560 OMP_sch_default = OMP_sch_static, 561 /// dist_schedule types 562 OMP_dist_sch_static_chunked = 91, 563 OMP_dist_sch_static = 92, 564 /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers. 565 /// Set if the monotonic schedule modifier was present. 566 OMP_sch_modifier_monotonic = (1 << 29), 567 /// Set if the nonmonotonic schedule modifier was present. 568 OMP_sch_modifier_nonmonotonic = (1 << 30), 569 }; 570 571 enum OpenMPRTLFunction { 572 /// Call to void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, 573 /// kmpc_micro microtask, ...); 574 OMPRTL__kmpc_fork_call, 575 /// Call to void *__kmpc_threadprivate_cached(ident_t *loc, 576 /// kmp_int32 global_tid, void *data, size_t size, void ***cache); 577 OMPRTL__kmpc_threadprivate_cached, 578 /// Call to void __kmpc_threadprivate_register( ident_t *, 579 /// void *data, kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor); 580 OMPRTL__kmpc_threadprivate_register, 581 // Call to __kmpc_int32 kmpc_global_thread_num(ident_t *loc); 582 OMPRTL__kmpc_global_thread_num, 583 // Call to void __kmpc_critical(ident_t *loc, kmp_int32 global_tid, 584 // kmp_critical_name *crit); 585 OMPRTL__kmpc_critical, 586 // Call to void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 587 // global_tid, kmp_critical_name *crit, uintptr_t hint); 588 OMPRTL__kmpc_critical_with_hint, 589 // Call to void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid, 590 // kmp_critical_name *crit); 591 OMPRTL__kmpc_end_critical, 592 // Call to kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32 593 // global_tid); 594 OMPRTL__kmpc_cancel_barrier, 595 // Call to void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid); 596 OMPRTL__kmpc_barrier, 597 // Call to void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid); 598 OMPRTL__kmpc_for_static_fini, 599 // Call to void __kmpc_serialized_parallel(ident_t *loc, kmp_int32 600 // global_tid); 601 OMPRTL__kmpc_serialized_parallel, 602 // Call to void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32 603 // global_tid); 604 OMPRTL__kmpc_end_serialized_parallel, 605 // Call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid, 606 // kmp_int32 num_threads); 607 OMPRTL__kmpc_push_num_threads, 608 // Call to void __kmpc_flush(ident_t *loc); 609 OMPRTL__kmpc_flush, 610 // Call to kmp_int32 __kmpc_master(ident_t *, kmp_int32 global_tid); 611 OMPRTL__kmpc_master, 612 // Call to void __kmpc_end_master(ident_t *, kmp_int32 global_tid); 613 OMPRTL__kmpc_end_master, 614 // Call to kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid, 615 // int end_part); 616 OMPRTL__kmpc_omp_taskyield, 617 // Call to kmp_int32 __kmpc_single(ident_t *, kmp_int32 global_tid); 618 OMPRTL__kmpc_single, 619 // Call to void __kmpc_end_single(ident_t *, kmp_int32 global_tid); 620 OMPRTL__kmpc_end_single, 621 // Call to kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 622 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 623 // kmp_routine_entry_t *task_entry); 624 OMPRTL__kmpc_omp_task_alloc, 625 // Call to kmp_task_t * __kmpc_omp_target_task_alloc(ident_t *, 626 // kmp_int32 gtid, kmp_int32 flags, size_t sizeof_kmp_task_t, 627 // size_t sizeof_shareds, kmp_routine_entry_t *task_entry, 628 // kmp_int64 device_id); 629 OMPRTL__kmpc_omp_target_task_alloc, 630 // Call to kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t * 631 // new_task); 632 OMPRTL__kmpc_omp_task, 633 // Call to void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid, 634 // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *), 635 // kmp_int32 didit); 636 OMPRTL__kmpc_copyprivate, 637 // Call to kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid, 638 // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void 639 // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck); 640 OMPRTL__kmpc_reduce, 641 // Call to kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32 642 // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, 643 // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name 644 // *lck); 645 OMPRTL__kmpc_reduce_nowait, 646 // Call to void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid, 647 // kmp_critical_name *lck); 648 OMPRTL__kmpc_end_reduce, 649 // Call to void __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid, 650 // kmp_critical_name *lck); 651 OMPRTL__kmpc_end_reduce_nowait, 652 // Call to void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 653 // kmp_task_t * new_task); 654 OMPRTL__kmpc_omp_task_begin_if0, 655 // Call to void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 656 // kmp_task_t * new_task); 657 OMPRTL__kmpc_omp_task_complete_if0, 658 // Call to void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid); 659 OMPRTL__kmpc_ordered, 660 // Call to void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid); 661 OMPRTL__kmpc_end_ordered, 662 // Call to kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 663 // global_tid); 664 OMPRTL__kmpc_omp_taskwait, 665 // Call to void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid); 666 OMPRTL__kmpc_taskgroup, 667 // Call to void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid); 668 OMPRTL__kmpc_end_taskgroup, 669 // Call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid, 670 // int proc_bind); 671 OMPRTL__kmpc_push_proc_bind, 672 // Call to kmp_int32 __kmpc_omp_task_with_deps(ident_t *loc_ref, kmp_int32 673 // gtid, kmp_task_t * new_task, kmp_int32 ndeps, kmp_depend_info_t 674 // *dep_list, kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 675 OMPRTL__kmpc_omp_task_with_deps, 676 // Call to void __kmpc_omp_wait_deps(ident_t *loc_ref, kmp_int32 677 // gtid, kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 678 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 679 OMPRTL__kmpc_omp_wait_deps, 680 // Call to kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 681 // global_tid, kmp_int32 cncl_kind); 682 OMPRTL__kmpc_cancellationpoint, 683 // Call to kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 684 // kmp_int32 cncl_kind); 685 OMPRTL__kmpc_cancel, 686 // Call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32 global_tid, 687 // kmp_int32 num_teams, kmp_int32 thread_limit); 688 OMPRTL__kmpc_push_num_teams, 689 // Call to void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro 690 // microtask, ...); 691 OMPRTL__kmpc_fork_teams, 692 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 693 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 694 // sched, kmp_uint64 grainsize, void *task_dup); 695 OMPRTL__kmpc_taskloop, 696 // Call to void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32 697 // num_dims, struct kmp_dim *dims); 698 OMPRTL__kmpc_doacross_init, 699 // Call to void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid); 700 OMPRTL__kmpc_doacross_fini, 701 // Call to void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64 702 // *vec); 703 OMPRTL__kmpc_doacross_post, 704 // Call to void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64 705 // *vec); 706 OMPRTL__kmpc_doacross_wait, 707 // Call to void *__kmpc_task_reduction_init(int gtid, int num_data, void 708 // *data); 709 OMPRTL__kmpc_task_reduction_init, 710 // Call to void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 711 // *d); 712 OMPRTL__kmpc_task_reduction_get_th_data, 713 // Call to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t al); 714 OMPRTL__kmpc_alloc, 715 // Call to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t al); 716 OMPRTL__kmpc_free, 717 718 // 719 // Offloading related calls 720 // 721 // Call to void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64 722 // size); 723 OMPRTL__kmpc_push_target_tripcount, 724 // Call to int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t 725 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 726 // *arg_types); 727 OMPRTL__tgt_target, 728 // Call to int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr, 729 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 730 // *arg_types); 731 OMPRTL__tgt_target_nowait, 732 // Call to int32_t __tgt_target_teams(int64_t device_id, void *host_ptr, 733 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 734 // *arg_types, int32_t num_teams, int32_t thread_limit); 735 OMPRTL__tgt_target_teams, 736 // Call to int32_t __tgt_target_teams_nowait(int64_t device_id, void 737 // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t 738 // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 739 OMPRTL__tgt_target_teams_nowait, 740 // Call to void __tgt_register_requires(int64_t flags); 741 OMPRTL__tgt_register_requires, 742 // Call to void __tgt_target_data_begin(int64_t device_id, int32_t arg_num, 743 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 744 OMPRTL__tgt_target_data_begin, 745 // Call to void __tgt_target_data_begin_nowait(int64_t device_id, int32_t 746 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 747 // *arg_types); 748 OMPRTL__tgt_target_data_begin_nowait, 749 // Call to void __tgt_target_data_end(int64_t device_id, int32_t arg_num, 750 // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types); 751 OMPRTL__tgt_target_data_end, 752 // Call to void __tgt_target_data_end_nowait(int64_t device_id, int32_t 753 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 754 // *arg_types); 755 OMPRTL__tgt_target_data_end_nowait, 756 // Call to void __tgt_target_data_update(int64_t device_id, int32_t arg_num, 757 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 758 OMPRTL__tgt_target_data_update, 759 // Call to void __tgt_target_data_update_nowait(int64_t device_id, int32_t 760 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 761 // *arg_types); 762 OMPRTL__tgt_target_data_update_nowait, 763 // Call to int64_t __tgt_mapper_num_components(void *rt_mapper_handle); 764 OMPRTL__tgt_mapper_num_components, 765 // Call to void __tgt_push_mapper_component(void *rt_mapper_handle, void 766 // *base, void *begin, int64_t size, int64_t type); 767 OMPRTL__tgt_push_mapper_component, 768 // Call to kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref, 769 // int gtid, kmp_task_t *task); 770 OMPRTL__kmpc_task_allow_completion_event, 771 }; 772 773 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP 774 /// region. 775 class CleanupTy final : public EHScopeStack::Cleanup { 776 PrePostActionTy *Action; 777 778 public: 779 explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {} 780 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 781 if (!CGF.HaveInsertPoint()) 782 return; 783 Action->Exit(CGF); 784 } 785 }; 786 787 } // anonymous namespace 788 789 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const { 790 CodeGenFunction::RunCleanupsScope Scope(CGF); 791 if (PrePostAction) { 792 CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction); 793 Callback(CodeGen, CGF, *PrePostAction); 794 } else { 795 PrePostActionTy Action; 796 Callback(CodeGen, CGF, Action); 797 } 798 } 799 800 /// Check if the combiner is a call to UDR combiner and if it is so return the 801 /// UDR decl used for reduction. 802 static const OMPDeclareReductionDecl * 803 getReductionInit(const Expr *ReductionOp) { 804 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 805 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 806 if (const auto *DRE = 807 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 808 if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) 809 return DRD; 810 return nullptr; 811 } 812 813 static void emitInitWithReductionInitializer(CodeGenFunction &CGF, 814 const OMPDeclareReductionDecl *DRD, 815 const Expr *InitOp, 816 Address Private, Address Original, 817 QualType Ty) { 818 if (DRD->getInitializer()) { 819 std::pair<llvm::Function *, llvm::Function *> Reduction = 820 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 821 const auto *CE = cast<CallExpr>(InitOp); 822 const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee()); 823 const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts(); 824 const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts(); 825 const auto *LHSDRE = 826 cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr()); 827 const auto *RHSDRE = 828 cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr()); 829 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 830 PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()), 831 [=]() { return Private; }); 832 PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()), 833 [=]() { return Original; }); 834 (void)PrivateScope.Privatize(); 835 RValue Func = RValue::get(Reduction.second); 836 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 837 CGF.EmitIgnoredExpr(InitOp); 838 } else { 839 llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty); 840 std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"}); 841 auto *GV = new llvm::GlobalVariable( 842 CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true, 843 llvm::GlobalValue::PrivateLinkage, Init, Name); 844 LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty); 845 RValue InitRVal; 846 switch (CGF.getEvaluationKind(Ty)) { 847 case TEK_Scalar: 848 InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation()); 849 break; 850 case TEK_Complex: 851 InitRVal = 852 RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation())); 853 break; 854 case TEK_Aggregate: 855 InitRVal = RValue::getAggregate(LV.getAddress(CGF)); 856 break; 857 } 858 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_RValue); 859 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal); 860 CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(), 861 /*IsInitializer=*/false); 862 } 863 } 864 865 /// Emit initialization of arrays of complex types. 866 /// \param DestAddr Address of the array. 867 /// \param Type Type of array. 868 /// \param Init Initial expression of array. 869 /// \param SrcAddr Address of the original array. 870 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr, 871 QualType Type, bool EmitDeclareReductionInit, 872 const Expr *Init, 873 const OMPDeclareReductionDecl *DRD, 874 Address SrcAddr = Address::invalid()) { 875 // Perform element-by-element initialization. 876 QualType ElementTy; 877 878 // Drill down to the base element type on both arrays. 879 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 880 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr); 881 DestAddr = 882 CGF.Builder.CreateElementBitCast(DestAddr, DestAddr.getElementType()); 883 if (DRD) 884 SrcAddr = 885 CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType()); 886 887 llvm::Value *SrcBegin = nullptr; 888 if (DRD) 889 SrcBegin = SrcAddr.getPointer(); 890 llvm::Value *DestBegin = DestAddr.getPointer(); 891 // Cast from pointer to array type to pointer to single element. 892 llvm::Value *DestEnd = CGF.Builder.CreateGEP(DestBegin, NumElements); 893 // The basic structure here is a while-do loop. 894 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body"); 895 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done"); 896 llvm::Value *IsEmpty = 897 CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty"); 898 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 899 900 // Enter the loop body, making that address the current address. 901 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 902 CGF.EmitBlock(BodyBB); 903 904 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 905 906 llvm::PHINode *SrcElementPHI = nullptr; 907 Address SrcElementCurrent = Address::invalid(); 908 if (DRD) { 909 SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2, 910 "omp.arraycpy.srcElementPast"); 911 SrcElementPHI->addIncoming(SrcBegin, EntryBB); 912 SrcElementCurrent = 913 Address(SrcElementPHI, 914 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 915 } 916 llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI( 917 DestBegin->getType(), 2, "omp.arraycpy.destElementPast"); 918 DestElementPHI->addIncoming(DestBegin, EntryBB); 919 Address DestElementCurrent = 920 Address(DestElementPHI, 921 DestAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 922 923 // Emit copy. 924 { 925 CodeGenFunction::RunCleanupsScope InitScope(CGF); 926 if (EmitDeclareReductionInit) { 927 emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent, 928 SrcElementCurrent, ElementTy); 929 } else 930 CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(), 931 /*IsInitializer=*/false); 932 } 933 934 if (DRD) { 935 // Shift the address forward by one element. 936 llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32( 937 SrcElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 938 SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock()); 939 } 940 941 // Shift the address forward by one element. 942 llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32( 943 DestElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 944 // Check whether we've reached the end. 945 llvm::Value *Done = 946 CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done"); 947 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 948 DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock()); 949 950 // Done. 951 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 952 } 953 954 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) { 955 return CGF.EmitOMPSharedLValue(E); 956 } 957 958 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF, 959 const Expr *E) { 960 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E)) 961 return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false); 962 return LValue(); 963 } 964 965 void ReductionCodeGen::emitAggregateInitialization( 966 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 967 const OMPDeclareReductionDecl *DRD) { 968 // Emit VarDecl with copy init for arrays. 969 // Get the address of the original variable captured in current 970 // captured region. 971 const auto *PrivateVD = 972 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 973 bool EmitDeclareReductionInit = 974 DRD && (DRD->getInitializer() || !PrivateVD->hasInit()); 975 EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(), 976 EmitDeclareReductionInit, 977 EmitDeclareReductionInit ? ClausesData[N].ReductionOp 978 : PrivateVD->getInit(), 979 DRD, SharedLVal.getAddress(CGF)); 980 } 981 982 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds, 983 ArrayRef<const Expr *> Privates, 984 ArrayRef<const Expr *> ReductionOps) { 985 ClausesData.reserve(Shareds.size()); 986 SharedAddresses.reserve(Shareds.size()); 987 Sizes.reserve(Shareds.size()); 988 BaseDecls.reserve(Shareds.size()); 989 auto IPriv = Privates.begin(); 990 auto IRed = ReductionOps.begin(); 991 for (const Expr *Ref : Shareds) { 992 ClausesData.emplace_back(Ref, *IPriv, *IRed); 993 std::advance(IPriv, 1); 994 std::advance(IRed, 1); 995 } 996 } 997 998 void ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, unsigned N) { 999 assert(SharedAddresses.size() == N && 1000 "Number of generated lvalues must be exactly N."); 1001 LValue First = emitSharedLValue(CGF, ClausesData[N].Ref); 1002 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref); 1003 SharedAddresses.emplace_back(First, Second); 1004 } 1005 1006 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) { 1007 const auto *PrivateVD = 1008 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1009 QualType PrivateType = PrivateVD->getType(); 1010 bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref); 1011 if (!PrivateType->isVariablyModifiedType()) { 1012 Sizes.emplace_back( 1013 CGF.getTypeSize( 1014 SharedAddresses[N].first.getType().getNonReferenceType()), 1015 nullptr); 1016 return; 1017 } 1018 llvm::Value *Size; 1019 llvm::Value *SizeInChars; 1020 auto *ElemType = cast<llvm::PointerType>( 1021 SharedAddresses[N].first.getPointer(CGF)->getType()) 1022 ->getElementType(); 1023 auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType); 1024 if (AsArraySection) { 1025 Size = CGF.Builder.CreatePtrDiff(SharedAddresses[N].second.getPointer(CGF), 1026 SharedAddresses[N].first.getPointer(CGF)); 1027 Size = CGF.Builder.CreateNUWAdd( 1028 Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1)); 1029 SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf); 1030 } else { 1031 SizeInChars = CGF.getTypeSize( 1032 SharedAddresses[N].first.getType().getNonReferenceType()); 1033 Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf); 1034 } 1035 Sizes.emplace_back(SizeInChars, Size); 1036 CodeGenFunction::OpaqueValueMapping OpaqueMap( 1037 CGF, 1038 cast<OpaqueValueExpr>( 1039 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 1040 RValue::get(Size)); 1041 CGF.EmitVariablyModifiedType(PrivateType); 1042 } 1043 1044 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N, 1045 llvm::Value *Size) { 1046 const auto *PrivateVD = 1047 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1048 QualType PrivateType = PrivateVD->getType(); 1049 if (!PrivateType->isVariablyModifiedType()) { 1050 assert(!Size && !Sizes[N].second && 1051 "Size should be nullptr for non-variably modified reduction " 1052 "items."); 1053 return; 1054 } 1055 CodeGenFunction::OpaqueValueMapping OpaqueMap( 1056 CGF, 1057 cast<OpaqueValueExpr>( 1058 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 1059 RValue::get(Size)); 1060 CGF.EmitVariablyModifiedType(PrivateType); 1061 } 1062 1063 void ReductionCodeGen::emitInitialization( 1064 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 1065 llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) { 1066 assert(SharedAddresses.size() > N && "No variable was generated"); 1067 const auto *PrivateVD = 1068 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1069 const OMPDeclareReductionDecl *DRD = 1070 getReductionInit(ClausesData[N].ReductionOp); 1071 QualType PrivateType = PrivateVD->getType(); 1072 PrivateAddr = CGF.Builder.CreateElementBitCast( 1073 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 1074 QualType SharedType = SharedAddresses[N].first.getType(); 1075 SharedLVal = CGF.MakeAddrLValue( 1076 CGF.Builder.CreateElementBitCast(SharedLVal.getAddress(CGF), 1077 CGF.ConvertTypeForMem(SharedType)), 1078 SharedType, SharedAddresses[N].first.getBaseInfo(), 1079 CGF.CGM.getTBAAInfoForSubobject(SharedAddresses[N].first, SharedType)); 1080 if (CGF.getContext().getAsArrayType(PrivateVD->getType())) { 1081 emitAggregateInitialization(CGF, N, PrivateAddr, SharedLVal, DRD); 1082 } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) { 1083 emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp, 1084 PrivateAddr, SharedLVal.getAddress(CGF), 1085 SharedLVal.getType()); 1086 } else if (!DefaultInit(CGF) && PrivateVD->hasInit() && 1087 !CGF.isTrivialInitializer(PrivateVD->getInit())) { 1088 CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr, 1089 PrivateVD->getType().getQualifiers(), 1090 /*IsInitializer=*/false); 1091 } 1092 } 1093 1094 bool ReductionCodeGen::needCleanups(unsigned N) { 1095 const auto *PrivateVD = 1096 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1097 QualType PrivateType = PrivateVD->getType(); 1098 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 1099 return DTorKind != QualType::DK_none; 1100 } 1101 1102 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N, 1103 Address PrivateAddr) { 1104 const auto *PrivateVD = 1105 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1106 QualType PrivateType = PrivateVD->getType(); 1107 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 1108 if (needCleanups(N)) { 1109 PrivateAddr = CGF.Builder.CreateElementBitCast( 1110 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 1111 CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType); 1112 } 1113 } 1114 1115 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 1116 LValue BaseLV) { 1117 BaseTy = BaseTy.getNonReferenceType(); 1118 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 1119 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 1120 if (const auto *PtrTy = BaseTy->getAs<PointerType>()) { 1121 BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(CGF), PtrTy); 1122 } else { 1123 LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(CGF), BaseTy); 1124 BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal); 1125 } 1126 BaseTy = BaseTy->getPointeeType(); 1127 } 1128 return CGF.MakeAddrLValue( 1129 CGF.Builder.CreateElementBitCast(BaseLV.getAddress(CGF), 1130 CGF.ConvertTypeForMem(ElTy)), 1131 BaseLV.getType(), BaseLV.getBaseInfo(), 1132 CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType())); 1133 } 1134 1135 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 1136 llvm::Type *BaseLVType, CharUnits BaseLVAlignment, 1137 llvm::Value *Addr) { 1138 Address Tmp = Address::invalid(); 1139 Address TopTmp = Address::invalid(); 1140 Address MostTopTmp = Address::invalid(); 1141 BaseTy = BaseTy.getNonReferenceType(); 1142 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 1143 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 1144 Tmp = CGF.CreateMemTemp(BaseTy); 1145 if (TopTmp.isValid()) 1146 CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp); 1147 else 1148 MostTopTmp = Tmp; 1149 TopTmp = Tmp; 1150 BaseTy = BaseTy->getPointeeType(); 1151 } 1152 llvm::Type *Ty = BaseLVType; 1153 if (Tmp.isValid()) 1154 Ty = Tmp.getElementType(); 1155 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty); 1156 if (Tmp.isValid()) { 1157 CGF.Builder.CreateStore(Addr, Tmp); 1158 return MostTopTmp; 1159 } 1160 return Address(Addr, BaseLVAlignment); 1161 } 1162 1163 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) { 1164 const VarDecl *OrigVD = nullptr; 1165 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) { 1166 const Expr *Base = OASE->getBase()->IgnoreParenImpCasts(); 1167 while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base)) 1168 Base = TempOASE->getBase()->IgnoreParenImpCasts(); 1169 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 1170 Base = TempASE->getBase()->IgnoreParenImpCasts(); 1171 DE = cast<DeclRefExpr>(Base); 1172 OrigVD = cast<VarDecl>(DE->getDecl()); 1173 } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) { 1174 const Expr *Base = ASE->getBase()->IgnoreParenImpCasts(); 1175 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 1176 Base = TempASE->getBase()->IgnoreParenImpCasts(); 1177 DE = cast<DeclRefExpr>(Base); 1178 OrigVD = cast<VarDecl>(DE->getDecl()); 1179 } 1180 return OrigVD; 1181 } 1182 1183 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N, 1184 Address PrivateAddr) { 1185 const DeclRefExpr *DE; 1186 if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) { 1187 BaseDecls.emplace_back(OrigVD); 1188 LValue OriginalBaseLValue = CGF.EmitLValue(DE); 1189 LValue BaseLValue = 1190 loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(), 1191 OriginalBaseLValue); 1192 llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff( 1193 BaseLValue.getPointer(CGF), SharedAddresses[N].first.getPointer(CGF)); 1194 llvm::Value *PrivatePointer = 1195 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 1196 PrivateAddr.getPointer(), 1197 SharedAddresses[N].first.getAddress(CGF).getType()); 1198 llvm::Value *Ptr = CGF.Builder.CreateGEP(PrivatePointer, Adjustment); 1199 return castToBase(CGF, OrigVD->getType(), 1200 SharedAddresses[N].first.getType(), 1201 OriginalBaseLValue.getAddress(CGF).getType(), 1202 OriginalBaseLValue.getAlignment(), Ptr); 1203 } 1204 BaseDecls.emplace_back( 1205 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl())); 1206 return PrivateAddr; 1207 } 1208 1209 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const { 1210 const OMPDeclareReductionDecl *DRD = 1211 getReductionInit(ClausesData[N].ReductionOp); 1212 return DRD && DRD->getInitializer(); 1213 } 1214 1215 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) { 1216 return CGF.EmitLoadOfPointerLValue( 1217 CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1218 getThreadIDVariable()->getType()->castAs<PointerType>()); 1219 } 1220 1221 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) { 1222 if (!CGF.HaveInsertPoint()) 1223 return; 1224 // 1.2.2 OpenMP Language Terminology 1225 // Structured block - An executable statement with a single entry at the 1226 // top and a single exit at the bottom. 1227 // The point of exit cannot be a branch out of the structured block. 1228 // longjmp() and throw() must not violate the entry/exit criteria. 1229 CGF.EHStack.pushTerminate(); 1230 CodeGen(CGF); 1231 CGF.EHStack.popTerminate(); 1232 } 1233 1234 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue( 1235 CodeGenFunction &CGF) { 1236 return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1237 getThreadIDVariable()->getType(), 1238 AlignmentSource::Decl); 1239 } 1240 1241 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC, 1242 QualType FieldTy) { 1243 auto *Field = FieldDecl::Create( 1244 C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy, 1245 C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()), 1246 /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit); 1247 Field->setAccess(AS_public); 1248 DC->addDecl(Field); 1249 return Field; 1250 } 1251 1252 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator, 1253 StringRef Separator) 1254 : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator), 1255 OffloadEntriesInfoManager(CGM) { 1256 ASTContext &C = CGM.getContext(); 1257 RecordDecl *RD = C.buildImplicitRecord("ident_t"); 1258 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 1259 RD->startDefinition(); 1260 // reserved_1 1261 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1262 // flags 1263 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1264 // reserved_2 1265 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1266 // reserved_3 1267 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1268 // psource 1269 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 1270 RD->completeDefinition(); 1271 IdentQTy = C.getRecordType(RD); 1272 IdentTy = CGM.getTypes().ConvertRecordDeclType(RD); 1273 KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8); 1274 1275 loadOffloadInfoMetadata(); 1276 } 1277 1278 void CGOpenMPRuntime::clear() { 1279 InternalVars.clear(); 1280 // Clean non-target variable declarations possibly used only in debug info. 1281 for (const auto &Data : EmittedNonTargetVariables) { 1282 if (!Data.getValue().pointsToAliveValue()) 1283 continue; 1284 auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue()); 1285 if (!GV) 1286 continue; 1287 if (!GV->isDeclaration() || GV->getNumUses() > 0) 1288 continue; 1289 GV->eraseFromParent(); 1290 } 1291 } 1292 1293 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const { 1294 SmallString<128> Buffer; 1295 llvm::raw_svector_ostream OS(Buffer); 1296 StringRef Sep = FirstSeparator; 1297 for (StringRef Part : Parts) { 1298 OS << Sep << Part; 1299 Sep = Separator; 1300 } 1301 return std::string(OS.str()); 1302 } 1303 1304 static llvm::Function * 1305 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty, 1306 const Expr *CombinerInitializer, const VarDecl *In, 1307 const VarDecl *Out, bool IsCombiner) { 1308 // void .omp_combiner.(Ty *in, Ty *out); 1309 ASTContext &C = CGM.getContext(); 1310 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 1311 FunctionArgList Args; 1312 ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(), 1313 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1314 ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(), 1315 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1316 Args.push_back(&OmpOutParm); 1317 Args.push_back(&OmpInParm); 1318 const CGFunctionInfo &FnInfo = 1319 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 1320 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 1321 std::string Name = CGM.getOpenMPRuntime().getName( 1322 {IsCombiner ? "omp_combiner" : "omp_initializer", ""}); 1323 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 1324 Name, &CGM.getModule()); 1325 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 1326 if (CGM.getLangOpts().Optimize) { 1327 Fn->removeFnAttr(llvm::Attribute::NoInline); 1328 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 1329 Fn->addFnAttr(llvm::Attribute::AlwaysInline); 1330 } 1331 CodeGenFunction CGF(CGM); 1332 // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions. 1333 // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions. 1334 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(), 1335 Out->getLocation()); 1336 CodeGenFunction::OMPPrivateScope Scope(CGF); 1337 Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm); 1338 Scope.addPrivate(In, [&CGF, AddrIn, PtrTy]() { 1339 return CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>()) 1340 .getAddress(CGF); 1341 }); 1342 Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm); 1343 Scope.addPrivate(Out, [&CGF, AddrOut, PtrTy]() { 1344 return CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>()) 1345 .getAddress(CGF); 1346 }); 1347 (void)Scope.Privatize(); 1348 if (!IsCombiner && Out->hasInit() && 1349 !CGF.isTrivialInitializer(Out->getInit())) { 1350 CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out), 1351 Out->getType().getQualifiers(), 1352 /*IsInitializer=*/true); 1353 } 1354 if (CombinerInitializer) 1355 CGF.EmitIgnoredExpr(CombinerInitializer); 1356 Scope.ForceCleanup(); 1357 CGF.FinishFunction(); 1358 return Fn; 1359 } 1360 1361 void CGOpenMPRuntime::emitUserDefinedReduction( 1362 CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) { 1363 if (UDRMap.count(D) > 0) 1364 return; 1365 llvm::Function *Combiner = emitCombinerOrInitializer( 1366 CGM, D->getType(), D->getCombiner(), 1367 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()), 1368 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()), 1369 /*IsCombiner=*/true); 1370 llvm::Function *Initializer = nullptr; 1371 if (const Expr *Init = D->getInitializer()) { 1372 Initializer = emitCombinerOrInitializer( 1373 CGM, D->getType(), 1374 D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init 1375 : nullptr, 1376 cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()), 1377 cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()), 1378 /*IsCombiner=*/false); 1379 } 1380 UDRMap.try_emplace(D, Combiner, Initializer); 1381 if (CGF) { 1382 auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn); 1383 Decls.second.push_back(D); 1384 } 1385 } 1386 1387 std::pair<llvm::Function *, llvm::Function *> 1388 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) { 1389 auto I = UDRMap.find(D); 1390 if (I != UDRMap.end()) 1391 return I->second; 1392 emitUserDefinedReduction(/*CGF=*/nullptr, D); 1393 return UDRMap.lookup(D); 1394 } 1395 1396 namespace { 1397 // Temporary RAII solution to perform a push/pop stack event on the OpenMP IR 1398 // Builder if one is present. 1399 struct PushAndPopStackRAII { 1400 PushAndPopStackRAII(llvm::OpenMPIRBuilder *OMPBuilder, CodeGenFunction &CGF, 1401 bool HasCancel) 1402 : OMPBuilder(OMPBuilder) { 1403 if (!OMPBuilder) 1404 return; 1405 1406 // The following callback is the crucial part of clangs cleanup process. 1407 // 1408 // NOTE: 1409 // Once the OpenMPIRBuilder is used to create parallel regions (and 1410 // similar), the cancellation destination (Dest below) is determined via 1411 // IP. That means if we have variables to finalize we split the block at IP, 1412 // use the new block (=BB) as destination to build a JumpDest (via 1413 // getJumpDestInCurrentScope(BB)) which then is fed to 1414 // EmitBranchThroughCleanup. Furthermore, there will not be the need 1415 // to push & pop an FinalizationInfo object. 1416 // The FiniCB will still be needed but at the point where the 1417 // OpenMPIRBuilder is asked to construct a parallel (or similar) construct. 1418 auto FiniCB = [&CGF](llvm::OpenMPIRBuilder::InsertPointTy IP) { 1419 assert(IP.getBlock()->end() == IP.getPoint() && 1420 "Clang CG should cause non-terminated block!"); 1421 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1422 CGF.Builder.restoreIP(IP); 1423 CodeGenFunction::JumpDest Dest = 1424 CGF.getOMPCancelDestination(OMPD_parallel); 1425 CGF.EmitBranchThroughCleanup(Dest); 1426 }; 1427 1428 // TODO: Remove this once we emit parallel regions through the 1429 // OpenMPIRBuilder as it can do this setup internally. 1430 llvm::OpenMPIRBuilder::FinalizationInfo FI( 1431 {FiniCB, OMPD_parallel, HasCancel}); 1432 OMPBuilder->pushFinalizationCB(std::move(FI)); 1433 } 1434 ~PushAndPopStackRAII() { 1435 if (OMPBuilder) 1436 OMPBuilder->popFinalizationCB(); 1437 } 1438 llvm::OpenMPIRBuilder *OMPBuilder; 1439 }; 1440 } // namespace 1441 1442 static llvm::Function *emitParallelOrTeamsOutlinedFunction( 1443 CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS, 1444 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, 1445 const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) { 1446 assert(ThreadIDVar->getType()->isPointerType() && 1447 "thread id variable must be of type kmp_int32 *"); 1448 CodeGenFunction CGF(CGM, true); 1449 bool HasCancel = false; 1450 if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D)) 1451 HasCancel = OPD->hasCancel(); 1452 else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D)) 1453 HasCancel = OPSD->hasCancel(); 1454 else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D)) 1455 HasCancel = OPFD->hasCancel(); 1456 else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D)) 1457 HasCancel = OPFD->hasCancel(); 1458 else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D)) 1459 HasCancel = OPFD->hasCancel(); 1460 else if (const auto *OPFD = 1461 dyn_cast<OMPTeamsDistributeParallelForDirective>(&D)) 1462 HasCancel = OPFD->hasCancel(); 1463 else if (const auto *OPFD = 1464 dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D)) 1465 HasCancel = OPFD->hasCancel(); 1466 1467 // TODO: Temporarily inform the OpenMPIRBuilder, if any, about the new 1468 // parallel region to make cancellation barriers work properly. 1469 llvm::OpenMPIRBuilder *OMPBuilder = CGM.getOpenMPIRBuilder(); 1470 PushAndPopStackRAII PSR(OMPBuilder, CGF, HasCancel); 1471 CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind, 1472 HasCancel, OutlinedHelperName); 1473 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1474 return CGF.GenerateOpenMPCapturedStmtFunction(*CS, D.getBeginLoc()); 1475 } 1476 1477 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction( 1478 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1479 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1480 const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel); 1481 return emitParallelOrTeamsOutlinedFunction( 1482 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1483 } 1484 1485 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction( 1486 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1487 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1488 const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams); 1489 return emitParallelOrTeamsOutlinedFunction( 1490 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1491 } 1492 1493 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction( 1494 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1495 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 1496 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 1497 bool Tied, unsigned &NumberOfParts) { 1498 auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF, 1499 PrePostActionTy &) { 1500 llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc()); 1501 llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 1502 llvm::Value *TaskArgs[] = { 1503 UpLoc, ThreadID, 1504 CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar), 1505 TaskTVar->getType()->castAs<PointerType>()) 1506 .getPointer(CGF)}; 1507 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs); 1508 }; 1509 CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar, 1510 UntiedCodeGen); 1511 CodeGen.setAction(Action); 1512 assert(!ThreadIDVar->getType()->isPointerType() && 1513 "thread id variable must be of type kmp_int32 for tasks"); 1514 const OpenMPDirectiveKind Region = 1515 isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop 1516 : OMPD_task; 1517 const CapturedStmt *CS = D.getCapturedStmt(Region); 1518 bool HasCancel = false; 1519 if (const auto *TD = dyn_cast<OMPTaskDirective>(&D)) 1520 HasCancel = TD->hasCancel(); 1521 else if (const auto *TD = dyn_cast<OMPTaskLoopDirective>(&D)) 1522 HasCancel = TD->hasCancel(); 1523 else if (const auto *TD = dyn_cast<OMPMasterTaskLoopDirective>(&D)) 1524 HasCancel = TD->hasCancel(); 1525 else if (const auto *TD = dyn_cast<OMPParallelMasterTaskLoopDirective>(&D)) 1526 HasCancel = TD->hasCancel(); 1527 1528 CodeGenFunction CGF(CGM, true); 1529 CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, 1530 InnermostKind, HasCancel, Action); 1531 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1532 llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS); 1533 if (!Tied) 1534 NumberOfParts = Action.getNumberOfParts(); 1535 return Res; 1536 } 1537 1538 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM, 1539 const RecordDecl *RD, const CGRecordLayout &RL, 1540 ArrayRef<llvm::Constant *> Data) { 1541 llvm::StructType *StructTy = RL.getLLVMType(); 1542 unsigned PrevIdx = 0; 1543 ConstantInitBuilder CIBuilder(CGM); 1544 auto DI = Data.begin(); 1545 for (const FieldDecl *FD : RD->fields()) { 1546 unsigned Idx = RL.getLLVMFieldNo(FD); 1547 // Fill the alignment. 1548 for (unsigned I = PrevIdx; I < Idx; ++I) 1549 Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I))); 1550 PrevIdx = Idx + 1; 1551 Fields.add(*DI); 1552 ++DI; 1553 } 1554 } 1555 1556 template <class... As> 1557 static llvm::GlobalVariable * 1558 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant, 1559 ArrayRef<llvm::Constant *> Data, const Twine &Name, 1560 As &&... Args) { 1561 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1562 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1563 ConstantInitBuilder CIBuilder(CGM); 1564 ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType()); 1565 buildStructValue(Fields, CGM, RD, RL, Data); 1566 return Fields.finishAndCreateGlobal( 1567 Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant, 1568 std::forward<As>(Args)...); 1569 } 1570 1571 template <typename T> 1572 static void 1573 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty, 1574 ArrayRef<llvm::Constant *> Data, 1575 T &Parent) { 1576 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1577 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1578 ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType()); 1579 buildStructValue(Fields, CGM, RD, RL, Data); 1580 Fields.finishAndAddTo(Parent); 1581 } 1582 1583 Address CGOpenMPRuntime::getOrCreateDefaultLocation(unsigned Flags) { 1584 CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy); 1585 unsigned Reserved2Flags = getDefaultLocationReserved2Flags(); 1586 FlagsTy FlagsKey(Flags, Reserved2Flags); 1587 llvm::Value *Entry = OpenMPDefaultLocMap.lookup(FlagsKey); 1588 if (!Entry) { 1589 if (!DefaultOpenMPPSource) { 1590 // Initialize default location for psource field of ident_t structure of 1591 // all ident_t objects. Format is ";file;function;line;column;;". 1592 // Taken from 1593 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp_str.cpp 1594 DefaultOpenMPPSource = 1595 CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;").getPointer(); 1596 DefaultOpenMPPSource = 1597 llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy); 1598 } 1599 1600 llvm::Constant *Data[] = { 1601 llvm::ConstantInt::getNullValue(CGM.Int32Ty), 1602 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 1603 llvm::ConstantInt::get(CGM.Int32Ty, Reserved2Flags), 1604 llvm::ConstantInt::getNullValue(CGM.Int32Ty), DefaultOpenMPPSource}; 1605 llvm::GlobalValue *DefaultOpenMPLocation = 1606 createGlobalStruct(CGM, IdentQTy, isDefaultLocationConstant(), Data, "", 1607 llvm::GlobalValue::PrivateLinkage); 1608 DefaultOpenMPLocation->setUnnamedAddr( 1609 llvm::GlobalValue::UnnamedAddr::Global); 1610 1611 OpenMPDefaultLocMap[FlagsKey] = Entry = DefaultOpenMPLocation; 1612 } 1613 return Address(Entry, Align); 1614 } 1615 1616 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF, 1617 bool AtCurrentPoint) { 1618 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1619 assert(!Elem.second.ServiceInsertPt && "Insert point is set already."); 1620 1621 llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty); 1622 if (AtCurrentPoint) { 1623 Elem.second.ServiceInsertPt = new llvm::BitCastInst( 1624 Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock()); 1625 } else { 1626 Elem.second.ServiceInsertPt = 1627 new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt"); 1628 Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt); 1629 } 1630 } 1631 1632 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) { 1633 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1634 if (Elem.second.ServiceInsertPt) { 1635 llvm::Instruction *Ptr = Elem.second.ServiceInsertPt; 1636 Elem.second.ServiceInsertPt = nullptr; 1637 Ptr->eraseFromParent(); 1638 } 1639 } 1640 1641 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF, 1642 SourceLocation Loc, 1643 unsigned Flags) { 1644 Flags |= OMP_IDENT_KMPC; 1645 // If no debug info is generated - return global default location. 1646 if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo || 1647 Loc.isInvalid()) 1648 return getOrCreateDefaultLocation(Flags).getPointer(); 1649 1650 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1651 1652 CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy); 1653 Address LocValue = Address::invalid(); 1654 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1655 if (I != OpenMPLocThreadIDMap.end()) 1656 LocValue = Address(I->second.DebugLoc, Align); 1657 1658 // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if 1659 // GetOpenMPThreadID was called before this routine. 1660 if (!LocValue.isValid()) { 1661 // Generate "ident_t .kmpc_loc.addr;" 1662 Address AI = CGF.CreateMemTemp(IdentQTy, ".kmpc_loc.addr"); 1663 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1664 Elem.second.DebugLoc = AI.getPointer(); 1665 LocValue = AI; 1666 1667 if (!Elem.second.ServiceInsertPt) 1668 setLocThreadIdInsertPt(CGF); 1669 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1670 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1671 CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags), 1672 CGF.getTypeSize(IdentQTy)); 1673 } 1674 1675 // char **psource = &.kmpc_loc_<flags>.addr.psource; 1676 LValue Base = CGF.MakeAddrLValue(LocValue, IdentQTy); 1677 auto Fields = cast<RecordDecl>(IdentQTy->getAsTagDecl())->field_begin(); 1678 LValue PSource = 1679 CGF.EmitLValueForField(Base, *std::next(Fields, IdentField_PSource)); 1680 1681 llvm::Value *OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding()); 1682 if (OMPDebugLoc == nullptr) { 1683 SmallString<128> Buffer2; 1684 llvm::raw_svector_ostream OS2(Buffer2); 1685 // Build debug location 1686 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc); 1687 OS2 << ";" << PLoc.getFilename() << ";"; 1688 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) 1689 OS2 << FD->getQualifiedNameAsString(); 1690 OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;"; 1691 OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str()); 1692 OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc; 1693 } 1694 // *psource = ";<File>;<Function>;<Line>;<Column>;;"; 1695 CGF.EmitStoreOfScalar(OMPDebugLoc, PSource); 1696 1697 // Our callers always pass this to a runtime function, so for 1698 // convenience, go ahead and return a naked pointer. 1699 return LocValue.getPointer(); 1700 } 1701 1702 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF, 1703 SourceLocation Loc) { 1704 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1705 1706 llvm::Value *ThreadID = nullptr; 1707 // Check whether we've already cached a load of the thread id in this 1708 // function. 1709 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1710 if (I != OpenMPLocThreadIDMap.end()) { 1711 ThreadID = I->second.ThreadID; 1712 if (ThreadID != nullptr) 1713 return ThreadID; 1714 } 1715 // If exceptions are enabled, do not use parameter to avoid possible crash. 1716 if (auto *OMPRegionInfo = 1717 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 1718 if (OMPRegionInfo->getThreadIDVariable()) { 1719 // Check if this an outlined function with thread id passed as argument. 1720 LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF); 1721 llvm::BasicBlock *TopBlock = CGF.AllocaInsertPt->getParent(); 1722 if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions || 1723 !CGF.getLangOpts().CXXExceptions || 1724 CGF.Builder.GetInsertBlock() == TopBlock || 1725 !isa<llvm::Instruction>(LVal.getPointer(CGF)) || 1726 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() == 1727 TopBlock || 1728 cast<llvm::Instruction>(LVal.getPointer(CGF))->getParent() == 1729 CGF.Builder.GetInsertBlock()) { 1730 ThreadID = CGF.EmitLoadOfScalar(LVal, Loc); 1731 // If value loaded in entry block, cache it and use it everywhere in 1732 // function. 1733 if (CGF.Builder.GetInsertBlock() == TopBlock) { 1734 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1735 Elem.second.ThreadID = ThreadID; 1736 } 1737 return ThreadID; 1738 } 1739 } 1740 } 1741 1742 // This is not an outlined function region - need to call __kmpc_int32 1743 // kmpc_global_thread_num(ident_t *loc). 1744 // Generate thread id value and cache this value for use across the 1745 // function. 1746 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1747 if (!Elem.second.ServiceInsertPt) 1748 setLocThreadIdInsertPt(CGF); 1749 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1750 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1751 llvm::CallInst *Call = CGF.Builder.CreateCall( 1752 createRuntimeFunction(OMPRTL__kmpc_global_thread_num), 1753 emitUpdateLocation(CGF, Loc)); 1754 Call->setCallingConv(CGF.getRuntimeCC()); 1755 Elem.second.ThreadID = Call; 1756 return Call; 1757 } 1758 1759 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) { 1760 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1761 if (OpenMPLocThreadIDMap.count(CGF.CurFn)) { 1762 clearLocThreadIdInsertPt(CGF); 1763 OpenMPLocThreadIDMap.erase(CGF.CurFn); 1764 } 1765 if (FunctionUDRMap.count(CGF.CurFn) > 0) { 1766 for(const auto *D : FunctionUDRMap[CGF.CurFn]) 1767 UDRMap.erase(D); 1768 FunctionUDRMap.erase(CGF.CurFn); 1769 } 1770 auto I = FunctionUDMMap.find(CGF.CurFn); 1771 if (I != FunctionUDMMap.end()) { 1772 for(const auto *D : I->second) 1773 UDMMap.erase(D); 1774 FunctionUDMMap.erase(I); 1775 } 1776 LastprivateConditionalToTypes.erase(CGF.CurFn); 1777 } 1778 1779 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() { 1780 return IdentTy->getPointerTo(); 1781 } 1782 1783 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() { 1784 if (!Kmpc_MicroTy) { 1785 // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...) 1786 llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty), 1787 llvm::PointerType::getUnqual(CGM.Int32Ty)}; 1788 Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true); 1789 } 1790 return llvm::PointerType::getUnqual(Kmpc_MicroTy); 1791 } 1792 1793 llvm::FunctionCallee CGOpenMPRuntime::createRuntimeFunction(unsigned Function) { 1794 llvm::FunctionCallee RTLFn = nullptr; 1795 switch (static_cast<OpenMPRTLFunction>(Function)) { 1796 case OMPRTL__kmpc_fork_call: { 1797 // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro 1798 // microtask, ...); 1799 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1800 getKmpc_MicroPointerTy()}; 1801 auto *FnTy = 1802 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true); 1803 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call"); 1804 if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) { 1805 if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) { 1806 llvm::LLVMContext &Ctx = F->getContext(); 1807 llvm::MDBuilder MDB(Ctx); 1808 // Annotate the callback behavior of the __kmpc_fork_call: 1809 // - The callback callee is argument number 2 (microtask). 1810 // - The first two arguments of the callback callee are unknown (-1). 1811 // - All variadic arguments to the __kmpc_fork_call are passed to the 1812 // callback callee. 1813 F->addMetadata( 1814 llvm::LLVMContext::MD_callback, 1815 *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding( 1816 2, {-1, -1}, 1817 /* VarArgsArePassed */ true)})); 1818 } 1819 } 1820 break; 1821 } 1822 case OMPRTL__kmpc_global_thread_num: { 1823 // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc); 1824 llvm::Type *TypeParams[] = {getIdentTyPointerTy()}; 1825 auto *FnTy = 1826 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1827 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num"); 1828 break; 1829 } 1830 case OMPRTL__kmpc_threadprivate_cached: { 1831 // Build void *__kmpc_threadprivate_cached(ident_t *loc, 1832 // kmp_int32 global_tid, void *data, size_t size, void ***cache); 1833 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1834 CGM.VoidPtrTy, CGM.SizeTy, 1835 CGM.VoidPtrTy->getPointerTo()->getPointerTo()}; 1836 auto *FnTy = 1837 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false); 1838 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached"); 1839 break; 1840 } 1841 case OMPRTL__kmpc_critical: { 1842 // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid, 1843 // kmp_critical_name *crit); 1844 llvm::Type *TypeParams[] = { 1845 getIdentTyPointerTy(), CGM.Int32Ty, 1846 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1847 auto *FnTy = 1848 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1849 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical"); 1850 break; 1851 } 1852 case OMPRTL__kmpc_critical_with_hint: { 1853 // Build void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 global_tid, 1854 // kmp_critical_name *crit, uintptr_t hint); 1855 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1856 llvm::PointerType::getUnqual(KmpCriticalNameTy), 1857 CGM.IntPtrTy}; 1858 auto *FnTy = 1859 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1860 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical_with_hint"); 1861 break; 1862 } 1863 case OMPRTL__kmpc_threadprivate_register: { 1864 // Build void __kmpc_threadprivate_register(ident_t *, void *data, 1865 // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor); 1866 // typedef void *(*kmpc_ctor)(void *); 1867 auto *KmpcCtorTy = 1868 llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 1869 /*isVarArg*/ false)->getPointerTo(); 1870 // typedef void *(*kmpc_cctor)(void *, void *); 1871 llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 1872 auto *KmpcCopyCtorTy = 1873 llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs, 1874 /*isVarArg*/ false) 1875 ->getPointerTo(); 1876 // typedef void (*kmpc_dtor)(void *); 1877 auto *KmpcDtorTy = 1878 llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false) 1879 ->getPointerTo(); 1880 llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy, 1881 KmpcCopyCtorTy, KmpcDtorTy}; 1882 auto *FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs, 1883 /*isVarArg*/ false); 1884 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register"); 1885 break; 1886 } 1887 case OMPRTL__kmpc_end_critical: { 1888 // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid, 1889 // kmp_critical_name *crit); 1890 llvm::Type *TypeParams[] = { 1891 getIdentTyPointerTy(), CGM.Int32Ty, 1892 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1893 auto *FnTy = 1894 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1895 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical"); 1896 break; 1897 } 1898 case OMPRTL__kmpc_cancel_barrier: { 1899 // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32 1900 // global_tid); 1901 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1902 auto *FnTy = 1903 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1904 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier"); 1905 break; 1906 } 1907 case OMPRTL__kmpc_barrier: { 1908 // Build void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid); 1909 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1910 auto *FnTy = 1911 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1912 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_barrier"); 1913 break; 1914 } 1915 case OMPRTL__kmpc_for_static_fini: { 1916 // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid); 1917 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1918 auto *FnTy = 1919 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1920 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini"); 1921 break; 1922 } 1923 case OMPRTL__kmpc_push_num_threads: { 1924 // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid, 1925 // kmp_int32 num_threads) 1926 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1927 CGM.Int32Ty}; 1928 auto *FnTy = 1929 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1930 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads"); 1931 break; 1932 } 1933 case OMPRTL__kmpc_serialized_parallel: { 1934 // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32 1935 // global_tid); 1936 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1937 auto *FnTy = 1938 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1939 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel"); 1940 break; 1941 } 1942 case OMPRTL__kmpc_end_serialized_parallel: { 1943 // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32 1944 // global_tid); 1945 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1946 auto *FnTy = 1947 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1948 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel"); 1949 break; 1950 } 1951 case OMPRTL__kmpc_flush: { 1952 // Build void __kmpc_flush(ident_t *loc); 1953 llvm::Type *TypeParams[] = {getIdentTyPointerTy()}; 1954 auto *FnTy = 1955 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1956 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush"); 1957 break; 1958 } 1959 case OMPRTL__kmpc_master: { 1960 // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid); 1961 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1962 auto *FnTy = 1963 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1964 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master"); 1965 break; 1966 } 1967 case OMPRTL__kmpc_end_master: { 1968 // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid); 1969 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1970 auto *FnTy = 1971 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1972 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master"); 1973 break; 1974 } 1975 case OMPRTL__kmpc_omp_taskyield: { 1976 // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid, 1977 // int end_part); 1978 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 1979 auto *FnTy = 1980 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1981 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield"); 1982 break; 1983 } 1984 case OMPRTL__kmpc_single: { 1985 // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid); 1986 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1987 auto *FnTy = 1988 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1989 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single"); 1990 break; 1991 } 1992 case OMPRTL__kmpc_end_single: { 1993 // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid); 1994 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1995 auto *FnTy = 1996 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1997 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single"); 1998 break; 1999 } 2000 case OMPRTL__kmpc_omp_task_alloc: { 2001 // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 2002 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 2003 // kmp_routine_entry_t *task_entry); 2004 assert(KmpRoutineEntryPtrTy != nullptr && 2005 "Type kmp_routine_entry_t must be created."); 2006 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2007 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy}; 2008 // Return void * and then cast to particular kmp_task_t type. 2009 auto *FnTy = 2010 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2011 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc"); 2012 break; 2013 } 2014 case OMPRTL__kmpc_omp_target_task_alloc: { 2015 // Build kmp_task_t *__kmpc_omp_target_task_alloc(ident_t *, kmp_int32 gtid, 2016 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 2017 // kmp_routine_entry_t *task_entry, kmp_int64 device_id); 2018 assert(KmpRoutineEntryPtrTy != nullptr && 2019 "Type kmp_routine_entry_t must be created."); 2020 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2021 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy, 2022 CGM.Int64Ty}; 2023 // Return void * and then cast to particular kmp_task_t type. 2024 auto *FnTy = 2025 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2026 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_target_task_alloc"); 2027 break; 2028 } 2029 case OMPRTL__kmpc_omp_task: { 2030 // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2031 // *new_task); 2032 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2033 CGM.VoidPtrTy}; 2034 auto *FnTy = 2035 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2036 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task"); 2037 break; 2038 } 2039 case OMPRTL__kmpc_copyprivate: { 2040 // Build void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid, 2041 // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *), 2042 // kmp_int32 didit); 2043 llvm::Type *CpyTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2044 auto *CpyFnTy = 2045 llvm::FunctionType::get(CGM.VoidTy, CpyTypeParams, /*isVarArg=*/false); 2046 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.SizeTy, 2047 CGM.VoidPtrTy, CpyFnTy->getPointerTo(), 2048 CGM.Int32Ty}; 2049 auto *FnTy = 2050 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2051 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_copyprivate"); 2052 break; 2053 } 2054 case OMPRTL__kmpc_reduce: { 2055 // Build kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid, 2056 // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void 2057 // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck); 2058 llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2059 auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams, 2060 /*isVarArg=*/false); 2061 llvm::Type *TypeParams[] = { 2062 getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy, 2063 CGM.VoidPtrTy, ReduceFnTy->getPointerTo(), 2064 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2065 auto *FnTy = 2066 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2067 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce"); 2068 break; 2069 } 2070 case OMPRTL__kmpc_reduce_nowait: { 2071 // Build kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32 2072 // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, 2073 // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name 2074 // *lck); 2075 llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2076 auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams, 2077 /*isVarArg=*/false); 2078 llvm::Type *TypeParams[] = { 2079 getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy, 2080 CGM.VoidPtrTy, ReduceFnTy->getPointerTo(), 2081 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2082 auto *FnTy = 2083 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2084 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce_nowait"); 2085 break; 2086 } 2087 case OMPRTL__kmpc_end_reduce: { 2088 // Build void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid, 2089 // kmp_critical_name *lck); 2090 llvm::Type *TypeParams[] = { 2091 getIdentTyPointerTy(), CGM.Int32Ty, 2092 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2093 auto *FnTy = 2094 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2095 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce"); 2096 break; 2097 } 2098 case OMPRTL__kmpc_end_reduce_nowait: { 2099 // Build __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid, 2100 // kmp_critical_name *lck); 2101 llvm::Type *TypeParams[] = { 2102 getIdentTyPointerTy(), CGM.Int32Ty, 2103 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2104 auto *FnTy = 2105 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2106 RTLFn = 2107 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce_nowait"); 2108 break; 2109 } 2110 case OMPRTL__kmpc_omp_task_begin_if0: { 2111 // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2112 // *new_task); 2113 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2114 CGM.VoidPtrTy}; 2115 auto *FnTy = 2116 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2117 RTLFn = 2118 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_begin_if0"); 2119 break; 2120 } 2121 case OMPRTL__kmpc_omp_task_complete_if0: { 2122 // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2123 // *new_task); 2124 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2125 CGM.VoidPtrTy}; 2126 auto *FnTy = 2127 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2128 RTLFn = CGM.CreateRuntimeFunction(FnTy, 2129 /*Name=*/"__kmpc_omp_task_complete_if0"); 2130 break; 2131 } 2132 case OMPRTL__kmpc_ordered: { 2133 // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid); 2134 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2135 auto *FnTy = 2136 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2137 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_ordered"); 2138 break; 2139 } 2140 case OMPRTL__kmpc_end_ordered: { 2141 // Build void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid); 2142 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2143 auto *FnTy = 2144 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2145 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_ordered"); 2146 break; 2147 } 2148 case OMPRTL__kmpc_omp_taskwait: { 2149 // Build kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 global_tid); 2150 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2151 auto *FnTy = 2152 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2153 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_omp_taskwait"); 2154 break; 2155 } 2156 case OMPRTL__kmpc_taskgroup: { 2157 // Build void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid); 2158 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2159 auto *FnTy = 2160 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2161 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_taskgroup"); 2162 break; 2163 } 2164 case OMPRTL__kmpc_end_taskgroup: { 2165 // Build void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid); 2166 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2167 auto *FnTy = 2168 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2169 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_taskgroup"); 2170 break; 2171 } 2172 case OMPRTL__kmpc_push_proc_bind: { 2173 // Build void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid, 2174 // int proc_bind) 2175 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2176 auto *FnTy = 2177 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2178 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_proc_bind"); 2179 break; 2180 } 2181 case OMPRTL__kmpc_omp_task_with_deps: { 2182 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 2183 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 2184 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 2185 llvm::Type *TypeParams[] = { 2186 getIdentTyPointerTy(), CGM.Int32Ty, CGM.VoidPtrTy, CGM.Int32Ty, 2187 CGM.VoidPtrTy, CGM.Int32Ty, CGM.VoidPtrTy}; 2188 auto *FnTy = 2189 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2190 RTLFn = 2191 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_with_deps"); 2192 break; 2193 } 2194 case OMPRTL__kmpc_omp_wait_deps: { 2195 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 2196 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 ndeps_noalias, 2197 // kmp_depend_info_t *noalias_dep_list); 2198 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2199 CGM.Int32Ty, CGM.VoidPtrTy, 2200 CGM.Int32Ty, CGM.VoidPtrTy}; 2201 auto *FnTy = 2202 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2203 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_wait_deps"); 2204 break; 2205 } 2206 case OMPRTL__kmpc_cancellationpoint: { 2207 // Build kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 2208 // global_tid, kmp_int32 cncl_kind) 2209 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2210 auto *FnTy = 2211 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2212 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancellationpoint"); 2213 break; 2214 } 2215 case OMPRTL__kmpc_cancel: { 2216 // Build kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 2217 // kmp_int32 cncl_kind) 2218 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2219 auto *FnTy = 2220 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2221 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancel"); 2222 break; 2223 } 2224 case OMPRTL__kmpc_push_num_teams: { 2225 // Build void kmpc_push_num_teams (ident_t loc, kmp_int32 global_tid, 2226 // kmp_int32 num_teams, kmp_int32 num_threads) 2227 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2228 CGM.Int32Ty}; 2229 auto *FnTy = 2230 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2231 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_teams"); 2232 break; 2233 } 2234 case OMPRTL__kmpc_fork_teams: { 2235 // Build void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro 2236 // microtask, ...); 2237 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2238 getKmpc_MicroPointerTy()}; 2239 auto *FnTy = 2240 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true); 2241 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_teams"); 2242 if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) { 2243 if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) { 2244 llvm::LLVMContext &Ctx = F->getContext(); 2245 llvm::MDBuilder MDB(Ctx); 2246 // Annotate the callback behavior of the __kmpc_fork_teams: 2247 // - The callback callee is argument number 2 (microtask). 2248 // - The first two arguments of the callback callee are unknown (-1). 2249 // - All variadic arguments to the __kmpc_fork_teams are passed to the 2250 // callback callee. 2251 F->addMetadata( 2252 llvm::LLVMContext::MD_callback, 2253 *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding( 2254 2, {-1, -1}, 2255 /* VarArgsArePassed */ true)})); 2256 } 2257 } 2258 break; 2259 } 2260 case OMPRTL__kmpc_taskloop: { 2261 // Build void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 2262 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 2263 // sched, kmp_uint64 grainsize, void *task_dup); 2264 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), 2265 CGM.IntTy, 2266 CGM.VoidPtrTy, 2267 CGM.IntTy, 2268 CGM.Int64Ty->getPointerTo(), 2269 CGM.Int64Ty->getPointerTo(), 2270 CGM.Int64Ty, 2271 CGM.IntTy, 2272 CGM.IntTy, 2273 CGM.Int64Ty, 2274 CGM.VoidPtrTy}; 2275 auto *FnTy = 2276 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2277 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_taskloop"); 2278 break; 2279 } 2280 case OMPRTL__kmpc_doacross_init: { 2281 // Build void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32 2282 // num_dims, struct kmp_dim *dims); 2283 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), 2284 CGM.Int32Ty, 2285 CGM.Int32Ty, 2286 CGM.VoidPtrTy}; 2287 auto *FnTy = 2288 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2289 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_init"); 2290 break; 2291 } 2292 case OMPRTL__kmpc_doacross_fini: { 2293 // Build void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid); 2294 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2295 auto *FnTy = 2296 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2297 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_fini"); 2298 break; 2299 } 2300 case OMPRTL__kmpc_doacross_post: { 2301 // Build void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64 2302 // *vec); 2303 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2304 CGM.Int64Ty->getPointerTo()}; 2305 auto *FnTy = 2306 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2307 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_post"); 2308 break; 2309 } 2310 case OMPRTL__kmpc_doacross_wait: { 2311 // Build void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64 2312 // *vec); 2313 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2314 CGM.Int64Ty->getPointerTo()}; 2315 auto *FnTy = 2316 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2317 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_wait"); 2318 break; 2319 } 2320 case OMPRTL__kmpc_task_reduction_init: { 2321 // Build void *__kmpc_task_reduction_init(int gtid, int num_data, void 2322 // *data); 2323 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.IntTy, CGM.VoidPtrTy}; 2324 auto *FnTy = 2325 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2326 RTLFn = 2327 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_task_reduction_init"); 2328 break; 2329 } 2330 case OMPRTL__kmpc_task_reduction_get_th_data: { 2331 // Build void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 2332 // *d); 2333 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy}; 2334 auto *FnTy = 2335 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2336 RTLFn = CGM.CreateRuntimeFunction( 2337 FnTy, /*Name=*/"__kmpc_task_reduction_get_th_data"); 2338 break; 2339 } 2340 case OMPRTL__kmpc_alloc: { 2341 // Build to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t 2342 // al); omp_allocator_handle_t type is void *. 2343 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.SizeTy, CGM.VoidPtrTy}; 2344 auto *FnTy = 2345 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2346 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_alloc"); 2347 break; 2348 } 2349 case OMPRTL__kmpc_free: { 2350 // Build to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t 2351 // al); omp_allocator_handle_t type is void *. 2352 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy}; 2353 auto *FnTy = 2354 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2355 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_free"); 2356 break; 2357 } 2358 case OMPRTL__kmpc_push_target_tripcount: { 2359 // Build void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64 2360 // size); 2361 llvm::Type *TypeParams[] = {CGM.Int64Ty, CGM.Int64Ty}; 2362 llvm::FunctionType *FnTy = 2363 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2364 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_target_tripcount"); 2365 break; 2366 } 2367 case OMPRTL__tgt_target: { 2368 // Build int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t 2369 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2370 // *arg_types); 2371 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2372 CGM.VoidPtrTy, 2373 CGM.Int32Ty, 2374 CGM.VoidPtrPtrTy, 2375 CGM.VoidPtrPtrTy, 2376 CGM.Int64Ty->getPointerTo(), 2377 CGM.Int64Ty->getPointerTo()}; 2378 auto *FnTy = 2379 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2380 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target"); 2381 break; 2382 } 2383 case OMPRTL__tgt_target_nowait: { 2384 // Build int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr, 2385 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, 2386 // int64_t *arg_types); 2387 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2388 CGM.VoidPtrTy, 2389 CGM.Int32Ty, 2390 CGM.VoidPtrPtrTy, 2391 CGM.VoidPtrPtrTy, 2392 CGM.Int64Ty->getPointerTo(), 2393 CGM.Int64Ty->getPointerTo()}; 2394 auto *FnTy = 2395 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2396 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_nowait"); 2397 break; 2398 } 2399 case OMPRTL__tgt_target_teams: { 2400 // Build int32_t __tgt_target_teams(int64_t device_id, void *host_ptr, 2401 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, 2402 // int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 2403 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2404 CGM.VoidPtrTy, 2405 CGM.Int32Ty, 2406 CGM.VoidPtrPtrTy, 2407 CGM.VoidPtrPtrTy, 2408 CGM.Int64Ty->getPointerTo(), 2409 CGM.Int64Ty->getPointerTo(), 2410 CGM.Int32Ty, 2411 CGM.Int32Ty}; 2412 auto *FnTy = 2413 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2414 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams"); 2415 break; 2416 } 2417 case OMPRTL__tgt_target_teams_nowait: { 2418 // Build int32_t __tgt_target_teams_nowait(int64_t device_id, void 2419 // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t 2420 // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 2421 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2422 CGM.VoidPtrTy, 2423 CGM.Int32Ty, 2424 CGM.VoidPtrPtrTy, 2425 CGM.VoidPtrPtrTy, 2426 CGM.Int64Ty->getPointerTo(), 2427 CGM.Int64Ty->getPointerTo(), 2428 CGM.Int32Ty, 2429 CGM.Int32Ty}; 2430 auto *FnTy = 2431 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2432 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams_nowait"); 2433 break; 2434 } 2435 case OMPRTL__tgt_register_requires: { 2436 // Build void __tgt_register_requires(int64_t flags); 2437 llvm::Type *TypeParams[] = {CGM.Int64Ty}; 2438 auto *FnTy = 2439 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2440 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_requires"); 2441 break; 2442 } 2443 case OMPRTL__tgt_target_data_begin: { 2444 // Build void __tgt_target_data_begin(int64_t device_id, int32_t arg_num, 2445 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2446 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2447 CGM.Int32Ty, 2448 CGM.VoidPtrPtrTy, 2449 CGM.VoidPtrPtrTy, 2450 CGM.Int64Ty->getPointerTo(), 2451 CGM.Int64Ty->getPointerTo()}; 2452 auto *FnTy = 2453 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2454 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin"); 2455 break; 2456 } 2457 case OMPRTL__tgt_target_data_begin_nowait: { 2458 // Build void __tgt_target_data_begin_nowait(int64_t device_id, int32_t 2459 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2460 // *arg_types); 2461 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2462 CGM.Int32Ty, 2463 CGM.VoidPtrPtrTy, 2464 CGM.VoidPtrPtrTy, 2465 CGM.Int64Ty->getPointerTo(), 2466 CGM.Int64Ty->getPointerTo()}; 2467 auto *FnTy = 2468 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2469 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin_nowait"); 2470 break; 2471 } 2472 case OMPRTL__tgt_target_data_end: { 2473 // Build void __tgt_target_data_end(int64_t device_id, int32_t arg_num, 2474 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2475 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2476 CGM.Int32Ty, 2477 CGM.VoidPtrPtrTy, 2478 CGM.VoidPtrPtrTy, 2479 CGM.Int64Ty->getPointerTo(), 2480 CGM.Int64Ty->getPointerTo()}; 2481 auto *FnTy = 2482 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2483 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end"); 2484 break; 2485 } 2486 case OMPRTL__tgt_target_data_end_nowait: { 2487 // Build void __tgt_target_data_end_nowait(int64_t device_id, int32_t 2488 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2489 // *arg_types); 2490 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2491 CGM.Int32Ty, 2492 CGM.VoidPtrPtrTy, 2493 CGM.VoidPtrPtrTy, 2494 CGM.Int64Ty->getPointerTo(), 2495 CGM.Int64Ty->getPointerTo()}; 2496 auto *FnTy = 2497 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2498 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end_nowait"); 2499 break; 2500 } 2501 case OMPRTL__tgt_target_data_update: { 2502 // Build void __tgt_target_data_update(int64_t device_id, int32_t arg_num, 2503 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2504 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2505 CGM.Int32Ty, 2506 CGM.VoidPtrPtrTy, 2507 CGM.VoidPtrPtrTy, 2508 CGM.Int64Ty->getPointerTo(), 2509 CGM.Int64Ty->getPointerTo()}; 2510 auto *FnTy = 2511 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2512 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update"); 2513 break; 2514 } 2515 case OMPRTL__tgt_target_data_update_nowait: { 2516 // Build void __tgt_target_data_update_nowait(int64_t device_id, int32_t 2517 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2518 // *arg_types); 2519 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2520 CGM.Int32Ty, 2521 CGM.VoidPtrPtrTy, 2522 CGM.VoidPtrPtrTy, 2523 CGM.Int64Ty->getPointerTo(), 2524 CGM.Int64Ty->getPointerTo()}; 2525 auto *FnTy = 2526 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2527 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update_nowait"); 2528 break; 2529 } 2530 case OMPRTL__tgt_mapper_num_components: { 2531 // Build int64_t __tgt_mapper_num_components(void *rt_mapper_handle); 2532 llvm::Type *TypeParams[] = {CGM.VoidPtrTy}; 2533 auto *FnTy = 2534 llvm::FunctionType::get(CGM.Int64Ty, TypeParams, /*isVarArg*/ false); 2535 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_mapper_num_components"); 2536 break; 2537 } 2538 case OMPRTL__tgt_push_mapper_component: { 2539 // Build void __tgt_push_mapper_component(void *rt_mapper_handle, void 2540 // *base, void *begin, int64_t size, int64_t type); 2541 llvm::Type *TypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy, CGM.VoidPtrTy, 2542 CGM.Int64Ty, CGM.Int64Ty}; 2543 auto *FnTy = 2544 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2545 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_push_mapper_component"); 2546 break; 2547 } 2548 case OMPRTL__kmpc_task_allow_completion_event: { 2549 // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref, 2550 // int gtid, kmp_task_t *task); 2551 auto *FnTy = llvm::FunctionType::get( 2552 CGM.VoidPtrTy, {getIdentTyPointerTy(), CGM.IntTy, CGM.VoidPtrTy}, 2553 /*isVarArg=*/false); 2554 RTLFn = 2555 CGM.CreateRuntimeFunction(FnTy, "__kmpc_task_allow_completion_event"); 2556 break; 2557 } 2558 } 2559 assert(RTLFn && "Unable to find OpenMP runtime function"); 2560 return RTLFn; 2561 } 2562 2563 llvm::FunctionCallee 2564 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned) { 2565 assert((IVSize == 32 || IVSize == 64) && 2566 "IV size is not compatible with the omp runtime"); 2567 StringRef Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4" 2568 : "__kmpc_for_static_init_4u") 2569 : (IVSigned ? "__kmpc_for_static_init_8" 2570 : "__kmpc_for_static_init_8u"); 2571 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2572 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 2573 llvm::Type *TypeParams[] = { 2574 getIdentTyPointerTy(), // loc 2575 CGM.Int32Ty, // tid 2576 CGM.Int32Ty, // schedtype 2577 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 2578 PtrTy, // p_lower 2579 PtrTy, // p_upper 2580 PtrTy, // p_stride 2581 ITy, // incr 2582 ITy // chunk 2583 }; 2584 auto *FnTy = 2585 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2586 return CGM.CreateRuntimeFunction(FnTy, Name); 2587 } 2588 2589 llvm::FunctionCallee 2590 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) { 2591 assert((IVSize == 32 || IVSize == 64) && 2592 "IV size is not compatible with the omp runtime"); 2593 StringRef Name = 2594 IVSize == 32 2595 ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u") 2596 : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u"); 2597 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2598 llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc 2599 CGM.Int32Ty, // tid 2600 CGM.Int32Ty, // schedtype 2601 ITy, // lower 2602 ITy, // upper 2603 ITy, // stride 2604 ITy // chunk 2605 }; 2606 auto *FnTy = 2607 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2608 return CGM.CreateRuntimeFunction(FnTy, Name); 2609 } 2610 2611 llvm::FunctionCallee 2612 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) { 2613 assert((IVSize == 32 || IVSize == 64) && 2614 "IV size is not compatible with the omp runtime"); 2615 StringRef Name = 2616 IVSize == 32 2617 ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u") 2618 : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u"); 2619 llvm::Type *TypeParams[] = { 2620 getIdentTyPointerTy(), // loc 2621 CGM.Int32Ty, // tid 2622 }; 2623 auto *FnTy = 2624 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2625 return CGM.CreateRuntimeFunction(FnTy, Name); 2626 } 2627 2628 llvm::FunctionCallee 2629 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) { 2630 assert((IVSize == 32 || IVSize == 64) && 2631 "IV size is not compatible with the omp runtime"); 2632 StringRef Name = 2633 IVSize == 32 2634 ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u") 2635 : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u"); 2636 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2637 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 2638 llvm::Type *TypeParams[] = { 2639 getIdentTyPointerTy(), // loc 2640 CGM.Int32Ty, // tid 2641 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 2642 PtrTy, // p_lower 2643 PtrTy, // p_upper 2644 PtrTy // p_stride 2645 }; 2646 auto *FnTy = 2647 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2648 return CGM.CreateRuntimeFunction(FnTy, Name); 2649 } 2650 2651 /// Obtain information that uniquely identifies a target entry. This 2652 /// consists of the file and device IDs as well as line number associated with 2653 /// the relevant entry source location. 2654 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc, 2655 unsigned &DeviceID, unsigned &FileID, 2656 unsigned &LineNum) { 2657 SourceManager &SM = C.getSourceManager(); 2658 2659 // The loc should be always valid and have a file ID (the user cannot use 2660 // #pragma directives in macros) 2661 2662 assert(Loc.isValid() && "Source location is expected to be always valid."); 2663 2664 PresumedLoc PLoc = SM.getPresumedLoc(Loc); 2665 assert(PLoc.isValid() && "Source location is expected to be always valid."); 2666 2667 llvm::sys::fs::UniqueID ID; 2668 if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID)) 2669 SM.getDiagnostics().Report(diag::err_cannot_open_file) 2670 << PLoc.getFilename() << EC.message(); 2671 2672 DeviceID = ID.getDevice(); 2673 FileID = ID.getFile(); 2674 LineNum = PLoc.getLine(); 2675 } 2676 2677 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) { 2678 if (CGM.getLangOpts().OpenMPSimd) 2679 return Address::invalid(); 2680 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 2681 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 2682 if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link || 2683 (*Res == OMPDeclareTargetDeclAttr::MT_To && 2684 HasRequiresUnifiedSharedMemory))) { 2685 SmallString<64> PtrName; 2686 { 2687 llvm::raw_svector_ostream OS(PtrName); 2688 OS << CGM.getMangledName(GlobalDecl(VD)); 2689 if (!VD->isExternallyVisible()) { 2690 unsigned DeviceID, FileID, Line; 2691 getTargetEntryUniqueInfo(CGM.getContext(), 2692 VD->getCanonicalDecl()->getBeginLoc(), 2693 DeviceID, FileID, Line); 2694 OS << llvm::format("_%x", FileID); 2695 } 2696 OS << "_decl_tgt_ref_ptr"; 2697 } 2698 llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName); 2699 if (!Ptr) { 2700 QualType PtrTy = CGM.getContext().getPointerType(VD->getType()); 2701 Ptr = getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(PtrTy), 2702 PtrName); 2703 2704 auto *GV = cast<llvm::GlobalVariable>(Ptr); 2705 GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 2706 2707 if (!CGM.getLangOpts().OpenMPIsDevice) 2708 GV->setInitializer(CGM.GetAddrOfGlobal(VD)); 2709 registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr)); 2710 } 2711 return Address(Ptr, CGM.getContext().getDeclAlign(VD)); 2712 } 2713 return Address::invalid(); 2714 } 2715 2716 llvm::Constant * 2717 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) { 2718 assert(!CGM.getLangOpts().OpenMPUseTLS || 2719 !CGM.getContext().getTargetInfo().isTLSSupported()); 2720 // Lookup the entry, lazily creating it if necessary. 2721 std::string Suffix = getName({"cache", ""}); 2722 return getOrCreateInternalVariable( 2723 CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix)); 2724 } 2725 2726 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 2727 const VarDecl *VD, 2728 Address VDAddr, 2729 SourceLocation Loc) { 2730 if (CGM.getLangOpts().OpenMPUseTLS && 2731 CGM.getContext().getTargetInfo().isTLSSupported()) 2732 return VDAddr; 2733 2734 llvm::Type *VarTy = VDAddr.getElementType(); 2735 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2736 CGF.Builder.CreatePointerCast(VDAddr.getPointer(), 2737 CGM.Int8PtrTy), 2738 CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)), 2739 getOrCreateThreadPrivateCache(VD)}; 2740 return Address(CGF.EmitRuntimeCall( 2741 createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args), 2742 VDAddr.getAlignment()); 2743 } 2744 2745 void CGOpenMPRuntime::emitThreadPrivateVarInit( 2746 CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor, 2747 llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) { 2748 // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime 2749 // library. 2750 llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc); 2751 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num), 2752 OMPLoc); 2753 // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor) 2754 // to register constructor/destructor for variable. 2755 llvm::Value *Args[] = { 2756 OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy), 2757 Ctor, CopyCtor, Dtor}; 2758 CGF.EmitRuntimeCall( 2759 createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args); 2760 } 2761 2762 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition( 2763 const VarDecl *VD, Address VDAddr, SourceLocation Loc, 2764 bool PerformInit, CodeGenFunction *CGF) { 2765 if (CGM.getLangOpts().OpenMPUseTLS && 2766 CGM.getContext().getTargetInfo().isTLSSupported()) 2767 return nullptr; 2768 2769 VD = VD->getDefinition(CGM.getContext()); 2770 if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) { 2771 QualType ASTTy = VD->getType(); 2772 2773 llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr; 2774 const Expr *Init = VD->getAnyInitializer(); 2775 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 2776 // Generate function that re-emits the declaration's initializer into the 2777 // threadprivate copy of the variable VD 2778 CodeGenFunction CtorCGF(CGM); 2779 FunctionArgList Args; 2780 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 2781 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 2782 ImplicitParamDecl::Other); 2783 Args.push_back(&Dst); 2784 2785 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 2786 CGM.getContext().VoidPtrTy, Args); 2787 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2788 std::string Name = getName({"__kmpc_global_ctor_", ""}); 2789 llvm::Function *Fn = 2790 CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc); 2791 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI, 2792 Args, Loc, Loc); 2793 llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar( 2794 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 2795 CGM.getContext().VoidPtrTy, Dst.getLocation()); 2796 Address Arg = Address(ArgVal, VDAddr.getAlignment()); 2797 Arg = CtorCGF.Builder.CreateElementBitCast( 2798 Arg, CtorCGF.ConvertTypeForMem(ASTTy)); 2799 CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(), 2800 /*IsInitializer=*/true); 2801 ArgVal = CtorCGF.EmitLoadOfScalar( 2802 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 2803 CGM.getContext().VoidPtrTy, Dst.getLocation()); 2804 CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue); 2805 CtorCGF.FinishFunction(); 2806 Ctor = Fn; 2807 } 2808 if (VD->getType().isDestructedType() != QualType::DK_none) { 2809 // Generate function that emits destructor call for the threadprivate copy 2810 // of the variable VD 2811 CodeGenFunction DtorCGF(CGM); 2812 FunctionArgList Args; 2813 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 2814 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 2815 ImplicitParamDecl::Other); 2816 Args.push_back(&Dst); 2817 2818 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 2819 CGM.getContext().VoidTy, Args); 2820 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2821 std::string Name = getName({"__kmpc_global_dtor_", ""}); 2822 llvm::Function *Fn = 2823 CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc); 2824 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 2825 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args, 2826 Loc, Loc); 2827 // Create a scope with an artificial location for the body of this function. 2828 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 2829 llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar( 2830 DtorCGF.GetAddrOfLocalVar(&Dst), 2831 /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation()); 2832 DtorCGF.emitDestroy(Address(ArgVal, VDAddr.getAlignment()), ASTTy, 2833 DtorCGF.getDestroyer(ASTTy.isDestructedType()), 2834 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 2835 DtorCGF.FinishFunction(); 2836 Dtor = Fn; 2837 } 2838 // Do not emit init function if it is not required. 2839 if (!Ctor && !Dtor) 2840 return nullptr; 2841 2842 llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2843 auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs, 2844 /*isVarArg=*/false) 2845 ->getPointerTo(); 2846 // Copying constructor for the threadprivate variable. 2847 // Must be NULL - reserved by runtime, but currently it requires that this 2848 // parameter is always NULL. Otherwise it fires assertion. 2849 CopyCtor = llvm::Constant::getNullValue(CopyCtorTy); 2850 if (Ctor == nullptr) { 2851 auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 2852 /*isVarArg=*/false) 2853 ->getPointerTo(); 2854 Ctor = llvm::Constant::getNullValue(CtorTy); 2855 } 2856 if (Dtor == nullptr) { 2857 auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, 2858 /*isVarArg=*/false) 2859 ->getPointerTo(); 2860 Dtor = llvm::Constant::getNullValue(DtorTy); 2861 } 2862 if (!CGF) { 2863 auto *InitFunctionTy = 2864 llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false); 2865 std::string Name = getName({"__omp_threadprivate_init_", ""}); 2866 llvm::Function *InitFunction = CGM.CreateGlobalInitOrDestructFunction( 2867 InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction()); 2868 CodeGenFunction InitCGF(CGM); 2869 FunctionArgList ArgList; 2870 InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction, 2871 CGM.getTypes().arrangeNullaryFunction(), ArgList, 2872 Loc, Loc); 2873 emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 2874 InitCGF.FinishFunction(); 2875 return InitFunction; 2876 } 2877 emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 2878 } 2879 return nullptr; 2880 } 2881 2882 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD, 2883 llvm::GlobalVariable *Addr, 2884 bool PerformInit) { 2885 if (CGM.getLangOpts().OMPTargetTriples.empty() && 2886 !CGM.getLangOpts().OpenMPIsDevice) 2887 return false; 2888 Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 2889 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 2890 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 2891 (*Res == OMPDeclareTargetDeclAttr::MT_To && 2892 HasRequiresUnifiedSharedMemory)) 2893 return CGM.getLangOpts().OpenMPIsDevice; 2894 VD = VD->getDefinition(CGM.getContext()); 2895 assert(VD && "Unknown VarDecl"); 2896 2897 if (!DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second) 2898 return CGM.getLangOpts().OpenMPIsDevice; 2899 2900 QualType ASTTy = VD->getType(); 2901 SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc(); 2902 2903 // Produce the unique prefix to identify the new target regions. We use 2904 // the source location of the variable declaration which we know to not 2905 // conflict with any target region. 2906 unsigned DeviceID; 2907 unsigned FileID; 2908 unsigned Line; 2909 getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line); 2910 SmallString<128> Buffer, Out; 2911 { 2912 llvm::raw_svector_ostream OS(Buffer); 2913 OS << "__omp_offloading_" << llvm::format("_%x", DeviceID) 2914 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 2915 } 2916 2917 const Expr *Init = VD->getAnyInitializer(); 2918 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 2919 llvm::Constant *Ctor; 2920 llvm::Constant *ID; 2921 if (CGM.getLangOpts().OpenMPIsDevice) { 2922 // Generate function that re-emits the declaration's initializer into 2923 // the threadprivate copy of the variable VD 2924 CodeGenFunction CtorCGF(CGM); 2925 2926 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 2927 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2928 llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction( 2929 FTy, Twine(Buffer, "_ctor"), FI, Loc); 2930 auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF); 2931 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 2932 FunctionArgList(), Loc, Loc); 2933 auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF); 2934 CtorCGF.EmitAnyExprToMem(Init, 2935 Address(Addr, CGM.getContext().getDeclAlign(VD)), 2936 Init->getType().getQualifiers(), 2937 /*IsInitializer=*/true); 2938 CtorCGF.FinishFunction(); 2939 Ctor = Fn; 2940 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 2941 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor)); 2942 } else { 2943 Ctor = new llvm::GlobalVariable( 2944 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 2945 llvm::GlobalValue::PrivateLinkage, 2946 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor")); 2947 ID = Ctor; 2948 } 2949 2950 // Register the information for the entry associated with the constructor. 2951 Out.clear(); 2952 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 2953 DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor, 2954 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor); 2955 } 2956 if (VD->getType().isDestructedType() != QualType::DK_none) { 2957 llvm::Constant *Dtor; 2958 llvm::Constant *ID; 2959 if (CGM.getLangOpts().OpenMPIsDevice) { 2960 // Generate function that emits destructor call for the threadprivate 2961 // copy of the variable VD 2962 CodeGenFunction DtorCGF(CGM); 2963 2964 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 2965 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2966 llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction( 2967 FTy, Twine(Buffer, "_dtor"), FI, Loc); 2968 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 2969 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 2970 FunctionArgList(), Loc, Loc); 2971 // Create a scope with an artificial location for the body of this 2972 // function. 2973 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 2974 DtorCGF.emitDestroy(Address(Addr, CGM.getContext().getDeclAlign(VD)), 2975 ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()), 2976 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 2977 DtorCGF.FinishFunction(); 2978 Dtor = Fn; 2979 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 2980 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor)); 2981 } else { 2982 Dtor = new llvm::GlobalVariable( 2983 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 2984 llvm::GlobalValue::PrivateLinkage, 2985 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor")); 2986 ID = Dtor; 2987 } 2988 // Register the information for the entry associated with the destructor. 2989 Out.clear(); 2990 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 2991 DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor, 2992 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor); 2993 } 2994 return CGM.getLangOpts().OpenMPIsDevice; 2995 } 2996 2997 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF, 2998 QualType VarType, 2999 StringRef Name) { 3000 std::string Suffix = getName({"artificial", ""}); 3001 llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType); 3002 llvm::Value *GAddr = 3003 getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix)); 3004 if (CGM.getLangOpts().OpenMP && CGM.getLangOpts().OpenMPUseTLS && 3005 CGM.getTarget().isTLSSupported()) { 3006 cast<llvm::GlobalVariable>(GAddr)->setThreadLocal(/*Val=*/true); 3007 return Address(GAddr, CGM.getContext().getTypeAlignInChars(VarType)); 3008 } 3009 std::string CacheSuffix = getName({"cache", ""}); 3010 llvm::Value *Args[] = { 3011 emitUpdateLocation(CGF, SourceLocation()), 3012 getThreadID(CGF, SourceLocation()), 3013 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy), 3014 CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy, 3015 /*isSigned=*/false), 3016 getOrCreateInternalVariable( 3017 CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))}; 3018 return Address( 3019 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3020 CGF.EmitRuntimeCall( 3021 createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args), 3022 VarLVType->getPointerTo(/*AddrSpace=*/0)), 3023 CGM.getContext().getTypeAlignInChars(VarType)); 3024 } 3025 3026 void CGOpenMPRuntime::emitIfClause(CodeGenFunction &CGF, const Expr *Cond, 3027 const RegionCodeGenTy &ThenGen, 3028 const RegionCodeGenTy &ElseGen) { 3029 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange()); 3030 3031 // If the condition constant folds and can be elided, try to avoid emitting 3032 // the condition and the dead arm of the if/else. 3033 bool CondConstant; 3034 if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) { 3035 if (CondConstant) 3036 ThenGen(CGF); 3037 else 3038 ElseGen(CGF); 3039 return; 3040 } 3041 3042 // Otherwise, the condition did not fold, or we couldn't elide it. Just 3043 // emit the conditional branch. 3044 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then"); 3045 llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else"); 3046 llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end"); 3047 CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0); 3048 3049 // Emit the 'then' code. 3050 CGF.EmitBlock(ThenBlock); 3051 ThenGen(CGF); 3052 CGF.EmitBranch(ContBlock); 3053 // Emit the 'else' code if present. 3054 // There is no need to emit line number for unconditional branch. 3055 (void)ApplyDebugLocation::CreateEmpty(CGF); 3056 CGF.EmitBlock(ElseBlock); 3057 ElseGen(CGF); 3058 // There is no need to emit line number for unconditional branch. 3059 (void)ApplyDebugLocation::CreateEmpty(CGF); 3060 CGF.EmitBranch(ContBlock); 3061 // Emit the continuation block for code after the if. 3062 CGF.EmitBlock(ContBlock, /*IsFinished=*/true); 3063 } 3064 3065 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc, 3066 llvm::Function *OutlinedFn, 3067 ArrayRef<llvm::Value *> CapturedVars, 3068 const Expr *IfCond) { 3069 if (!CGF.HaveInsertPoint()) 3070 return; 3071 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 3072 auto &&ThenGen = [OutlinedFn, CapturedVars, RTLoc](CodeGenFunction &CGF, 3073 PrePostActionTy &) { 3074 // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn); 3075 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 3076 llvm::Value *Args[] = { 3077 RTLoc, 3078 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 3079 CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())}; 3080 llvm::SmallVector<llvm::Value *, 16> RealArgs; 3081 RealArgs.append(std::begin(Args), std::end(Args)); 3082 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 3083 3084 llvm::FunctionCallee RTLFn = 3085 RT.createRuntimeFunction(OMPRTL__kmpc_fork_call); 3086 CGF.EmitRuntimeCall(RTLFn, RealArgs); 3087 }; 3088 auto &&ElseGen = [OutlinedFn, CapturedVars, RTLoc, Loc](CodeGenFunction &CGF, 3089 PrePostActionTy &) { 3090 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 3091 llvm::Value *ThreadID = RT.getThreadID(CGF, Loc); 3092 // Build calls: 3093 // __kmpc_serialized_parallel(&Loc, GTid); 3094 llvm::Value *Args[] = {RTLoc, ThreadID}; 3095 CGF.EmitRuntimeCall( 3096 RT.createRuntimeFunction(OMPRTL__kmpc_serialized_parallel), Args); 3097 3098 // OutlinedFn(>id, &zero_bound, CapturedStruct); 3099 Address ThreadIDAddr = RT.emitThreadIDAddress(CGF, Loc); 3100 Address ZeroAddrBound = 3101 CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty, 3102 /*Name=*/".bound.zero.addr"); 3103 CGF.InitTempAlloca(ZeroAddrBound, CGF.Builder.getInt32(/*C*/ 0)); 3104 llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs; 3105 // ThreadId for serialized parallels is 0. 3106 OutlinedFnArgs.push_back(ThreadIDAddr.getPointer()); 3107 OutlinedFnArgs.push_back(ZeroAddrBound.getPointer()); 3108 OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end()); 3109 RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs); 3110 3111 // __kmpc_end_serialized_parallel(&Loc, GTid); 3112 llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID}; 3113 CGF.EmitRuntimeCall( 3114 RT.createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel), 3115 EndArgs); 3116 }; 3117 if (IfCond) { 3118 emitIfClause(CGF, IfCond, ThenGen, ElseGen); 3119 } else { 3120 RegionCodeGenTy ThenRCG(ThenGen); 3121 ThenRCG(CGF); 3122 } 3123 } 3124 3125 // If we're inside an (outlined) parallel region, use the region info's 3126 // thread-ID variable (it is passed in a first argument of the outlined function 3127 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in 3128 // regular serial code region, get thread ID by calling kmp_int32 3129 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and 3130 // return the address of that temp. 3131 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF, 3132 SourceLocation Loc) { 3133 if (auto *OMPRegionInfo = 3134 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 3135 if (OMPRegionInfo->getThreadIDVariable()) 3136 return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(CGF); 3137 3138 llvm::Value *ThreadID = getThreadID(CGF, Loc); 3139 QualType Int32Ty = 3140 CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true); 3141 Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp."); 3142 CGF.EmitStoreOfScalar(ThreadID, 3143 CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty)); 3144 3145 return ThreadIDTemp; 3146 } 3147 3148 llvm::Constant *CGOpenMPRuntime::getOrCreateInternalVariable( 3149 llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) { 3150 SmallString<256> Buffer; 3151 llvm::raw_svector_ostream Out(Buffer); 3152 Out << Name; 3153 StringRef RuntimeName = Out.str(); 3154 auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first; 3155 if (Elem.second) { 3156 assert(Elem.second->getType()->getPointerElementType() == Ty && 3157 "OMP internal variable has different type than requested"); 3158 return &*Elem.second; 3159 } 3160 3161 return Elem.second = new llvm::GlobalVariable( 3162 CGM.getModule(), Ty, /*IsConstant*/ false, 3163 llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty), 3164 Elem.first(), /*InsertBefore=*/nullptr, 3165 llvm::GlobalValue::NotThreadLocal, AddressSpace); 3166 } 3167 3168 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) { 3169 std::string Prefix = Twine("gomp_critical_user_", CriticalName).str(); 3170 std::string Name = getName({Prefix, "var"}); 3171 return getOrCreateInternalVariable(KmpCriticalNameTy, Name); 3172 } 3173 3174 namespace { 3175 /// Common pre(post)-action for different OpenMP constructs. 3176 class CommonActionTy final : public PrePostActionTy { 3177 llvm::FunctionCallee EnterCallee; 3178 ArrayRef<llvm::Value *> EnterArgs; 3179 llvm::FunctionCallee ExitCallee; 3180 ArrayRef<llvm::Value *> ExitArgs; 3181 bool Conditional; 3182 llvm::BasicBlock *ContBlock = nullptr; 3183 3184 public: 3185 CommonActionTy(llvm::FunctionCallee EnterCallee, 3186 ArrayRef<llvm::Value *> EnterArgs, 3187 llvm::FunctionCallee ExitCallee, 3188 ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false) 3189 : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee), 3190 ExitArgs(ExitArgs), Conditional(Conditional) {} 3191 void Enter(CodeGenFunction &CGF) override { 3192 llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs); 3193 if (Conditional) { 3194 llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes); 3195 auto *ThenBlock = CGF.createBasicBlock("omp_if.then"); 3196 ContBlock = CGF.createBasicBlock("omp_if.end"); 3197 // Generate the branch (If-stmt) 3198 CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock); 3199 CGF.EmitBlock(ThenBlock); 3200 } 3201 } 3202 void Done(CodeGenFunction &CGF) { 3203 // Emit the rest of blocks/branches 3204 CGF.EmitBranch(ContBlock); 3205 CGF.EmitBlock(ContBlock, true); 3206 } 3207 void Exit(CodeGenFunction &CGF) override { 3208 CGF.EmitRuntimeCall(ExitCallee, ExitArgs); 3209 } 3210 }; 3211 } // anonymous namespace 3212 3213 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF, 3214 StringRef CriticalName, 3215 const RegionCodeGenTy &CriticalOpGen, 3216 SourceLocation Loc, const Expr *Hint) { 3217 // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]); 3218 // CriticalOpGen(); 3219 // __kmpc_end_critical(ident_t *, gtid, Lock); 3220 // Prepare arguments and build a call to __kmpc_critical 3221 if (!CGF.HaveInsertPoint()) 3222 return; 3223 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3224 getCriticalRegionLock(CriticalName)}; 3225 llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args), 3226 std::end(Args)); 3227 if (Hint) { 3228 EnterArgs.push_back(CGF.Builder.CreateIntCast( 3229 CGF.EmitScalarExpr(Hint), CGM.IntPtrTy, /*isSigned=*/false)); 3230 } 3231 CommonActionTy Action( 3232 createRuntimeFunction(Hint ? OMPRTL__kmpc_critical_with_hint 3233 : OMPRTL__kmpc_critical), 3234 EnterArgs, createRuntimeFunction(OMPRTL__kmpc_end_critical), Args); 3235 CriticalOpGen.setAction(Action); 3236 emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen); 3237 } 3238 3239 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF, 3240 const RegionCodeGenTy &MasterOpGen, 3241 SourceLocation Loc) { 3242 if (!CGF.HaveInsertPoint()) 3243 return; 3244 // if(__kmpc_master(ident_t *, gtid)) { 3245 // MasterOpGen(); 3246 // __kmpc_end_master(ident_t *, gtid); 3247 // } 3248 // Prepare arguments and build a call to __kmpc_master 3249 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3250 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_master), Args, 3251 createRuntimeFunction(OMPRTL__kmpc_end_master), Args, 3252 /*Conditional=*/true); 3253 MasterOpGen.setAction(Action); 3254 emitInlinedDirective(CGF, OMPD_master, MasterOpGen); 3255 Action.Done(CGF); 3256 } 3257 3258 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 3259 SourceLocation Loc) { 3260 if (!CGF.HaveInsertPoint()) 3261 return; 3262 llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder(); 3263 if (OMPBuilder) { 3264 OMPBuilder->CreateTaskyield(CGF.Builder); 3265 } else { 3266 // Build call __kmpc_omp_taskyield(loc, thread_id, 0); 3267 llvm::Value *Args[] = { 3268 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3269 llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)}; 3270 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield), 3271 Args); 3272 } 3273 3274 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 3275 Region->emitUntiedSwitch(CGF); 3276 } 3277 3278 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF, 3279 const RegionCodeGenTy &TaskgroupOpGen, 3280 SourceLocation Loc) { 3281 if (!CGF.HaveInsertPoint()) 3282 return; 3283 // __kmpc_taskgroup(ident_t *, gtid); 3284 // TaskgroupOpGen(); 3285 // __kmpc_end_taskgroup(ident_t *, gtid); 3286 // Prepare arguments and build a call to __kmpc_taskgroup 3287 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3288 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_taskgroup), Args, 3289 createRuntimeFunction(OMPRTL__kmpc_end_taskgroup), 3290 Args); 3291 TaskgroupOpGen.setAction(Action); 3292 emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen); 3293 } 3294 3295 /// Given an array of pointers to variables, project the address of a 3296 /// given variable. 3297 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array, 3298 unsigned Index, const VarDecl *Var) { 3299 // Pull out the pointer to the variable. 3300 Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index); 3301 llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr); 3302 3303 Address Addr = Address(Ptr, CGF.getContext().getDeclAlign(Var)); 3304 Addr = CGF.Builder.CreateElementBitCast( 3305 Addr, CGF.ConvertTypeForMem(Var->getType())); 3306 return Addr; 3307 } 3308 3309 static llvm::Value *emitCopyprivateCopyFunction( 3310 CodeGenModule &CGM, llvm::Type *ArgsType, 3311 ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs, 3312 ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps, 3313 SourceLocation Loc) { 3314 ASTContext &C = CGM.getContext(); 3315 // void copy_func(void *LHSArg, void *RHSArg); 3316 FunctionArgList Args; 3317 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 3318 ImplicitParamDecl::Other); 3319 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 3320 ImplicitParamDecl::Other); 3321 Args.push_back(&LHSArg); 3322 Args.push_back(&RHSArg); 3323 const auto &CGFI = 3324 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 3325 std::string Name = 3326 CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"}); 3327 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 3328 llvm::GlobalValue::InternalLinkage, Name, 3329 &CGM.getModule()); 3330 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 3331 Fn->setDoesNotRecurse(); 3332 CodeGenFunction CGF(CGM); 3333 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 3334 // Dest = (void*[n])(LHSArg); 3335 // Src = (void*[n])(RHSArg); 3336 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3337 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 3338 ArgsType), CGF.getPointerAlign()); 3339 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3340 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 3341 ArgsType), CGF.getPointerAlign()); 3342 // *(Type0*)Dst[0] = *(Type0*)Src[0]; 3343 // *(Type1*)Dst[1] = *(Type1*)Src[1]; 3344 // ... 3345 // *(Typen*)Dst[n] = *(Typen*)Src[n]; 3346 for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) { 3347 const auto *DestVar = 3348 cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl()); 3349 Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar); 3350 3351 const auto *SrcVar = 3352 cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl()); 3353 Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar); 3354 3355 const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl(); 3356 QualType Type = VD->getType(); 3357 CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]); 3358 } 3359 CGF.FinishFunction(); 3360 return Fn; 3361 } 3362 3363 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF, 3364 const RegionCodeGenTy &SingleOpGen, 3365 SourceLocation Loc, 3366 ArrayRef<const Expr *> CopyprivateVars, 3367 ArrayRef<const Expr *> SrcExprs, 3368 ArrayRef<const Expr *> DstExprs, 3369 ArrayRef<const Expr *> AssignmentOps) { 3370 if (!CGF.HaveInsertPoint()) 3371 return; 3372 assert(CopyprivateVars.size() == SrcExprs.size() && 3373 CopyprivateVars.size() == DstExprs.size() && 3374 CopyprivateVars.size() == AssignmentOps.size()); 3375 ASTContext &C = CGM.getContext(); 3376 // int32 did_it = 0; 3377 // if(__kmpc_single(ident_t *, gtid)) { 3378 // SingleOpGen(); 3379 // __kmpc_end_single(ident_t *, gtid); 3380 // did_it = 1; 3381 // } 3382 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 3383 // <copy_func>, did_it); 3384 3385 Address DidIt = Address::invalid(); 3386 if (!CopyprivateVars.empty()) { 3387 // int32 did_it = 0; 3388 QualType KmpInt32Ty = 3389 C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 3390 DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it"); 3391 CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt); 3392 } 3393 // Prepare arguments and build a call to __kmpc_single 3394 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3395 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_single), Args, 3396 createRuntimeFunction(OMPRTL__kmpc_end_single), Args, 3397 /*Conditional=*/true); 3398 SingleOpGen.setAction(Action); 3399 emitInlinedDirective(CGF, OMPD_single, SingleOpGen); 3400 if (DidIt.isValid()) { 3401 // did_it = 1; 3402 CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt); 3403 } 3404 Action.Done(CGF); 3405 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 3406 // <copy_func>, did_it); 3407 if (DidIt.isValid()) { 3408 llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size()); 3409 QualType CopyprivateArrayTy = C.getConstantArrayType( 3410 C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 3411 /*IndexTypeQuals=*/0); 3412 // Create a list of all private variables for copyprivate. 3413 Address CopyprivateList = 3414 CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list"); 3415 for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) { 3416 Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I); 3417 CGF.Builder.CreateStore( 3418 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3419 CGF.EmitLValue(CopyprivateVars[I]).getPointer(CGF), 3420 CGF.VoidPtrTy), 3421 Elem); 3422 } 3423 // Build function that copies private values from single region to all other 3424 // threads in the corresponding parallel region. 3425 llvm::Value *CpyFn = emitCopyprivateCopyFunction( 3426 CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(), 3427 CopyprivateVars, SrcExprs, DstExprs, AssignmentOps, Loc); 3428 llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy); 3429 Address CL = 3430 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList, 3431 CGF.VoidPtrTy); 3432 llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt); 3433 llvm::Value *Args[] = { 3434 emitUpdateLocation(CGF, Loc), // ident_t *<loc> 3435 getThreadID(CGF, Loc), // i32 <gtid> 3436 BufSize, // size_t <buf_size> 3437 CL.getPointer(), // void *<copyprivate list> 3438 CpyFn, // void (*) (void *, void *) <copy_func> 3439 DidItVal // i32 did_it 3440 }; 3441 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_copyprivate), Args); 3442 } 3443 } 3444 3445 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF, 3446 const RegionCodeGenTy &OrderedOpGen, 3447 SourceLocation Loc, bool IsThreads) { 3448 if (!CGF.HaveInsertPoint()) 3449 return; 3450 // __kmpc_ordered(ident_t *, gtid); 3451 // OrderedOpGen(); 3452 // __kmpc_end_ordered(ident_t *, gtid); 3453 // Prepare arguments and build a call to __kmpc_ordered 3454 if (IsThreads) { 3455 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3456 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_ordered), Args, 3457 createRuntimeFunction(OMPRTL__kmpc_end_ordered), 3458 Args); 3459 OrderedOpGen.setAction(Action); 3460 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 3461 return; 3462 } 3463 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 3464 } 3465 3466 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) { 3467 unsigned Flags; 3468 if (Kind == OMPD_for) 3469 Flags = OMP_IDENT_BARRIER_IMPL_FOR; 3470 else if (Kind == OMPD_sections) 3471 Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS; 3472 else if (Kind == OMPD_single) 3473 Flags = OMP_IDENT_BARRIER_IMPL_SINGLE; 3474 else if (Kind == OMPD_barrier) 3475 Flags = OMP_IDENT_BARRIER_EXPL; 3476 else 3477 Flags = OMP_IDENT_BARRIER_IMPL; 3478 return Flags; 3479 } 3480 3481 void CGOpenMPRuntime::getDefaultScheduleAndChunk( 3482 CodeGenFunction &CGF, const OMPLoopDirective &S, 3483 OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const { 3484 // Check if the loop directive is actually a doacross loop directive. In this 3485 // case choose static, 1 schedule. 3486 if (llvm::any_of( 3487 S.getClausesOfKind<OMPOrderedClause>(), 3488 [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) { 3489 ScheduleKind = OMPC_SCHEDULE_static; 3490 // Chunk size is 1 in this case. 3491 llvm::APInt ChunkSize(32, 1); 3492 ChunkExpr = IntegerLiteral::Create( 3493 CGF.getContext(), ChunkSize, 3494 CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0), 3495 SourceLocation()); 3496 } 3497 } 3498 3499 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc, 3500 OpenMPDirectiveKind Kind, bool EmitChecks, 3501 bool ForceSimpleCall) { 3502 // Check if we should use the OMPBuilder 3503 auto *OMPRegionInfo = 3504 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo); 3505 llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder(); 3506 if (OMPBuilder) { 3507 CGF.Builder.restoreIP(OMPBuilder->CreateBarrier( 3508 CGF.Builder, Kind, ForceSimpleCall, EmitChecks)); 3509 return; 3510 } 3511 3512 if (!CGF.HaveInsertPoint()) 3513 return; 3514 // Build call __kmpc_cancel_barrier(loc, thread_id); 3515 // Build call __kmpc_barrier(loc, thread_id); 3516 unsigned Flags = getDefaultFlagsForBarriers(Kind); 3517 // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc, 3518 // thread_id); 3519 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags), 3520 getThreadID(CGF, Loc)}; 3521 if (OMPRegionInfo) { 3522 if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) { 3523 llvm::Value *Result = CGF.EmitRuntimeCall( 3524 createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args); 3525 if (EmitChecks) { 3526 // if (__kmpc_cancel_barrier()) { 3527 // exit from construct; 3528 // } 3529 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 3530 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 3531 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 3532 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 3533 CGF.EmitBlock(ExitBB); 3534 // exit from construct; 3535 CodeGenFunction::JumpDest CancelDestination = 3536 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 3537 CGF.EmitBranchThroughCleanup(CancelDestination); 3538 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 3539 } 3540 return; 3541 } 3542 } 3543 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_barrier), Args); 3544 } 3545 3546 /// Map the OpenMP loop schedule to the runtime enumeration. 3547 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind, 3548 bool Chunked, bool Ordered) { 3549 switch (ScheduleKind) { 3550 case OMPC_SCHEDULE_static: 3551 return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked) 3552 : (Ordered ? OMP_ord_static : OMP_sch_static); 3553 case OMPC_SCHEDULE_dynamic: 3554 return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked; 3555 case OMPC_SCHEDULE_guided: 3556 return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked; 3557 case OMPC_SCHEDULE_runtime: 3558 return Ordered ? OMP_ord_runtime : OMP_sch_runtime; 3559 case OMPC_SCHEDULE_auto: 3560 return Ordered ? OMP_ord_auto : OMP_sch_auto; 3561 case OMPC_SCHEDULE_unknown: 3562 assert(!Chunked && "chunk was specified but schedule kind not known"); 3563 return Ordered ? OMP_ord_static : OMP_sch_static; 3564 } 3565 llvm_unreachable("Unexpected runtime schedule"); 3566 } 3567 3568 /// Map the OpenMP distribute schedule to the runtime enumeration. 3569 static OpenMPSchedType 3570 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) { 3571 // only static is allowed for dist_schedule 3572 return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static; 3573 } 3574 3575 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind, 3576 bool Chunked) const { 3577 OpenMPSchedType Schedule = 3578 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 3579 return Schedule == OMP_sch_static; 3580 } 3581 3582 bool CGOpenMPRuntime::isStaticNonchunked( 3583 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 3584 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 3585 return Schedule == OMP_dist_sch_static; 3586 } 3587 3588 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind, 3589 bool Chunked) const { 3590 OpenMPSchedType Schedule = 3591 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 3592 return Schedule == OMP_sch_static_chunked; 3593 } 3594 3595 bool CGOpenMPRuntime::isStaticChunked( 3596 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 3597 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 3598 return Schedule == OMP_dist_sch_static_chunked; 3599 } 3600 3601 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const { 3602 OpenMPSchedType Schedule = 3603 getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false); 3604 assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here"); 3605 return Schedule != OMP_sch_static; 3606 } 3607 3608 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule, 3609 OpenMPScheduleClauseModifier M1, 3610 OpenMPScheduleClauseModifier M2) { 3611 int Modifier = 0; 3612 switch (M1) { 3613 case OMPC_SCHEDULE_MODIFIER_monotonic: 3614 Modifier = OMP_sch_modifier_monotonic; 3615 break; 3616 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 3617 Modifier = OMP_sch_modifier_nonmonotonic; 3618 break; 3619 case OMPC_SCHEDULE_MODIFIER_simd: 3620 if (Schedule == OMP_sch_static_chunked) 3621 Schedule = OMP_sch_static_balanced_chunked; 3622 break; 3623 case OMPC_SCHEDULE_MODIFIER_last: 3624 case OMPC_SCHEDULE_MODIFIER_unknown: 3625 break; 3626 } 3627 switch (M2) { 3628 case OMPC_SCHEDULE_MODIFIER_monotonic: 3629 Modifier = OMP_sch_modifier_monotonic; 3630 break; 3631 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 3632 Modifier = OMP_sch_modifier_nonmonotonic; 3633 break; 3634 case OMPC_SCHEDULE_MODIFIER_simd: 3635 if (Schedule == OMP_sch_static_chunked) 3636 Schedule = OMP_sch_static_balanced_chunked; 3637 break; 3638 case OMPC_SCHEDULE_MODIFIER_last: 3639 case OMPC_SCHEDULE_MODIFIER_unknown: 3640 break; 3641 } 3642 // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription. 3643 // If the static schedule kind is specified or if the ordered clause is 3644 // specified, and if the nonmonotonic modifier is not specified, the effect is 3645 // as if the monotonic modifier is specified. Otherwise, unless the monotonic 3646 // modifier is specified, the effect is as if the nonmonotonic modifier is 3647 // specified. 3648 if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) { 3649 if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static || 3650 Schedule == OMP_sch_static_balanced_chunked || 3651 Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static || 3652 Schedule == OMP_dist_sch_static_chunked || 3653 Schedule == OMP_dist_sch_static)) 3654 Modifier = OMP_sch_modifier_nonmonotonic; 3655 } 3656 return Schedule | Modifier; 3657 } 3658 3659 void CGOpenMPRuntime::emitForDispatchInit( 3660 CodeGenFunction &CGF, SourceLocation Loc, 3661 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 3662 bool Ordered, const DispatchRTInput &DispatchValues) { 3663 if (!CGF.HaveInsertPoint()) 3664 return; 3665 OpenMPSchedType Schedule = getRuntimeSchedule( 3666 ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered); 3667 assert(Ordered || 3668 (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked && 3669 Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked && 3670 Schedule != OMP_sch_static_balanced_chunked)); 3671 // Call __kmpc_dispatch_init( 3672 // ident_t *loc, kmp_int32 tid, kmp_int32 schedule, 3673 // kmp_int[32|64] lower, kmp_int[32|64] upper, 3674 // kmp_int[32|64] stride, kmp_int[32|64] chunk); 3675 3676 // If the Chunk was not specified in the clause - use default value 1. 3677 llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk 3678 : CGF.Builder.getIntN(IVSize, 1); 3679 llvm::Value *Args[] = { 3680 emitUpdateLocation(CGF, Loc), 3681 getThreadID(CGF, Loc), 3682 CGF.Builder.getInt32(addMonoNonMonoModifier( 3683 CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type 3684 DispatchValues.LB, // Lower 3685 DispatchValues.UB, // Upper 3686 CGF.Builder.getIntN(IVSize, 1), // Stride 3687 Chunk // Chunk 3688 }; 3689 CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args); 3690 } 3691 3692 static void emitForStaticInitCall( 3693 CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId, 3694 llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule, 3695 OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2, 3696 const CGOpenMPRuntime::StaticRTInput &Values) { 3697 if (!CGF.HaveInsertPoint()) 3698 return; 3699 3700 assert(!Values.Ordered); 3701 assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked || 3702 Schedule == OMP_sch_static_balanced_chunked || 3703 Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked || 3704 Schedule == OMP_dist_sch_static || 3705 Schedule == OMP_dist_sch_static_chunked); 3706 3707 // Call __kmpc_for_static_init( 3708 // ident_t *loc, kmp_int32 tid, kmp_int32 schedtype, 3709 // kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower, 3710 // kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride, 3711 // kmp_int[32|64] incr, kmp_int[32|64] chunk); 3712 llvm::Value *Chunk = Values.Chunk; 3713 if (Chunk == nullptr) { 3714 assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static || 3715 Schedule == OMP_dist_sch_static) && 3716 "expected static non-chunked schedule"); 3717 // If the Chunk was not specified in the clause - use default value 1. 3718 Chunk = CGF.Builder.getIntN(Values.IVSize, 1); 3719 } else { 3720 assert((Schedule == OMP_sch_static_chunked || 3721 Schedule == OMP_sch_static_balanced_chunked || 3722 Schedule == OMP_ord_static_chunked || 3723 Schedule == OMP_dist_sch_static_chunked) && 3724 "expected static chunked schedule"); 3725 } 3726 llvm::Value *Args[] = { 3727 UpdateLocation, 3728 ThreadId, 3729 CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1, 3730 M2)), // Schedule type 3731 Values.IL.getPointer(), // &isLastIter 3732 Values.LB.getPointer(), // &LB 3733 Values.UB.getPointer(), // &UB 3734 Values.ST.getPointer(), // &Stride 3735 CGF.Builder.getIntN(Values.IVSize, 1), // Incr 3736 Chunk // Chunk 3737 }; 3738 CGF.EmitRuntimeCall(ForStaticInitFunction, Args); 3739 } 3740 3741 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF, 3742 SourceLocation Loc, 3743 OpenMPDirectiveKind DKind, 3744 const OpenMPScheduleTy &ScheduleKind, 3745 const StaticRTInput &Values) { 3746 OpenMPSchedType ScheduleNum = getRuntimeSchedule( 3747 ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered); 3748 assert(isOpenMPWorksharingDirective(DKind) && 3749 "Expected loop-based or sections-based directive."); 3750 llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc, 3751 isOpenMPLoopDirective(DKind) 3752 ? OMP_IDENT_WORK_LOOP 3753 : OMP_IDENT_WORK_SECTIONS); 3754 llvm::Value *ThreadId = getThreadID(CGF, Loc); 3755 llvm::FunctionCallee StaticInitFunction = 3756 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 3757 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 3758 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 3759 ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values); 3760 } 3761 3762 void CGOpenMPRuntime::emitDistributeStaticInit( 3763 CodeGenFunction &CGF, SourceLocation Loc, 3764 OpenMPDistScheduleClauseKind SchedKind, 3765 const CGOpenMPRuntime::StaticRTInput &Values) { 3766 OpenMPSchedType ScheduleNum = 3767 getRuntimeSchedule(SchedKind, Values.Chunk != nullptr); 3768 llvm::Value *UpdatedLocation = 3769 emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE); 3770 llvm::Value *ThreadId = getThreadID(CGF, Loc); 3771 llvm::FunctionCallee StaticInitFunction = 3772 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 3773 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 3774 ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown, 3775 OMPC_SCHEDULE_MODIFIER_unknown, Values); 3776 } 3777 3778 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF, 3779 SourceLocation Loc, 3780 OpenMPDirectiveKind DKind) { 3781 if (!CGF.HaveInsertPoint()) 3782 return; 3783 // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid); 3784 llvm::Value *Args[] = { 3785 emitUpdateLocation(CGF, Loc, 3786 isOpenMPDistributeDirective(DKind) 3787 ? OMP_IDENT_WORK_DISTRIBUTE 3788 : isOpenMPLoopDirective(DKind) 3789 ? OMP_IDENT_WORK_LOOP 3790 : OMP_IDENT_WORK_SECTIONS), 3791 getThreadID(CGF, Loc)}; 3792 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 3793 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini), 3794 Args); 3795 } 3796 3797 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 3798 SourceLocation Loc, 3799 unsigned IVSize, 3800 bool IVSigned) { 3801 if (!CGF.HaveInsertPoint()) 3802 return; 3803 // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid); 3804 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3805 CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args); 3806 } 3807 3808 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF, 3809 SourceLocation Loc, unsigned IVSize, 3810 bool IVSigned, Address IL, 3811 Address LB, Address UB, 3812 Address ST) { 3813 // Call __kmpc_dispatch_next( 3814 // ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter, 3815 // kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper, 3816 // kmp_int[32|64] *p_stride); 3817 llvm::Value *Args[] = { 3818 emitUpdateLocation(CGF, Loc), 3819 getThreadID(CGF, Loc), 3820 IL.getPointer(), // &isLastIter 3821 LB.getPointer(), // &Lower 3822 UB.getPointer(), // &Upper 3823 ST.getPointer() // &Stride 3824 }; 3825 llvm::Value *Call = 3826 CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args); 3827 return CGF.EmitScalarConversion( 3828 Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1), 3829 CGF.getContext().BoolTy, Loc); 3830 } 3831 3832 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 3833 llvm::Value *NumThreads, 3834 SourceLocation Loc) { 3835 if (!CGF.HaveInsertPoint()) 3836 return; 3837 // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads) 3838 llvm::Value *Args[] = { 3839 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3840 CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)}; 3841 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads), 3842 Args); 3843 } 3844 3845 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF, 3846 ProcBindKind ProcBind, 3847 SourceLocation Loc) { 3848 if (!CGF.HaveInsertPoint()) 3849 return; 3850 assert(ProcBind != OMP_PROC_BIND_unknown && "Unsupported proc_bind value."); 3851 // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind) 3852 llvm::Value *Args[] = { 3853 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3854 llvm::ConstantInt::get(CGM.IntTy, unsigned(ProcBind), /*isSigned=*/true)}; 3855 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_proc_bind), Args); 3856 } 3857 3858 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>, 3859 SourceLocation Loc, llvm::AtomicOrdering AO) { 3860 llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder(); 3861 if (OMPBuilder) { 3862 OMPBuilder->CreateFlush(CGF.Builder); 3863 } else { 3864 if (!CGF.HaveInsertPoint()) 3865 return; 3866 // Build call void __kmpc_flush(ident_t *loc) 3867 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush), 3868 emitUpdateLocation(CGF, Loc)); 3869 } 3870 } 3871 3872 namespace { 3873 /// Indexes of fields for type kmp_task_t. 3874 enum KmpTaskTFields { 3875 /// List of shared variables. 3876 KmpTaskTShareds, 3877 /// Task routine. 3878 KmpTaskTRoutine, 3879 /// Partition id for the untied tasks. 3880 KmpTaskTPartId, 3881 /// Function with call of destructors for private variables. 3882 Data1, 3883 /// Task priority. 3884 Data2, 3885 /// (Taskloops only) Lower bound. 3886 KmpTaskTLowerBound, 3887 /// (Taskloops only) Upper bound. 3888 KmpTaskTUpperBound, 3889 /// (Taskloops only) Stride. 3890 KmpTaskTStride, 3891 /// (Taskloops only) Is last iteration flag. 3892 KmpTaskTLastIter, 3893 /// (Taskloops only) Reduction data. 3894 KmpTaskTReductions, 3895 }; 3896 } // anonymous namespace 3897 3898 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const { 3899 return OffloadEntriesTargetRegion.empty() && 3900 OffloadEntriesDeviceGlobalVar.empty(); 3901 } 3902 3903 /// Initialize target region entry. 3904 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3905 initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3906 StringRef ParentName, unsigned LineNum, 3907 unsigned Order) { 3908 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3909 "only required for the device " 3910 "code generation."); 3911 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = 3912 OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr, 3913 OMPTargetRegionEntryTargetRegion); 3914 ++OffloadingEntriesNum; 3915 } 3916 3917 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3918 registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3919 StringRef ParentName, unsigned LineNum, 3920 llvm::Constant *Addr, llvm::Constant *ID, 3921 OMPTargetRegionEntryKind Flags) { 3922 // If we are emitting code for a target, the entry is already initialized, 3923 // only has to be registered. 3924 if (CGM.getLangOpts().OpenMPIsDevice) { 3925 if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) { 3926 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3927 DiagnosticsEngine::Error, 3928 "Unable to find target region on line '%0' in the device code."); 3929 CGM.getDiags().Report(DiagID) << LineNum; 3930 return; 3931 } 3932 auto &Entry = 3933 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum]; 3934 assert(Entry.isValid() && "Entry not initialized!"); 3935 Entry.setAddress(Addr); 3936 Entry.setID(ID); 3937 Entry.setFlags(Flags); 3938 } else { 3939 OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags); 3940 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry; 3941 ++OffloadingEntriesNum; 3942 } 3943 } 3944 3945 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo( 3946 unsigned DeviceID, unsigned FileID, StringRef ParentName, 3947 unsigned LineNum) const { 3948 auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID); 3949 if (PerDevice == OffloadEntriesTargetRegion.end()) 3950 return false; 3951 auto PerFile = PerDevice->second.find(FileID); 3952 if (PerFile == PerDevice->second.end()) 3953 return false; 3954 auto PerParentName = PerFile->second.find(ParentName); 3955 if (PerParentName == PerFile->second.end()) 3956 return false; 3957 auto PerLine = PerParentName->second.find(LineNum); 3958 if (PerLine == PerParentName->second.end()) 3959 return false; 3960 // Fail if this entry is already registered. 3961 if (PerLine->second.getAddress() || PerLine->second.getID()) 3962 return false; 3963 return true; 3964 } 3965 3966 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo( 3967 const OffloadTargetRegionEntryInfoActTy &Action) { 3968 // Scan all target region entries and perform the provided action. 3969 for (const auto &D : OffloadEntriesTargetRegion) 3970 for (const auto &F : D.second) 3971 for (const auto &P : F.second) 3972 for (const auto &L : P.second) 3973 Action(D.first, F.first, P.first(), L.first, L.second); 3974 } 3975 3976 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3977 initializeDeviceGlobalVarEntryInfo(StringRef Name, 3978 OMPTargetGlobalVarEntryKind Flags, 3979 unsigned Order) { 3980 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3981 "only required for the device " 3982 "code generation."); 3983 OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags); 3984 ++OffloadingEntriesNum; 3985 } 3986 3987 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3988 registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr, 3989 CharUnits VarSize, 3990 OMPTargetGlobalVarEntryKind Flags, 3991 llvm::GlobalValue::LinkageTypes Linkage) { 3992 if (CGM.getLangOpts().OpenMPIsDevice) { 3993 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 3994 assert(Entry.isValid() && Entry.getFlags() == Flags && 3995 "Entry not initialized!"); 3996 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 3997 "Resetting with the new address."); 3998 if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) { 3999 if (Entry.getVarSize().isZero()) { 4000 Entry.setVarSize(VarSize); 4001 Entry.setLinkage(Linkage); 4002 } 4003 return; 4004 } 4005 Entry.setVarSize(VarSize); 4006 Entry.setLinkage(Linkage); 4007 Entry.setAddress(Addr); 4008 } else { 4009 if (hasDeviceGlobalVarEntryInfo(VarName)) { 4010 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 4011 assert(Entry.isValid() && Entry.getFlags() == Flags && 4012 "Entry not initialized!"); 4013 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 4014 "Resetting with the new address."); 4015 if (Entry.getVarSize().isZero()) { 4016 Entry.setVarSize(VarSize); 4017 Entry.setLinkage(Linkage); 4018 } 4019 return; 4020 } 4021 OffloadEntriesDeviceGlobalVar.try_emplace( 4022 VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage); 4023 ++OffloadingEntriesNum; 4024 } 4025 } 4026 4027 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 4028 actOnDeviceGlobalVarEntriesInfo( 4029 const OffloadDeviceGlobalVarEntryInfoActTy &Action) { 4030 // Scan all target region entries and perform the provided action. 4031 for (const auto &E : OffloadEntriesDeviceGlobalVar) 4032 Action(E.getKey(), E.getValue()); 4033 } 4034 4035 void CGOpenMPRuntime::createOffloadEntry( 4036 llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags, 4037 llvm::GlobalValue::LinkageTypes Linkage) { 4038 StringRef Name = Addr->getName(); 4039 llvm::Module &M = CGM.getModule(); 4040 llvm::LLVMContext &C = M.getContext(); 4041 4042 // Create constant string with the name. 4043 llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name); 4044 4045 std::string StringName = getName({"omp_offloading", "entry_name"}); 4046 auto *Str = new llvm::GlobalVariable( 4047 M, StrPtrInit->getType(), /*isConstant=*/true, 4048 llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName); 4049 Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 4050 4051 llvm::Constant *Data[] = {llvm::ConstantExpr::getBitCast(ID, CGM.VoidPtrTy), 4052 llvm::ConstantExpr::getBitCast(Str, CGM.Int8PtrTy), 4053 llvm::ConstantInt::get(CGM.SizeTy, Size), 4054 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 4055 llvm::ConstantInt::get(CGM.Int32Ty, 0)}; 4056 std::string EntryName = getName({"omp_offloading", "entry", ""}); 4057 llvm::GlobalVariable *Entry = createGlobalStruct( 4058 CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data, 4059 Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage); 4060 4061 // The entry has to be created in the section the linker expects it to be. 4062 Entry->setSection("omp_offloading_entries"); 4063 } 4064 4065 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() { 4066 // Emit the offloading entries and metadata so that the device codegen side 4067 // can easily figure out what to emit. The produced metadata looks like 4068 // this: 4069 // 4070 // !omp_offload.info = !{!1, ...} 4071 // 4072 // Right now we only generate metadata for function that contain target 4073 // regions. 4074 4075 // If we are in simd mode or there are no entries, we don't need to do 4076 // anything. 4077 if (CGM.getLangOpts().OpenMPSimd || OffloadEntriesInfoManager.empty()) 4078 return; 4079 4080 llvm::Module &M = CGM.getModule(); 4081 llvm::LLVMContext &C = M.getContext(); 4082 SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *, 4083 SourceLocation, StringRef>, 4084 16> 4085 OrderedEntries(OffloadEntriesInfoManager.size()); 4086 llvm::SmallVector<StringRef, 16> ParentFunctions( 4087 OffloadEntriesInfoManager.size()); 4088 4089 // Auxiliary methods to create metadata values and strings. 4090 auto &&GetMDInt = [this](unsigned V) { 4091 return llvm::ConstantAsMetadata::get( 4092 llvm::ConstantInt::get(CGM.Int32Ty, V)); 4093 }; 4094 4095 auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); }; 4096 4097 // Create the offloading info metadata node. 4098 llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info"); 4099 4100 // Create function that emits metadata for each target region entry; 4101 auto &&TargetRegionMetadataEmitter = 4102 [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt, 4103 &GetMDString]( 4104 unsigned DeviceID, unsigned FileID, StringRef ParentName, 4105 unsigned Line, 4106 const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) { 4107 // Generate metadata for target regions. Each entry of this metadata 4108 // contains: 4109 // - Entry 0 -> Kind of this type of metadata (0). 4110 // - Entry 1 -> Device ID of the file where the entry was identified. 4111 // - Entry 2 -> File ID of the file where the entry was identified. 4112 // - Entry 3 -> Mangled name of the function where the entry was 4113 // identified. 4114 // - Entry 4 -> Line in the file where the entry was identified. 4115 // - Entry 5 -> Order the entry was created. 4116 // The first element of the metadata node is the kind. 4117 llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID), 4118 GetMDInt(FileID), GetMDString(ParentName), 4119 GetMDInt(Line), GetMDInt(E.getOrder())}; 4120 4121 SourceLocation Loc; 4122 for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(), 4123 E = CGM.getContext().getSourceManager().fileinfo_end(); 4124 I != E; ++I) { 4125 if (I->getFirst()->getUniqueID().getDevice() == DeviceID && 4126 I->getFirst()->getUniqueID().getFile() == FileID) { 4127 Loc = CGM.getContext().getSourceManager().translateFileLineCol( 4128 I->getFirst(), Line, 1); 4129 break; 4130 } 4131 } 4132 // Save this entry in the right position of the ordered entries array. 4133 OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName); 4134 ParentFunctions[E.getOrder()] = ParentName; 4135 4136 // Add metadata to the named metadata node. 4137 MD->addOperand(llvm::MDNode::get(C, Ops)); 4138 }; 4139 4140 OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo( 4141 TargetRegionMetadataEmitter); 4142 4143 // Create function that emits metadata for each device global variable entry; 4144 auto &&DeviceGlobalVarMetadataEmitter = 4145 [&C, &OrderedEntries, &GetMDInt, &GetMDString, 4146 MD](StringRef MangledName, 4147 const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar 4148 &E) { 4149 // Generate metadata for global variables. Each entry of this metadata 4150 // contains: 4151 // - Entry 0 -> Kind of this type of metadata (1). 4152 // - Entry 1 -> Mangled name of the variable. 4153 // - Entry 2 -> Declare target kind. 4154 // - Entry 3 -> Order the entry was created. 4155 // The first element of the metadata node is the kind. 4156 llvm::Metadata *Ops[] = { 4157 GetMDInt(E.getKind()), GetMDString(MangledName), 4158 GetMDInt(E.getFlags()), GetMDInt(E.getOrder())}; 4159 4160 // Save this entry in the right position of the ordered entries array. 4161 OrderedEntries[E.getOrder()] = 4162 std::make_tuple(&E, SourceLocation(), MangledName); 4163 4164 // Add metadata to the named metadata node. 4165 MD->addOperand(llvm::MDNode::get(C, Ops)); 4166 }; 4167 4168 OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo( 4169 DeviceGlobalVarMetadataEmitter); 4170 4171 for (const auto &E : OrderedEntries) { 4172 assert(std::get<0>(E) && "All ordered entries must exist!"); 4173 if (const auto *CE = 4174 dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>( 4175 std::get<0>(E))) { 4176 if (!CE->getID() || !CE->getAddress()) { 4177 // Do not blame the entry if the parent funtion is not emitted. 4178 StringRef FnName = ParentFunctions[CE->getOrder()]; 4179 if (!CGM.GetGlobalValue(FnName)) 4180 continue; 4181 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4182 DiagnosticsEngine::Error, 4183 "Offloading entry for target region in %0 is incorrect: either the " 4184 "address or the ID is invalid."); 4185 CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName; 4186 continue; 4187 } 4188 createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0, 4189 CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage); 4190 } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy:: 4191 OffloadEntryInfoDeviceGlobalVar>( 4192 std::get<0>(E))) { 4193 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags = 4194 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 4195 CE->getFlags()); 4196 switch (Flags) { 4197 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: { 4198 if (CGM.getLangOpts().OpenMPIsDevice && 4199 CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory()) 4200 continue; 4201 if (!CE->getAddress()) { 4202 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4203 DiagnosticsEngine::Error, "Offloading entry for declare target " 4204 "variable %0 is incorrect: the " 4205 "address is invalid."); 4206 CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E); 4207 continue; 4208 } 4209 // The vaiable has no definition - no need to add the entry. 4210 if (CE->getVarSize().isZero()) 4211 continue; 4212 break; 4213 } 4214 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink: 4215 assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) || 4216 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) && 4217 "Declaret target link address is set."); 4218 if (CGM.getLangOpts().OpenMPIsDevice) 4219 continue; 4220 if (!CE->getAddress()) { 4221 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4222 DiagnosticsEngine::Error, 4223 "Offloading entry for declare target variable is incorrect: the " 4224 "address is invalid."); 4225 CGM.getDiags().Report(DiagID); 4226 continue; 4227 } 4228 break; 4229 } 4230 createOffloadEntry(CE->getAddress(), CE->getAddress(), 4231 CE->getVarSize().getQuantity(), Flags, 4232 CE->getLinkage()); 4233 } else { 4234 llvm_unreachable("Unsupported entry kind."); 4235 } 4236 } 4237 } 4238 4239 /// Loads all the offload entries information from the host IR 4240 /// metadata. 4241 void CGOpenMPRuntime::loadOffloadInfoMetadata() { 4242 // If we are in target mode, load the metadata from the host IR. This code has 4243 // to match the metadaata creation in createOffloadEntriesAndInfoMetadata(). 4244 4245 if (!CGM.getLangOpts().OpenMPIsDevice) 4246 return; 4247 4248 if (CGM.getLangOpts().OMPHostIRFile.empty()) 4249 return; 4250 4251 auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile); 4252 if (auto EC = Buf.getError()) { 4253 CGM.getDiags().Report(diag::err_cannot_open_file) 4254 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 4255 return; 4256 } 4257 4258 llvm::LLVMContext C; 4259 auto ME = expectedToErrorOrAndEmitErrors( 4260 C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C)); 4261 4262 if (auto EC = ME.getError()) { 4263 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4264 DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'"); 4265 CGM.getDiags().Report(DiagID) 4266 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 4267 return; 4268 } 4269 4270 llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info"); 4271 if (!MD) 4272 return; 4273 4274 for (llvm::MDNode *MN : MD->operands()) { 4275 auto &&GetMDInt = [MN](unsigned Idx) { 4276 auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx)); 4277 return cast<llvm::ConstantInt>(V->getValue())->getZExtValue(); 4278 }; 4279 4280 auto &&GetMDString = [MN](unsigned Idx) { 4281 auto *V = cast<llvm::MDString>(MN->getOperand(Idx)); 4282 return V->getString(); 4283 }; 4284 4285 switch (GetMDInt(0)) { 4286 default: 4287 llvm_unreachable("Unexpected metadata!"); 4288 break; 4289 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 4290 OffloadingEntryInfoTargetRegion: 4291 OffloadEntriesInfoManager.initializeTargetRegionEntryInfo( 4292 /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2), 4293 /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4), 4294 /*Order=*/GetMDInt(5)); 4295 break; 4296 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 4297 OffloadingEntryInfoDeviceGlobalVar: 4298 OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo( 4299 /*MangledName=*/GetMDString(1), 4300 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 4301 /*Flags=*/GetMDInt(2)), 4302 /*Order=*/GetMDInt(3)); 4303 break; 4304 } 4305 } 4306 } 4307 4308 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) { 4309 if (!KmpRoutineEntryPtrTy) { 4310 // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type. 4311 ASTContext &C = CGM.getContext(); 4312 QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy}; 4313 FunctionProtoType::ExtProtoInfo EPI; 4314 KmpRoutineEntryPtrQTy = C.getPointerType( 4315 C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI)); 4316 KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy); 4317 } 4318 } 4319 4320 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() { 4321 // Make sure the type of the entry is already created. This is the type we 4322 // have to create: 4323 // struct __tgt_offload_entry{ 4324 // void *addr; // Pointer to the offload entry info. 4325 // // (function or global) 4326 // char *name; // Name of the function or global. 4327 // size_t size; // Size of the entry info (0 if it a function). 4328 // int32_t flags; // Flags associated with the entry, e.g. 'link'. 4329 // int32_t reserved; // Reserved, to use by the runtime library. 4330 // }; 4331 if (TgtOffloadEntryQTy.isNull()) { 4332 ASTContext &C = CGM.getContext(); 4333 RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry"); 4334 RD->startDefinition(); 4335 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4336 addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy)); 4337 addFieldToRecordDecl(C, RD, C.getSizeType()); 4338 addFieldToRecordDecl( 4339 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4340 addFieldToRecordDecl( 4341 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4342 RD->completeDefinition(); 4343 RD->addAttr(PackedAttr::CreateImplicit(C)); 4344 TgtOffloadEntryQTy = C.getRecordType(RD); 4345 } 4346 return TgtOffloadEntryQTy; 4347 } 4348 4349 namespace { 4350 struct PrivateHelpersTy { 4351 PrivateHelpersTy(const Expr *OriginalRef, const VarDecl *Original, 4352 const VarDecl *PrivateCopy, const VarDecl *PrivateElemInit) 4353 : OriginalRef(OriginalRef), Original(Original), PrivateCopy(PrivateCopy), 4354 PrivateElemInit(PrivateElemInit) {} 4355 const Expr *OriginalRef = nullptr; 4356 const VarDecl *Original = nullptr; 4357 const VarDecl *PrivateCopy = nullptr; 4358 const VarDecl *PrivateElemInit = nullptr; 4359 }; 4360 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy; 4361 } // anonymous namespace 4362 4363 static RecordDecl * 4364 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) { 4365 if (!Privates.empty()) { 4366 ASTContext &C = CGM.getContext(); 4367 // Build struct .kmp_privates_t. { 4368 // /* private vars */ 4369 // }; 4370 RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t"); 4371 RD->startDefinition(); 4372 for (const auto &Pair : Privates) { 4373 const VarDecl *VD = Pair.second.Original; 4374 QualType Type = VD->getType().getNonReferenceType(); 4375 FieldDecl *FD = addFieldToRecordDecl(C, RD, Type); 4376 if (VD->hasAttrs()) { 4377 for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()), 4378 E(VD->getAttrs().end()); 4379 I != E; ++I) 4380 FD->addAttr(*I); 4381 } 4382 } 4383 RD->completeDefinition(); 4384 return RD; 4385 } 4386 return nullptr; 4387 } 4388 4389 static RecordDecl * 4390 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind, 4391 QualType KmpInt32Ty, 4392 QualType KmpRoutineEntryPointerQTy) { 4393 ASTContext &C = CGM.getContext(); 4394 // Build struct kmp_task_t { 4395 // void * shareds; 4396 // kmp_routine_entry_t routine; 4397 // kmp_int32 part_id; 4398 // kmp_cmplrdata_t data1; 4399 // kmp_cmplrdata_t data2; 4400 // For taskloops additional fields: 4401 // kmp_uint64 lb; 4402 // kmp_uint64 ub; 4403 // kmp_int64 st; 4404 // kmp_int32 liter; 4405 // void * reductions; 4406 // }; 4407 RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union); 4408 UD->startDefinition(); 4409 addFieldToRecordDecl(C, UD, KmpInt32Ty); 4410 addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy); 4411 UD->completeDefinition(); 4412 QualType KmpCmplrdataTy = C.getRecordType(UD); 4413 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t"); 4414 RD->startDefinition(); 4415 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4416 addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy); 4417 addFieldToRecordDecl(C, RD, KmpInt32Ty); 4418 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 4419 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 4420 if (isOpenMPTaskLoopDirective(Kind)) { 4421 QualType KmpUInt64Ty = 4422 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0); 4423 QualType KmpInt64Ty = 4424 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 4425 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 4426 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 4427 addFieldToRecordDecl(C, RD, KmpInt64Ty); 4428 addFieldToRecordDecl(C, RD, KmpInt32Ty); 4429 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4430 } 4431 RD->completeDefinition(); 4432 return RD; 4433 } 4434 4435 static RecordDecl * 4436 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy, 4437 ArrayRef<PrivateDataTy> Privates) { 4438 ASTContext &C = CGM.getContext(); 4439 // Build struct kmp_task_t_with_privates { 4440 // kmp_task_t task_data; 4441 // .kmp_privates_t. privates; 4442 // }; 4443 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates"); 4444 RD->startDefinition(); 4445 addFieldToRecordDecl(C, RD, KmpTaskTQTy); 4446 if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates)) 4447 addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD)); 4448 RD->completeDefinition(); 4449 return RD; 4450 } 4451 4452 /// Emit a proxy function which accepts kmp_task_t as the second 4453 /// argument. 4454 /// \code 4455 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) { 4456 /// TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt, 4457 /// For taskloops: 4458 /// tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 4459 /// tt->reductions, tt->shareds); 4460 /// return 0; 4461 /// } 4462 /// \endcode 4463 static llvm::Function * 4464 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc, 4465 OpenMPDirectiveKind Kind, QualType KmpInt32Ty, 4466 QualType KmpTaskTWithPrivatesPtrQTy, 4467 QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy, 4468 QualType SharedsPtrTy, llvm::Function *TaskFunction, 4469 llvm::Value *TaskPrivatesMap) { 4470 ASTContext &C = CGM.getContext(); 4471 FunctionArgList Args; 4472 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 4473 ImplicitParamDecl::Other); 4474 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4475 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 4476 ImplicitParamDecl::Other); 4477 Args.push_back(&GtidArg); 4478 Args.push_back(&TaskTypeArg); 4479 const auto &TaskEntryFnInfo = 4480 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 4481 llvm::FunctionType *TaskEntryTy = 4482 CGM.getTypes().GetFunctionType(TaskEntryFnInfo); 4483 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""}); 4484 auto *TaskEntry = llvm::Function::Create( 4485 TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 4486 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo); 4487 TaskEntry->setDoesNotRecurse(); 4488 CodeGenFunction CGF(CGM); 4489 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args, 4490 Loc, Loc); 4491 4492 // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map, 4493 // tt, 4494 // For taskloops: 4495 // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 4496 // tt->task_data.shareds); 4497 llvm::Value *GtidParam = CGF.EmitLoadOfScalar( 4498 CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc); 4499 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4500 CGF.GetAddrOfLocalVar(&TaskTypeArg), 4501 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4502 const auto *KmpTaskTWithPrivatesQTyRD = 4503 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 4504 LValue Base = 4505 CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4506 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 4507 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 4508 LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI); 4509 llvm::Value *PartidParam = PartIdLVal.getPointer(CGF); 4510 4511 auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds); 4512 LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI); 4513 llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4514 CGF.EmitLoadOfScalar(SharedsLVal, Loc), 4515 CGF.ConvertTypeForMem(SharedsPtrTy)); 4516 4517 auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 4518 llvm::Value *PrivatesParam; 4519 if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) { 4520 LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI); 4521 PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4522 PrivatesLVal.getPointer(CGF), CGF.VoidPtrTy); 4523 } else { 4524 PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 4525 } 4526 4527 llvm::Value *CommonArgs[] = {GtidParam, PartidParam, PrivatesParam, 4528 TaskPrivatesMap, 4529 CGF.Builder 4530 .CreatePointerBitCastOrAddrSpaceCast( 4531 TDBase.getAddress(CGF), CGF.VoidPtrTy) 4532 .getPointer()}; 4533 SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs), 4534 std::end(CommonArgs)); 4535 if (isOpenMPTaskLoopDirective(Kind)) { 4536 auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound); 4537 LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI); 4538 llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc); 4539 auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound); 4540 LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI); 4541 llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc); 4542 auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride); 4543 LValue StLVal = CGF.EmitLValueForField(Base, *StFI); 4544 llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc); 4545 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 4546 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 4547 llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc); 4548 auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions); 4549 LValue RLVal = CGF.EmitLValueForField(Base, *RFI); 4550 llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc); 4551 CallArgs.push_back(LBParam); 4552 CallArgs.push_back(UBParam); 4553 CallArgs.push_back(StParam); 4554 CallArgs.push_back(LIParam); 4555 CallArgs.push_back(RParam); 4556 } 4557 CallArgs.push_back(SharedsParam); 4558 4559 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction, 4560 CallArgs); 4561 CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)), 4562 CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty)); 4563 CGF.FinishFunction(); 4564 return TaskEntry; 4565 } 4566 4567 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM, 4568 SourceLocation Loc, 4569 QualType KmpInt32Ty, 4570 QualType KmpTaskTWithPrivatesPtrQTy, 4571 QualType KmpTaskTWithPrivatesQTy) { 4572 ASTContext &C = CGM.getContext(); 4573 FunctionArgList Args; 4574 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 4575 ImplicitParamDecl::Other); 4576 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4577 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 4578 ImplicitParamDecl::Other); 4579 Args.push_back(&GtidArg); 4580 Args.push_back(&TaskTypeArg); 4581 const auto &DestructorFnInfo = 4582 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 4583 llvm::FunctionType *DestructorFnTy = 4584 CGM.getTypes().GetFunctionType(DestructorFnInfo); 4585 std::string Name = 4586 CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""}); 4587 auto *DestructorFn = 4588 llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage, 4589 Name, &CGM.getModule()); 4590 CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn, 4591 DestructorFnInfo); 4592 DestructorFn->setDoesNotRecurse(); 4593 CodeGenFunction CGF(CGM); 4594 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo, 4595 Args, Loc, Loc); 4596 4597 LValue Base = CGF.EmitLoadOfPointerLValue( 4598 CGF.GetAddrOfLocalVar(&TaskTypeArg), 4599 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4600 const auto *KmpTaskTWithPrivatesQTyRD = 4601 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 4602 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4603 Base = CGF.EmitLValueForField(Base, *FI); 4604 for (const auto *Field : 4605 cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) { 4606 if (QualType::DestructionKind DtorKind = 4607 Field->getType().isDestructedType()) { 4608 LValue FieldLValue = CGF.EmitLValueForField(Base, Field); 4609 CGF.pushDestroy(DtorKind, FieldLValue.getAddress(CGF), Field->getType()); 4610 } 4611 } 4612 CGF.FinishFunction(); 4613 return DestructorFn; 4614 } 4615 4616 /// Emit a privates mapping function for correct handling of private and 4617 /// firstprivate variables. 4618 /// \code 4619 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1> 4620 /// **noalias priv1,..., <tyn> **noalias privn) { 4621 /// *priv1 = &.privates.priv1; 4622 /// ...; 4623 /// *privn = &.privates.privn; 4624 /// } 4625 /// \endcode 4626 static llvm::Value * 4627 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc, 4628 ArrayRef<const Expr *> PrivateVars, 4629 ArrayRef<const Expr *> FirstprivateVars, 4630 ArrayRef<const Expr *> LastprivateVars, 4631 QualType PrivatesQTy, 4632 ArrayRef<PrivateDataTy> Privates) { 4633 ASTContext &C = CGM.getContext(); 4634 FunctionArgList Args; 4635 ImplicitParamDecl TaskPrivatesArg( 4636 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4637 C.getPointerType(PrivatesQTy).withConst().withRestrict(), 4638 ImplicitParamDecl::Other); 4639 Args.push_back(&TaskPrivatesArg); 4640 llvm::DenseMap<const VarDecl *, unsigned> PrivateVarsPos; 4641 unsigned Counter = 1; 4642 for (const Expr *E : PrivateVars) { 4643 Args.push_back(ImplicitParamDecl::Create( 4644 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4645 C.getPointerType(C.getPointerType(E->getType())) 4646 .withConst() 4647 .withRestrict(), 4648 ImplicitParamDecl::Other)); 4649 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4650 PrivateVarsPos[VD] = Counter; 4651 ++Counter; 4652 } 4653 for (const Expr *E : FirstprivateVars) { 4654 Args.push_back(ImplicitParamDecl::Create( 4655 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4656 C.getPointerType(C.getPointerType(E->getType())) 4657 .withConst() 4658 .withRestrict(), 4659 ImplicitParamDecl::Other)); 4660 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4661 PrivateVarsPos[VD] = Counter; 4662 ++Counter; 4663 } 4664 for (const Expr *E : LastprivateVars) { 4665 Args.push_back(ImplicitParamDecl::Create( 4666 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4667 C.getPointerType(C.getPointerType(E->getType())) 4668 .withConst() 4669 .withRestrict(), 4670 ImplicitParamDecl::Other)); 4671 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4672 PrivateVarsPos[VD] = Counter; 4673 ++Counter; 4674 } 4675 const auto &TaskPrivatesMapFnInfo = 4676 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4677 llvm::FunctionType *TaskPrivatesMapTy = 4678 CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo); 4679 std::string Name = 4680 CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""}); 4681 auto *TaskPrivatesMap = llvm::Function::Create( 4682 TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name, 4683 &CGM.getModule()); 4684 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap, 4685 TaskPrivatesMapFnInfo); 4686 if (CGM.getLangOpts().Optimize) { 4687 TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline); 4688 TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone); 4689 TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline); 4690 } 4691 CodeGenFunction CGF(CGM); 4692 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap, 4693 TaskPrivatesMapFnInfo, Args, Loc, Loc); 4694 4695 // *privi = &.privates.privi; 4696 LValue Base = CGF.EmitLoadOfPointerLValue( 4697 CGF.GetAddrOfLocalVar(&TaskPrivatesArg), 4698 TaskPrivatesArg.getType()->castAs<PointerType>()); 4699 const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl()); 4700 Counter = 0; 4701 for (const FieldDecl *Field : PrivatesQTyRD->fields()) { 4702 LValue FieldLVal = CGF.EmitLValueForField(Base, Field); 4703 const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]]; 4704 LValue RefLVal = 4705 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType()); 4706 LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue( 4707 RefLVal.getAddress(CGF), RefLVal.getType()->castAs<PointerType>()); 4708 CGF.EmitStoreOfScalar(FieldLVal.getPointer(CGF), RefLoadLVal); 4709 ++Counter; 4710 } 4711 CGF.FinishFunction(); 4712 return TaskPrivatesMap; 4713 } 4714 4715 /// Emit initialization for private variables in task-based directives. 4716 static void emitPrivatesInit(CodeGenFunction &CGF, 4717 const OMPExecutableDirective &D, 4718 Address KmpTaskSharedsPtr, LValue TDBase, 4719 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 4720 QualType SharedsTy, QualType SharedsPtrTy, 4721 const OMPTaskDataTy &Data, 4722 ArrayRef<PrivateDataTy> Privates, bool ForDup) { 4723 ASTContext &C = CGF.getContext(); 4724 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4725 LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI); 4726 OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind()) 4727 ? OMPD_taskloop 4728 : OMPD_task; 4729 const CapturedStmt &CS = *D.getCapturedStmt(Kind); 4730 CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS); 4731 LValue SrcBase; 4732 bool IsTargetTask = 4733 isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) || 4734 isOpenMPTargetExecutionDirective(D.getDirectiveKind()); 4735 // For target-based directives skip 3 firstprivate arrays BasePointersArray, 4736 // PointersArray and SizesArray. The original variables for these arrays are 4737 // not captured and we get their addresses explicitly. 4738 if ((!IsTargetTask && !Data.FirstprivateVars.empty() && ForDup) || 4739 (IsTargetTask && KmpTaskSharedsPtr.isValid())) { 4740 SrcBase = CGF.MakeAddrLValue( 4741 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4742 KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)), 4743 SharedsTy); 4744 } 4745 FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin(); 4746 for (const PrivateDataTy &Pair : Privates) { 4747 const VarDecl *VD = Pair.second.PrivateCopy; 4748 const Expr *Init = VD->getAnyInitializer(); 4749 if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) && 4750 !CGF.isTrivialInitializer(Init)))) { 4751 LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI); 4752 if (const VarDecl *Elem = Pair.second.PrivateElemInit) { 4753 const VarDecl *OriginalVD = Pair.second.Original; 4754 // Check if the variable is the target-based BasePointersArray, 4755 // PointersArray or SizesArray. 4756 LValue SharedRefLValue; 4757 QualType Type = PrivateLValue.getType(); 4758 const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD); 4759 if (IsTargetTask && !SharedField) { 4760 assert(isa<ImplicitParamDecl>(OriginalVD) && 4761 isa<CapturedDecl>(OriginalVD->getDeclContext()) && 4762 cast<CapturedDecl>(OriginalVD->getDeclContext()) 4763 ->getNumParams() == 0 && 4764 isa<TranslationUnitDecl>( 4765 cast<CapturedDecl>(OriginalVD->getDeclContext()) 4766 ->getDeclContext()) && 4767 "Expected artificial target data variable."); 4768 SharedRefLValue = 4769 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type); 4770 } else if (ForDup) { 4771 SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField); 4772 SharedRefLValue = CGF.MakeAddrLValue( 4773 Address(SharedRefLValue.getPointer(CGF), 4774 C.getDeclAlign(OriginalVD)), 4775 SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl), 4776 SharedRefLValue.getTBAAInfo()); 4777 } else { 4778 InlinedOpenMPRegionRAII Region( 4779 CGF, [](CodeGenFunction &, PrePostActionTy &) {}, OMPD_unknown, 4780 /*HasCancel=*/false); 4781 SharedRefLValue = CGF.EmitLValue(Pair.second.OriginalRef); 4782 } 4783 if (Type->isArrayType()) { 4784 // Initialize firstprivate array. 4785 if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) { 4786 // Perform simple memcpy. 4787 CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type); 4788 } else { 4789 // Initialize firstprivate array using element-by-element 4790 // initialization. 4791 CGF.EmitOMPAggregateAssign( 4792 PrivateLValue.getAddress(CGF), SharedRefLValue.getAddress(CGF), 4793 Type, 4794 [&CGF, Elem, Init, &CapturesInfo](Address DestElement, 4795 Address SrcElement) { 4796 // Clean up any temporaries needed by the initialization. 4797 CodeGenFunction::OMPPrivateScope InitScope(CGF); 4798 InitScope.addPrivate( 4799 Elem, [SrcElement]() -> Address { return SrcElement; }); 4800 (void)InitScope.Privatize(); 4801 // Emit initialization for single element. 4802 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII( 4803 CGF, &CapturesInfo); 4804 CGF.EmitAnyExprToMem(Init, DestElement, 4805 Init->getType().getQualifiers(), 4806 /*IsInitializer=*/false); 4807 }); 4808 } 4809 } else { 4810 CodeGenFunction::OMPPrivateScope InitScope(CGF); 4811 InitScope.addPrivate(Elem, [SharedRefLValue, &CGF]() -> Address { 4812 return SharedRefLValue.getAddress(CGF); 4813 }); 4814 (void)InitScope.Privatize(); 4815 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo); 4816 CGF.EmitExprAsInit(Init, VD, PrivateLValue, 4817 /*capturedByInit=*/false); 4818 } 4819 } else { 4820 CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false); 4821 } 4822 } 4823 ++FI; 4824 } 4825 } 4826 4827 /// Check if duplication function is required for taskloops. 4828 static bool checkInitIsRequired(CodeGenFunction &CGF, 4829 ArrayRef<PrivateDataTy> Privates) { 4830 bool InitRequired = false; 4831 for (const PrivateDataTy &Pair : Privates) { 4832 const VarDecl *VD = Pair.second.PrivateCopy; 4833 const Expr *Init = VD->getAnyInitializer(); 4834 InitRequired = InitRequired || (Init && isa<CXXConstructExpr>(Init) && 4835 !CGF.isTrivialInitializer(Init)); 4836 if (InitRequired) 4837 break; 4838 } 4839 return InitRequired; 4840 } 4841 4842 4843 /// Emit task_dup function (for initialization of 4844 /// private/firstprivate/lastprivate vars and last_iter flag) 4845 /// \code 4846 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int 4847 /// lastpriv) { 4848 /// // setup lastprivate flag 4849 /// task_dst->last = lastpriv; 4850 /// // could be constructor calls here... 4851 /// } 4852 /// \endcode 4853 static llvm::Value * 4854 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc, 4855 const OMPExecutableDirective &D, 4856 QualType KmpTaskTWithPrivatesPtrQTy, 4857 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 4858 const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy, 4859 QualType SharedsPtrTy, const OMPTaskDataTy &Data, 4860 ArrayRef<PrivateDataTy> Privates, bool WithLastIter) { 4861 ASTContext &C = CGM.getContext(); 4862 FunctionArgList Args; 4863 ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4864 KmpTaskTWithPrivatesPtrQTy, 4865 ImplicitParamDecl::Other); 4866 ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4867 KmpTaskTWithPrivatesPtrQTy, 4868 ImplicitParamDecl::Other); 4869 ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy, 4870 ImplicitParamDecl::Other); 4871 Args.push_back(&DstArg); 4872 Args.push_back(&SrcArg); 4873 Args.push_back(&LastprivArg); 4874 const auto &TaskDupFnInfo = 4875 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4876 llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo); 4877 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""}); 4878 auto *TaskDup = llvm::Function::Create( 4879 TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 4880 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo); 4881 TaskDup->setDoesNotRecurse(); 4882 CodeGenFunction CGF(CGM); 4883 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc, 4884 Loc); 4885 4886 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4887 CGF.GetAddrOfLocalVar(&DstArg), 4888 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4889 // task_dst->liter = lastpriv; 4890 if (WithLastIter) { 4891 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 4892 LValue Base = CGF.EmitLValueForField( 4893 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4894 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 4895 llvm::Value *Lastpriv = CGF.EmitLoadOfScalar( 4896 CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc); 4897 CGF.EmitStoreOfScalar(Lastpriv, LILVal); 4898 } 4899 4900 // Emit initial values for private copies (if any). 4901 assert(!Privates.empty()); 4902 Address KmpTaskSharedsPtr = Address::invalid(); 4903 if (!Data.FirstprivateVars.empty()) { 4904 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4905 CGF.GetAddrOfLocalVar(&SrcArg), 4906 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4907 LValue Base = CGF.EmitLValueForField( 4908 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4909 KmpTaskSharedsPtr = Address( 4910 CGF.EmitLoadOfScalar(CGF.EmitLValueForField( 4911 Base, *std::next(KmpTaskTQTyRD->field_begin(), 4912 KmpTaskTShareds)), 4913 Loc), 4914 CGF.getNaturalTypeAlignment(SharedsTy)); 4915 } 4916 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD, 4917 SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true); 4918 CGF.FinishFunction(); 4919 return TaskDup; 4920 } 4921 4922 /// Checks if destructor function is required to be generated. 4923 /// \return true if cleanups are required, false otherwise. 4924 static bool 4925 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD) { 4926 bool NeedsCleanup = false; 4927 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 4928 const auto *PrivateRD = cast<RecordDecl>(FI->getType()->getAsTagDecl()); 4929 for (const FieldDecl *FD : PrivateRD->fields()) { 4930 NeedsCleanup = NeedsCleanup || FD->getType().isDestructedType(); 4931 if (NeedsCleanup) 4932 break; 4933 } 4934 return NeedsCleanup; 4935 } 4936 4937 CGOpenMPRuntime::TaskResultTy 4938 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc, 4939 const OMPExecutableDirective &D, 4940 llvm::Function *TaskFunction, QualType SharedsTy, 4941 Address Shareds, const OMPTaskDataTy &Data) { 4942 ASTContext &C = CGM.getContext(); 4943 llvm::SmallVector<PrivateDataTy, 4> Privates; 4944 // Aggregate privates and sort them by the alignment. 4945 const auto *I = Data.PrivateCopies.begin(); 4946 for (const Expr *E : Data.PrivateVars) { 4947 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4948 Privates.emplace_back( 4949 C.getDeclAlign(VD), 4950 PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4951 /*PrivateElemInit=*/nullptr)); 4952 ++I; 4953 } 4954 I = Data.FirstprivateCopies.begin(); 4955 const auto *IElemInitRef = Data.FirstprivateInits.begin(); 4956 for (const Expr *E : Data.FirstprivateVars) { 4957 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4958 Privates.emplace_back( 4959 C.getDeclAlign(VD), 4960 PrivateHelpersTy( 4961 E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4962 cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl()))); 4963 ++I; 4964 ++IElemInitRef; 4965 } 4966 I = Data.LastprivateCopies.begin(); 4967 for (const Expr *E : Data.LastprivateVars) { 4968 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4969 Privates.emplace_back( 4970 C.getDeclAlign(VD), 4971 PrivateHelpersTy(E, VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4972 /*PrivateElemInit=*/nullptr)); 4973 ++I; 4974 } 4975 llvm::stable_sort(Privates, [](PrivateDataTy L, PrivateDataTy R) { 4976 return L.first > R.first; 4977 }); 4978 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 4979 // Build type kmp_routine_entry_t (if not built yet). 4980 emitKmpRoutineEntryT(KmpInt32Ty); 4981 // Build type kmp_task_t (if not built yet). 4982 if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) { 4983 if (SavedKmpTaskloopTQTy.isNull()) { 4984 SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl( 4985 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 4986 } 4987 KmpTaskTQTy = SavedKmpTaskloopTQTy; 4988 } else { 4989 assert((D.getDirectiveKind() == OMPD_task || 4990 isOpenMPTargetExecutionDirective(D.getDirectiveKind()) || 4991 isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) && 4992 "Expected taskloop, task or target directive"); 4993 if (SavedKmpTaskTQTy.isNull()) { 4994 SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl( 4995 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 4996 } 4997 KmpTaskTQTy = SavedKmpTaskTQTy; 4998 } 4999 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 5000 // Build particular struct kmp_task_t for the given task. 5001 const RecordDecl *KmpTaskTWithPrivatesQTyRD = 5002 createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates); 5003 QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD); 5004 QualType KmpTaskTWithPrivatesPtrQTy = 5005 C.getPointerType(KmpTaskTWithPrivatesQTy); 5006 llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy); 5007 llvm::Type *KmpTaskTWithPrivatesPtrTy = 5008 KmpTaskTWithPrivatesTy->getPointerTo(); 5009 llvm::Value *KmpTaskTWithPrivatesTySize = 5010 CGF.getTypeSize(KmpTaskTWithPrivatesQTy); 5011 QualType SharedsPtrTy = C.getPointerType(SharedsTy); 5012 5013 // Emit initial values for private copies (if any). 5014 llvm::Value *TaskPrivatesMap = nullptr; 5015 llvm::Type *TaskPrivatesMapTy = 5016 std::next(TaskFunction->arg_begin(), 3)->getType(); 5017 if (!Privates.empty()) { 5018 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 5019 TaskPrivatesMap = emitTaskPrivateMappingFunction( 5020 CGM, Loc, Data.PrivateVars, Data.FirstprivateVars, Data.LastprivateVars, 5021 FI->getType(), Privates); 5022 TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5023 TaskPrivatesMap, TaskPrivatesMapTy); 5024 } else { 5025 TaskPrivatesMap = llvm::ConstantPointerNull::get( 5026 cast<llvm::PointerType>(TaskPrivatesMapTy)); 5027 } 5028 // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid, 5029 // kmp_task_t *tt); 5030 llvm::Function *TaskEntry = emitProxyTaskFunction( 5031 CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 5032 KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction, 5033 TaskPrivatesMap); 5034 5035 // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 5036 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 5037 // kmp_routine_entry_t *task_entry); 5038 // Task flags. Format is taken from 5039 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h, 5040 // description of kmp_tasking_flags struct. 5041 enum { 5042 TiedFlag = 0x1, 5043 FinalFlag = 0x2, 5044 DestructorsFlag = 0x8, 5045 PriorityFlag = 0x20, 5046 DetachableFlag = 0x40, 5047 }; 5048 unsigned Flags = Data.Tied ? TiedFlag : 0; 5049 bool NeedsCleanup = false; 5050 if (!Privates.empty()) { 5051 NeedsCleanup = checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD); 5052 if (NeedsCleanup) 5053 Flags = Flags | DestructorsFlag; 5054 } 5055 if (Data.Priority.getInt()) 5056 Flags = Flags | PriorityFlag; 5057 if (D.hasClausesOfKind<OMPDetachClause>()) 5058 Flags = Flags | DetachableFlag; 5059 llvm::Value *TaskFlags = 5060 Data.Final.getPointer() 5061 ? CGF.Builder.CreateSelect(Data.Final.getPointer(), 5062 CGF.Builder.getInt32(FinalFlag), 5063 CGF.Builder.getInt32(/*C=*/0)) 5064 : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0); 5065 TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags)); 5066 llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy)); 5067 SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc), 5068 getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize, 5069 SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5070 TaskEntry, KmpRoutineEntryPtrTy)}; 5071 llvm::Value *NewTask; 5072 if (D.hasClausesOfKind<OMPNowaitClause>()) { 5073 // Check if we have any device clause associated with the directive. 5074 const Expr *Device = nullptr; 5075 if (auto *C = D.getSingleClause<OMPDeviceClause>()) 5076 Device = C->getDevice(); 5077 // Emit device ID if any otherwise use default value. 5078 llvm::Value *DeviceID; 5079 if (Device) 5080 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 5081 CGF.Int64Ty, /*isSigned=*/true); 5082 else 5083 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 5084 AllocArgs.push_back(DeviceID); 5085 NewTask = CGF.EmitRuntimeCall( 5086 createRuntimeFunction(OMPRTL__kmpc_omp_target_task_alloc), AllocArgs); 5087 } else { 5088 NewTask = CGF.EmitRuntimeCall( 5089 createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs); 5090 } 5091 // Emit detach clause initialization. 5092 // evt = (typeof(evt))__kmpc_task_allow_completion_event(loc, tid, 5093 // task_descriptor); 5094 if (const auto *DC = D.getSingleClause<OMPDetachClause>()) { 5095 const Expr *Evt = DC->getEventHandler()->IgnoreParenImpCasts(); 5096 LValue EvtLVal = CGF.EmitLValue(Evt); 5097 5098 // Build kmp_event_t *__kmpc_task_allow_completion_event(ident_t *loc_ref, 5099 // int gtid, kmp_task_t *task); 5100 llvm::Value *Loc = emitUpdateLocation(CGF, DC->getBeginLoc()); 5101 llvm::Value *Tid = getThreadID(CGF, DC->getBeginLoc()); 5102 Tid = CGF.Builder.CreateIntCast(Tid, CGF.IntTy, /*isSigned=*/false); 5103 llvm::Value *EvtVal = CGF.EmitRuntimeCall( 5104 createRuntimeFunction(OMPRTL__kmpc_task_allow_completion_event), 5105 {Loc, Tid, NewTask}); 5106 EvtVal = CGF.EmitScalarConversion(EvtVal, C.VoidPtrTy, Evt->getType(), 5107 Evt->getExprLoc()); 5108 CGF.EmitStoreOfScalar(EvtVal, EvtLVal); 5109 } 5110 llvm::Value *NewTaskNewTaskTTy = 5111 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5112 NewTask, KmpTaskTWithPrivatesPtrTy); 5113 LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy, 5114 KmpTaskTWithPrivatesQTy); 5115 LValue TDBase = 5116 CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin()); 5117 // Fill the data in the resulting kmp_task_t record. 5118 // Copy shareds if there are any. 5119 Address KmpTaskSharedsPtr = Address::invalid(); 5120 if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) { 5121 KmpTaskSharedsPtr = 5122 Address(CGF.EmitLoadOfScalar( 5123 CGF.EmitLValueForField( 5124 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), 5125 KmpTaskTShareds)), 5126 Loc), 5127 CGF.getNaturalTypeAlignment(SharedsTy)); 5128 LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy); 5129 LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy); 5130 CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap); 5131 } 5132 // Emit initial values for private copies (if any). 5133 TaskResultTy Result; 5134 if (!Privates.empty()) { 5135 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD, 5136 SharedsTy, SharedsPtrTy, Data, Privates, 5137 /*ForDup=*/false); 5138 if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) && 5139 (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) { 5140 Result.TaskDupFn = emitTaskDupFunction( 5141 CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD, 5142 KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates, 5143 /*WithLastIter=*/!Data.LastprivateVars.empty()); 5144 } 5145 } 5146 // Fields of union "kmp_cmplrdata_t" for destructors and priority. 5147 enum { Priority = 0, Destructors = 1 }; 5148 // Provide pointer to function with destructors for privates. 5149 auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1); 5150 const RecordDecl *KmpCmplrdataUD = 5151 (*FI)->getType()->getAsUnionType()->getDecl(); 5152 if (NeedsCleanup) { 5153 llvm::Value *DestructorFn = emitDestructorsFunction( 5154 CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 5155 KmpTaskTWithPrivatesQTy); 5156 LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI); 5157 LValue DestructorsLV = CGF.EmitLValueForField( 5158 Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors)); 5159 CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5160 DestructorFn, KmpRoutineEntryPtrTy), 5161 DestructorsLV); 5162 } 5163 // Set priority. 5164 if (Data.Priority.getInt()) { 5165 LValue Data2LV = CGF.EmitLValueForField( 5166 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2)); 5167 LValue PriorityLV = CGF.EmitLValueForField( 5168 Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority)); 5169 CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV); 5170 } 5171 Result.NewTask = NewTask; 5172 Result.TaskEntry = TaskEntry; 5173 Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy; 5174 Result.TDBase = TDBase; 5175 Result.KmpTaskTQTyRD = KmpTaskTQTyRD; 5176 return Result; 5177 } 5178 5179 namespace { 5180 /// Dependence kind for RTL. 5181 enum RTLDependenceKindTy { 5182 DepIn = 0x01, 5183 DepInOut = 0x3, 5184 DepMutexInOutSet = 0x4 5185 }; 5186 /// Fields ids in kmp_depend_info record. 5187 enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags }; 5188 } // namespace 5189 5190 /// Translates internal dependency kind into the runtime kind. 5191 static RTLDependenceKindTy translateDependencyKind(OpenMPDependClauseKind K) { 5192 RTLDependenceKindTy DepKind; 5193 switch (K) { 5194 case OMPC_DEPEND_in: 5195 DepKind = DepIn; 5196 break; 5197 // Out and InOut dependencies must use the same code. 5198 case OMPC_DEPEND_out: 5199 case OMPC_DEPEND_inout: 5200 DepKind = DepInOut; 5201 break; 5202 case OMPC_DEPEND_mutexinoutset: 5203 DepKind = DepMutexInOutSet; 5204 break; 5205 case OMPC_DEPEND_source: 5206 case OMPC_DEPEND_sink: 5207 case OMPC_DEPEND_depobj: 5208 case OMPC_DEPEND_unknown: 5209 llvm_unreachable("Unknown task dependence type"); 5210 } 5211 return DepKind; 5212 } 5213 5214 /// Builds kmp_depend_info, if it is not built yet, and builds flags type. 5215 static void getDependTypes(ASTContext &C, QualType &KmpDependInfoTy, 5216 QualType &FlagsTy) { 5217 FlagsTy = C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false); 5218 if (KmpDependInfoTy.isNull()) { 5219 RecordDecl *KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info"); 5220 KmpDependInfoRD->startDefinition(); 5221 addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType()); 5222 addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType()); 5223 addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy); 5224 KmpDependInfoRD->completeDefinition(); 5225 KmpDependInfoTy = C.getRecordType(KmpDependInfoRD); 5226 } 5227 } 5228 5229 std::pair<llvm::Value *, LValue> 5230 CGOpenMPRuntime::getDepobjElements(CodeGenFunction &CGF, LValue DepobjLVal, 5231 SourceLocation Loc) { 5232 ASTContext &C = CGM.getContext(); 5233 QualType FlagsTy; 5234 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5235 RecordDecl *KmpDependInfoRD = 5236 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5237 LValue Base = CGF.EmitLoadOfPointerLValue( 5238 DepobjLVal.getAddress(CGF), 5239 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 5240 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 5241 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5242 Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy)); 5243 Base = CGF.MakeAddrLValue(Addr, KmpDependInfoTy, Base.getBaseInfo(), 5244 Base.getTBAAInfo()); 5245 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP( 5246 Addr.getPointer(), 5247 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 5248 LValue NumDepsBase = CGF.MakeAddrLValue( 5249 Address(DepObjAddr, Addr.getAlignment()), KmpDependInfoTy, 5250 Base.getBaseInfo(), Base.getTBAAInfo()); 5251 // NumDeps = deps[i].base_addr; 5252 LValue BaseAddrLVal = CGF.EmitLValueForField( 5253 NumDepsBase, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 5254 llvm::Value *NumDeps = CGF.EmitLoadOfScalar(BaseAddrLVal, Loc); 5255 return std::make_pair(NumDeps, Base); 5256 } 5257 5258 std::pair<llvm::Value *, Address> CGOpenMPRuntime::emitDependClause( 5259 CodeGenFunction &CGF, 5260 ArrayRef<std::pair<OpenMPDependClauseKind, const Expr *>> Dependencies, 5261 bool ForDepobj, SourceLocation Loc) { 5262 // Process list of dependencies. 5263 ASTContext &C = CGM.getContext(); 5264 Address DependenciesArray = Address::invalid(); 5265 unsigned NumDependencies = Dependencies.size(); 5266 llvm::Value *NumOfElements = nullptr; 5267 if (NumDependencies) { 5268 QualType FlagsTy; 5269 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5270 RecordDecl *KmpDependInfoRD = 5271 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5272 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 5273 unsigned NumDepobjDependecies = 0; 5274 SmallVector<std::pair<llvm::Value *, LValue>, 4> Depobjs; 5275 llvm::Value *NumOfDepobjElements = llvm::ConstantInt::get(CGF.IntPtrTy, 0); 5276 // Calculate number of depobj dependecies. 5277 for (const std::pair<OpenMPDependClauseKind, const Expr *> &Pair : 5278 Dependencies) { 5279 if (Pair.first != OMPC_DEPEND_depobj) 5280 continue; 5281 LValue DepobjLVal = CGF.EmitLValue(Pair.second); 5282 llvm::Value *NumDeps; 5283 LValue Base; 5284 std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc); 5285 NumOfDepobjElements = 5286 CGF.Builder.CreateNUWAdd(NumOfDepobjElements, NumDeps); 5287 Depobjs.emplace_back(NumDeps, Base); 5288 ++NumDepobjDependecies; 5289 } 5290 5291 QualType KmpDependInfoArrayTy; 5292 // Define type kmp_depend_info[<Dependencies.size()>]; 5293 // For depobj reserve one extra element to store the number of elements. 5294 // It is required to handle depobj(x) update(in) construct. 5295 // kmp_depend_info[<Dependencies.size()>] deps; 5296 if (ForDepobj) { 5297 assert(NumDepobjDependecies == 0 && 5298 "depobj dependency kind is not expected in depobj directive."); 5299 KmpDependInfoArrayTy = C.getConstantArrayType( 5300 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies + 1), 5301 nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 5302 // Need to allocate on the dynamic memory. 5303 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5304 // Use default allocator. 5305 llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5306 CharUnits Align = C.getTypeAlignInChars(KmpDependInfoArrayTy); 5307 CharUnits Sz = C.getTypeSizeInChars(KmpDependInfoArrayTy); 5308 llvm::Value *Size = CGF.CGM.getSize(Sz.alignTo(Align)); 5309 llvm::Value *Args[] = {ThreadID, Size, Allocator}; 5310 5311 llvm::Value *Addr = CGF.EmitRuntimeCall( 5312 createRuntimeFunction(OMPRTL__kmpc_alloc), Args, ".dep.arr.addr"); 5313 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5314 Addr, CGF.ConvertTypeForMem(KmpDependInfoArrayTy)->getPointerTo()); 5315 DependenciesArray = Address(Addr, Align); 5316 NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies, 5317 /*isSigned=*/false); 5318 } else if (NumDepobjDependecies > 0) { 5319 NumOfElements = CGF.Builder.CreateNUWAdd( 5320 NumOfDepobjElements, 5321 llvm::ConstantInt::get(CGM.IntPtrTy, 5322 NumDependencies - NumDepobjDependecies, 5323 /*isSigned=*/false)); 5324 NumOfElements = CGF.Builder.CreateIntCast(NumOfElements, CGF.Int32Ty, 5325 /*isSigned=*/false); 5326 OpaqueValueExpr OVE( 5327 Loc, C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0), 5328 VK_RValue); 5329 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, 5330 RValue::get(NumOfElements)); 5331 KmpDependInfoArrayTy = 5332 C.getVariableArrayType(KmpDependInfoTy, &OVE, ArrayType::Normal, 5333 /*IndexTypeQuals=*/0, SourceRange(Loc, Loc)); 5334 // CGF.EmitVariablyModifiedType(KmpDependInfoArrayTy); 5335 // Properly emit variable-sized array. 5336 auto *PD = ImplicitParamDecl::Create(C, KmpDependInfoArrayTy, 5337 ImplicitParamDecl::Other); 5338 CGF.EmitVarDecl(*PD); 5339 DependenciesArray = CGF.GetAddrOfLocalVar(PD); 5340 } else { 5341 KmpDependInfoArrayTy = C.getConstantArrayType( 5342 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies), 5343 nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 5344 DependenciesArray = 5345 CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr"); 5346 NumOfElements = llvm::ConstantInt::get(CGM.Int32Ty, NumDependencies, 5347 /*isSigned=*/false); 5348 } 5349 if (ForDepobj) { 5350 // Write number of elements in the first element of array for depobj. 5351 llvm::Value *NumVal = 5352 llvm::ConstantInt::get(CGF.IntPtrTy, NumDependencies); 5353 LValue Base = CGF.MakeAddrLValue( 5354 CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0), 5355 KmpDependInfoTy); 5356 // deps[i].base_addr = NumDependencies; 5357 LValue BaseAddrLVal = CGF.EmitLValueForField( 5358 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 5359 CGF.EmitStoreOfScalar(NumVal, BaseAddrLVal); 5360 } 5361 unsigned Pos = ForDepobj ? 1 : 0; 5362 for (unsigned I = 0; I < NumDependencies; ++I) { 5363 if (Dependencies[I].first == OMPC_DEPEND_depobj) 5364 continue; 5365 const Expr *E = Dependencies[I].second; 5366 LValue Addr = CGF.EmitLValue(E); 5367 llvm::Value *Size; 5368 QualType Ty = E->getType(); 5369 if (const auto *ASE = 5370 dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) { 5371 LValue UpAddrLVal = 5372 CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false); 5373 llvm::Value *UpAddr = CGF.Builder.CreateConstGEP1_32( 5374 UpAddrLVal.getPointer(CGF), /*Idx0=*/1); 5375 llvm::Value *LowIntPtr = 5376 CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGM.SizeTy); 5377 llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGM.SizeTy); 5378 Size = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr); 5379 } else { 5380 Size = CGF.getTypeSize(Ty); 5381 } 5382 LValue Base; 5383 if (NumDepobjDependecies > 0) { 5384 Base = CGF.MakeAddrLValue( 5385 CGF.Builder.CreateConstGEP(DependenciesArray, Pos), 5386 KmpDependInfoTy); 5387 } else { 5388 Base = CGF.MakeAddrLValue( 5389 CGF.Builder.CreateConstArrayGEP(DependenciesArray, Pos), 5390 KmpDependInfoTy); 5391 } 5392 // deps[i].base_addr = &<Dependencies[i].second>; 5393 LValue BaseAddrLVal = CGF.EmitLValueForField( 5394 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 5395 CGF.EmitStoreOfScalar( 5396 CGF.Builder.CreatePtrToInt(Addr.getPointer(CGF), CGF.IntPtrTy), 5397 BaseAddrLVal); 5398 // deps[i].len = sizeof(<Dependencies[i].second>); 5399 LValue LenLVal = CGF.EmitLValueForField( 5400 Base, *std::next(KmpDependInfoRD->field_begin(), Len)); 5401 CGF.EmitStoreOfScalar(Size, LenLVal); 5402 // deps[i].flags = <Dependencies[i].first>; 5403 RTLDependenceKindTy DepKind = 5404 translateDependencyKind(Dependencies[I].first); 5405 LValue FlagsLVal = CGF.EmitLValueForField( 5406 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 5407 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 5408 FlagsLVal); 5409 ++Pos; 5410 } 5411 // Copy final depobj arrays. 5412 if (NumDepobjDependecies > 0) { 5413 llvm::Value *ElSize = CGF.getTypeSize(KmpDependInfoTy); 5414 Address Addr = CGF.Builder.CreateConstGEP(DependenciesArray, Pos); 5415 for (const std::pair<llvm::Value *, LValue> &Pair : Depobjs) { 5416 llvm::Value *Size = CGF.Builder.CreateNUWMul(ElSize, Pair.first); 5417 CGF.Builder.CreateMemCpy(Addr, Pair.second.getAddress(CGF), Size); 5418 Addr = 5419 Address(CGF.Builder.CreateGEP( 5420 Addr.getElementType(), Addr.getPointer(), Pair.first), 5421 DependenciesArray.getAlignment().alignmentOfArrayElement( 5422 C.getTypeSizeInChars(KmpDependInfoTy))); 5423 } 5424 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5425 DependenciesArray, CGF.VoidPtrTy); 5426 } else { 5427 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5428 CGF.Builder.CreateConstArrayGEP(DependenciesArray, ForDepobj ? 1 : 0), 5429 CGF.VoidPtrTy); 5430 } 5431 } 5432 return std::make_pair(NumOfElements, DependenciesArray); 5433 } 5434 5435 void CGOpenMPRuntime::emitDestroyClause(CodeGenFunction &CGF, LValue DepobjLVal, 5436 SourceLocation Loc) { 5437 ASTContext &C = CGM.getContext(); 5438 QualType FlagsTy; 5439 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5440 LValue Base = CGF.EmitLoadOfPointerLValue( 5441 DepobjLVal.getAddress(CGF), 5442 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 5443 QualType KmpDependInfoPtrTy = C.getPointerType(KmpDependInfoTy); 5444 Address Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5445 Base.getAddress(CGF), CGF.ConvertTypeForMem(KmpDependInfoPtrTy)); 5446 llvm::Value *DepObjAddr = CGF.Builder.CreateGEP( 5447 Addr.getPointer(), 5448 llvm::ConstantInt::get(CGF.IntPtrTy, -1, /*isSigned=*/true)); 5449 DepObjAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(DepObjAddr, 5450 CGF.VoidPtrTy); 5451 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5452 // Use default allocator. 5453 llvm::Value *Allocator = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5454 llvm::Value *Args[] = {ThreadID, DepObjAddr, Allocator}; 5455 5456 // _kmpc_free(gtid, addr, nullptr); 5457 (void)CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_free), Args); 5458 } 5459 5460 void CGOpenMPRuntime::emitUpdateClause(CodeGenFunction &CGF, LValue DepobjLVal, 5461 OpenMPDependClauseKind NewDepKind, 5462 SourceLocation Loc) { 5463 ASTContext &C = CGM.getContext(); 5464 QualType FlagsTy; 5465 getDependTypes(C, KmpDependInfoTy, FlagsTy); 5466 RecordDecl *KmpDependInfoRD = 5467 cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5468 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 5469 llvm::Value *NumDeps; 5470 LValue Base; 5471 std::tie(NumDeps, Base) = getDepobjElements(CGF, DepobjLVal, Loc); 5472 5473 Address Begin = Base.getAddress(CGF); 5474 // Cast from pointer to array type to pointer to single element. 5475 llvm::Value *End = CGF.Builder.CreateGEP(Begin.getPointer(), NumDeps); 5476 // The basic structure here is a while-do loop. 5477 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.body"); 5478 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.done"); 5479 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5480 CGF.EmitBlock(BodyBB); 5481 llvm::PHINode *ElementPHI = 5482 CGF.Builder.CreatePHI(Begin.getType(), 2, "omp.elementPast"); 5483 ElementPHI->addIncoming(Begin.getPointer(), EntryBB); 5484 Begin = Address(ElementPHI, Begin.getAlignment()); 5485 Base = CGF.MakeAddrLValue(Begin, KmpDependInfoTy, Base.getBaseInfo(), 5486 Base.getTBAAInfo()); 5487 // deps[i].flags = NewDepKind; 5488 RTLDependenceKindTy DepKind = translateDependencyKind(NewDepKind); 5489 LValue FlagsLVal = CGF.EmitLValueForField( 5490 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 5491 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 5492 FlagsLVal); 5493 5494 // Shift the address forward by one element. 5495 Address ElementNext = 5496 CGF.Builder.CreateConstGEP(Begin, /*Index=*/1, "omp.elementNext"); 5497 ElementPHI->addIncoming(ElementNext.getPointer(), 5498 CGF.Builder.GetInsertBlock()); 5499 llvm::Value *IsEmpty = 5500 CGF.Builder.CreateICmpEQ(ElementNext.getPointer(), End, "omp.isempty"); 5501 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5502 // Done. 5503 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5504 } 5505 5506 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 5507 const OMPExecutableDirective &D, 5508 llvm::Function *TaskFunction, 5509 QualType SharedsTy, Address Shareds, 5510 const Expr *IfCond, 5511 const OMPTaskDataTy &Data) { 5512 if (!CGF.HaveInsertPoint()) 5513 return; 5514 5515 TaskResultTy Result = 5516 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5517 llvm::Value *NewTask = Result.NewTask; 5518 llvm::Function *TaskEntry = Result.TaskEntry; 5519 llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy; 5520 LValue TDBase = Result.TDBase; 5521 const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD; 5522 // Process list of dependences. 5523 Address DependenciesArray = Address::invalid(); 5524 llvm::Value *NumOfElements; 5525 std::tie(NumOfElements, DependenciesArray) = 5526 emitDependClause(CGF, Data.Dependences, /*ForDepobj=*/false, Loc); 5527 5528 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5529 // libcall. 5530 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 5531 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 5532 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence 5533 // list is not empty 5534 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5535 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5536 llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask }; 5537 llvm::Value *DepTaskArgs[7]; 5538 if (!Data.Dependences.empty()) { 5539 DepTaskArgs[0] = UpLoc; 5540 DepTaskArgs[1] = ThreadID; 5541 DepTaskArgs[2] = NewTask; 5542 DepTaskArgs[3] = NumOfElements; 5543 DepTaskArgs[4] = DependenciesArray.getPointer(); 5544 DepTaskArgs[5] = CGF.Builder.getInt32(0); 5545 DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5546 } 5547 auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, &TaskArgs, 5548 &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) { 5549 if (!Data.Tied) { 5550 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 5551 LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI); 5552 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal); 5553 } 5554 if (!Data.Dependences.empty()) { 5555 CGF.EmitRuntimeCall( 5556 createRuntimeFunction(OMPRTL__kmpc_omp_task_with_deps), DepTaskArgs); 5557 } else { 5558 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), 5559 TaskArgs); 5560 } 5561 // Check if parent region is untied and build return for untied task; 5562 if (auto *Region = 5563 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 5564 Region->emitUntiedSwitch(CGF); 5565 }; 5566 5567 llvm::Value *DepWaitTaskArgs[6]; 5568 if (!Data.Dependences.empty()) { 5569 DepWaitTaskArgs[0] = UpLoc; 5570 DepWaitTaskArgs[1] = ThreadID; 5571 DepWaitTaskArgs[2] = NumOfElements; 5572 DepWaitTaskArgs[3] = DependenciesArray.getPointer(); 5573 DepWaitTaskArgs[4] = CGF.Builder.getInt32(0); 5574 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5575 } 5576 auto &&ElseCodeGen = [&TaskArgs, ThreadID, NewTaskNewTaskTTy, TaskEntry, 5577 &Data, &DepWaitTaskArgs, 5578 Loc](CodeGenFunction &CGF, PrePostActionTy &) { 5579 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5580 CodeGenFunction::RunCleanupsScope LocalScope(CGF); 5581 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 5582 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 5583 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info 5584 // is specified. 5585 if (!Data.Dependences.empty()) 5586 CGF.EmitRuntimeCall(RT.createRuntimeFunction(OMPRTL__kmpc_omp_wait_deps), 5587 DepWaitTaskArgs); 5588 // Call proxy_task_entry(gtid, new_task); 5589 auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy, 5590 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 5591 Action.Enter(CGF); 5592 llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy}; 5593 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry, 5594 OutlinedFnArgs); 5595 }; 5596 5597 // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 5598 // kmp_task_t *new_task); 5599 // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 5600 // kmp_task_t *new_task); 5601 RegionCodeGenTy RCG(CodeGen); 5602 CommonActionTy Action( 5603 RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_begin_if0), TaskArgs, 5604 RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_complete_if0), TaskArgs); 5605 RCG.setAction(Action); 5606 RCG(CGF); 5607 }; 5608 5609 if (IfCond) { 5610 emitIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen); 5611 } else { 5612 RegionCodeGenTy ThenRCG(ThenCodeGen); 5613 ThenRCG(CGF); 5614 } 5615 } 5616 5617 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc, 5618 const OMPLoopDirective &D, 5619 llvm::Function *TaskFunction, 5620 QualType SharedsTy, Address Shareds, 5621 const Expr *IfCond, 5622 const OMPTaskDataTy &Data) { 5623 if (!CGF.HaveInsertPoint()) 5624 return; 5625 TaskResultTy Result = 5626 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5627 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5628 // libcall. 5629 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 5630 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 5631 // sched, kmp_uint64 grainsize, void *task_dup); 5632 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5633 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5634 llvm::Value *IfVal; 5635 if (IfCond) { 5636 IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy, 5637 /*isSigned=*/true); 5638 } else { 5639 IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1); 5640 } 5641 5642 LValue LBLVal = CGF.EmitLValueForField( 5643 Result.TDBase, 5644 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound)); 5645 const auto *LBVar = 5646 cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl()); 5647 CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(CGF), 5648 LBLVal.getQuals(), 5649 /*IsInitializer=*/true); 5650 LValue UBLVal = CGF.EmitLValueForField( 5651 Result.TDBase, 5652 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound)); 5653 const auto *UBVar = 5654 cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl()); 5655 CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(CGF), 5656 UBLVal.getQuals(), 5657 /*IsInitializer=*/true); 5658 LValue StLVal = CGF.EmitLValueForField( 5659 Result.TDBase, 5660 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride)); 5661 const auto *StVar = 5662 cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl()); 5663 CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(CGF), 5664 StLVal.getQuals(), 5665 /*IsInitializer=*/true); 5666 // Store reductions address. 5667 LValue RedLVal = CGF.EmitLValueForField( 5668 Result.TDBase, 5669 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions)); 5670 if (Data.Reductions) { 5671 CGF.EmitStoreOfScalar(Data.Reductions, RedLVal); 5672 } else { 5673 CGF.EmitNullInitialization(RedLVal.getAddress(CGF), 5674 CGF.getContext().VoidPtrTy); 5675 } 5676 enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 }; 5677 llvm::Value *TaskArgs[] = { 5678 UpLoc, 5679 ThreadID, 5680 Result.NewTask, 5681 IfVal, 5682 LBLVal.getPointer(CGF), 5683 UBLVal.getPointer(CGF), 5684 CGF.EmitLoadOfScalar(StLVal, Loc), 5685 llvm::ConstantInt::getSigned( 5686 CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler 5687 llvm::ConstantInt::getSigned( 5688 CGF.IntTy, Data.Schedule.getPointer() 5689 ? Data.Schedule.getInt() ? NumTasks : Grainsize 5690 : NoSchedule), 5691 Data.Schedule.getPointer() 5692 ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty, 5693 /*isSigned=*/false) 5694 : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0), 5695 Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5696 Result.TaskDupFn, CGF.VoidPtrTy) 5697 : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)}; 5698 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_taskloop), TaskArgs); 5699 } 5700 5701 /// Emit reduction operation for each element of array (required for 5702 /// array sections) LHS op = RHS. 5703 /// \param Type Type of array. 5704 /// \param LHSVar Variable on the left side of the reduction operation 5705 /// (references element of array in original variable). 5706 /// \param RHSVar Variable on the right side of the reduction operation 5707 /// (references element of array in original variable). 5708 /// \param RedOpGen Generator of reduction operation with use of LHSVar and 5709 /// RHSVar. 5710 static void EmitOMPAggregateReduction( 5711 CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar, 5712 const VarDecl *RHSVar, 5713 const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *, 5714 const Expr *, const Expr *)> &RedOpGen, 5715 const Expr *XExpr = nullptr, const Expr *EExpr = nullptr, 5716 const Expr *UpExpr = nullptr) { 5717 // Perform element-by-element initialization. 5718 QualType ElementTy; 5719 Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar); 5720 Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar); 5721 5722 // Drill down to the base element type on both arrays. 5723 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 5724 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr); 5725 5726 llvm::Value *RHSBegin = RHSAddr.getPointer(); 5727 llvm::Value *LHSBegin = LHSAddr.getPointer(); 5728 // Cast from pointer to array type to pointer to single element. 5729 llvm::Value *LHSEnd = CGF.Builder.CreateGEP(LHSBegin, NumElements); 5730 // The basic structure here is a while-do loop. 5731 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body"); 5732 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done"); 5733 llvm::Value *IsEmpty = 5734 CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty"); 5735 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5736 5737 // Enter the loop body, making that address the current address. 5738 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5739 CGF.EmitBlock(BodyBB); 5740 5741 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 5742 5743 llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI( 5744 RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast"); 5745 RHSElementPHI->addIncoming(RHSBegin, EntryBB); 5746 Address RHSElementCurrent = 5747 Address(RHSElementPHI, 5748 RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5749 5750 llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI( 5751 LHSBegin->getType(), 2, "omp.arraycpy.destElementPast"); 5752 LHSElementPHI->addIncoming(LHSBegin, EntryBB); 5753 Address LHSElementCurrent = 5754 Address(LHSElementPHI, 5755 LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5756 5757 // Emit copy. 5758 CodeGenFunction::OMPPrivateScope Scope(CGF); 5759 Scope.addPrivate(LHSVar, [=]() { return LHSElementCurrent; }); 5760 Scope.addPrivate(RHSVar, [=]() { return RHSElementCurrent; }); 5761 Scope.Privatize(); 5762 RedOpGen(CGF, XExpr, EExpr, UpExpr); 5763 Scope.ForceCleanup(); 5764 5765 // Shift the address forward by one element. 5766 llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32( 5767 LHSElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 5768 llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32( 5769 RHSElementPHI, /*Idx0=*/1, "omp.arraycpy.src.element"); 5770 // Check whether we've reached the end. 5771 llvm::Value *Done = 5772 CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done"); 5773 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 5774 LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock()); 5775 RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock()); 5776 5777 // Done. 5778 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5779 } 5780 5781 /// Emit reduction combiner. If the combiner is a simple expression emit it as 5782 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of 5783 /// UDR combiner function. 5784 static void emitReductionCombiner(CodeGenFunction &CGF, 5785 const Expr *ReductionOp) { 5786 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 5787 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 5788 if (const auto *DRE = 5789 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 5790 if (const auto *DRD = 5791 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) { 5792 std::pair<llvm::Function *, llvm::Function *> Reduction = 5793 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 5794 RValue Func = RValue::get(Reduction.first); 5795 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 5796 CGF.EmitIgnoredExpr(ReductionOp); 5797 return; 5798 } 5799 CGF.EmitIgnoredExpr(ReductionOp); 5800 } 5801 5802 llvm::Function *CGOpenMPRuntime::emitReductionFunction( 5803 SourceLocation Loc, llvm::Type *ArgsType, ArrayRef<const Expr *> Privates, 5804 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 5805 ArrayRef<const Expr *> ReductionOps) { 5806 ASTContext &C = CGM.getContext(); 5807 5808 // void reduction_func(void *LHSArg, void *RHSArg); 5809 FunctionArgList Args; 5810 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5811 ImplicitParamDecl::Other); 5812 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5813 ImplicitParamDecl::Other); 5814 Args.push_back(&LHSArg); 5815 Args.push_back(&RHSArg); 5816 const auto &CGFI = 5817 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5818 std::string Name = getName({"omp", "reduction", "reduction_func"}); 5819 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 5820 llvm::GlobalValue::InternalLinkage, Name, 5821 &CGM.getModule()); 5822 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 5823 Fn->setDoesNotRecurse(); 5824 CodeGenFunction CGF(CGM); 5825 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 5826 5827 // Dst = (void*[n])(LHSArg); 5828 // Src = (void*[n])(RHSArg); 5829 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5830 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 5831 ArgsType), CGF.getPointerAlign()); 5832 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5833 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 5834 ArgsType), CGF.getPointerAlign()); 5835 5836 // ... 5837 // *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]); 5838 // ... 5839 CodeGenFunction::OMPPrivateScope Scope(CGF); 5840 auto IPriv = Privates.begin(); 5841 unsigned Idx = 0; 5842 for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) { 5843 const auto *RHSVar = 5844 cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl()); 5845 Scope.addPrivate(RHSVar, [&CGF, RHS, Idx, RHSVar]() { 5846 return emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar); 5847 }); 5848 const auto *LHSVar = 5849 cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl()); 5850 Scope.addPrivate(LHSVar, [&CGF, LHS, Idx, LHSVar]() { 5851 return emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar); 5852 }); 5853 QualType PrivTy = (*IPriv)->getType(); 5854 if (PrivTy->isVariablyModifiedType()) { 5855 // Get array size and emit VLA type. 5856 ++Idx; 5857 Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx); 5858 llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem); 5859 const VariableArrayType *VLA = 5860 CGF.getContext().getAsVariableArrayType(PrivTy); 5861 const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr()); 5862 CodeGenFunction::OpaqueValueMapping OpaqueMap( 5863 CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy))); 5864 CGF.EmitVariablyModifiedType(PrivTy); 5865 } 5866 } 5867 Scope.Privatize(); 5868 IPriv = Privates.begin(); 5869 auto ILHS = LHSExprs.begin(); 5870 auto IRHS = RHSExprs.begin(); 5871 for (const Expr *E : ReductionOps) { 5872 if ((*IPriv)->getType()->isArrayType()) { 5873 // Emit reduction for array section. 5874 const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5875 const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5876 EmitOMPAggregateReduction( 5877 CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5878 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5879 emitReductionCombiner(CGF, E); 5880 }); 5881 } else { 5882 // Emit reduction for array subscript or single variable. 5883 emitReductionCombiner(CGF, E); 5884 } 5885 ++IPriv; 5886 ++ILHS; 5887 ++IRHS; 5888 } 5889 Scope.ForceCleanup(); 5890 CGF.FinishFunction(); 5891 return Fn; 5892 } 5893 5894 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF, 5895 const Expr *ReductionOp, 5896 const Expr *PrivateRef, 5897 const DeclRefExpr *LHS, 5898 const DeclRefExpr *RHS) { 5899 if (PrivateRef->getType()->isArrayType()) { 5900 // Emit reduction for array section. 5901 const auto *LHSVar = cast<VarDecl>(LHS->getDecl()); 5902 const auto *RHSVar = cast<VarDecl>(RHS->getDecl()); 5903 EmitOMPAggregateReduction( 5904 CGF, PrivateRef->getType(), LHSVar, RHSVar, 5905 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5906 emitReductionCombiner(CGF, ReductionOp); 5907 }); 5908 } else { 5909 // Emit reduction for array subscript or single variable. 5910 emitReductionCombiner(CGF, ReductionOp); 5911 } 5912 } 5913 5914 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc, 5915 ArrayRef<const Expr *> Privates, 5916 ArrayRef<const Expr *> LHSExprs, 5917 ArrayRef<const Expr *> RHSExprs, 5918 ArrayRef<const Expr *> ReductionOps, 5919 ReductionOptionsTy Options) { 5920 if (!CGF.HaveInsertPoint()) 5921 return; 5922 5923 bool WithNowait = Options.WithNowait; 5924 bool SimpleReduction = Options.SimpleReduction; 5925 5926 // Next code should be emitted for reduction: 5927 // 5928 // static kmp_critical_name lock = { 0 }; 5929 // 5930 // void reduce_func(void *lhs[<n>], void *rhs[<n>]) { 5931 // *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]); 5932 // ... 5933 // *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1], 5934 // *(Type<n>-1*)rhs[<n>-1]); 5935 // } 5936 // 5937 // ... 5938 // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]}; 5939 // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5940 // RedList, reduce_func, &<lock>)) { 5941 // case 1: 5942 // ... 5943 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5944 // ... 5945 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5946 // break; 5947 // case 2: 5948 // ... 5949 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5950 // ... 5951 // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);] 5952 // break; 5953 // default:; 5954 // } 5955 // 5956 // if SimpleReduction is true, only the next code is generated: 5957 // ... 5958 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5959 // ... 5960 5961 ASTContext &C = CGM.getContext(); 5962 5963 if (SimpleReduction) { 5964 CodeGenFunction::RunCleanupsScope Scope(CGF); 5965 auto IPriv = Privates.begin(); 5966 auto ILHS = LHSExprs.begin(); 5967 auto IRHS = RHSExprs.begin(); 5968 for (const Expr *E : ReductionOps) { 5969 emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5970 cast<DeclRefExpr>(*IRHS)); 5971 ++IPriv; 5972 ++ILHS; 5973 ++IRHS; 5974 } 5975 return; 5976 } 5977 5978 // 1. Build a list of reduction variables. 5979 // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]}; 5980 auto Size = RHSExprs.size(); 5981 for (const Expr *E : Privates) { 5982 if (E->getType()->isVariablyModifiedType()) 5983 // Reserve place for array size. 5984 ++Size; 5985 } 5986 llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size); 5987 QualType ReductionArrayTy = 5988 C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 5989 /*IndexTypeQuals=*/0); 5990 Address ReductionList = 5991 CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list"); 5992 auto IPriv = Privates.begin(); 5993 unsigned Idx = 0; 5994 for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) { 5995 Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 5996 CGF.Builder.CreateStore( 5997 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5998 CGF.EmitLValue(RHSExprs[I]).getPointer(CGF), CGF.VoidPtrTy), 5999 Elem); 6000 if ((*IPriv)->getType()->isVariablyModifiedType()) { 6001 // Store array size. 6002 ++Idx; 6003 Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 6004 llvm::Value *Size = CGF.Builder.CreateIntCast( 6005 CGF.getVLASize( 6006 CGF.getContext().getAsVariableArrayType((*IPriv)->getType())) 6007 .NumElts, 6008 CGF.SizeTy, /*isSigned=*/false); 6009 CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy), 6010 Elem); 6011 } 6012 } 6013 6014 // 2. Emit reduce_func(). 6015 llvm::Function *ReductionFn = emitReductionFunction( 6016 Loc, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), Privates, 6017 LHSExprs, RHSExprs, ReductionOps); 6018 6019 // 3. Create static kmp_critical_name lock = { 0 }; 6020 std::string Name = getName({"reduction"}); 6021 llvm::Value *Lock = getCriticalRegionLock(Name); 6022 6023 // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 6024 // RedList, reduce_func, &<lock>); 6025 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE); 6026 llvm::Value *ThreadId = getThreadID(CGF, Loc); 6027 llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy); 6028 llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6029 ReductionList.getPointer(), CGF.VoidPtrTy); 6030 llvm::Value *Args[] = { 6031 IdentTLoc, // ident_t *<loc> 6032 ThreadId, // i32 <gtid> 6033 CGF.Builder.getInt32(RHSExprs.size()), // i32 <n> 6034 ReductionArrayTySize, // size_type sizeof(RedList) 6035 RL, // void *RedList 6036 ReductionFn, // void (*) (void *, void *) <reduce_func> 6037 Lock // kmp_critical_name *&<lock> 6038 }; 6039 llvm::Value *Res = CGF.EmitRuntimeCall( 6040 createRuntimeFunction(WithNowait ? OMPRTL__kmpc_reduce_nowait 6041 : OMPRTL__kmpc_reduce), 6042 Args); 6043 6044 // 5. Build switch(res) 6045 llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default"); 6046 llvm::SwitchInst *SwInst = 6047 CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2); 6048 6049 // 6. Build case 1: 6050 // ... 6051 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 6052 // ... 6053 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 6054 // break; 6055 llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1"); 6056 SwInst->addCase(CGF.Builder.getInt32(1), Case1BB); 6057 CGF.EmitBlock(Case1BB); 6058 6059 // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 6060 llvm::Value *EndArgs[] = { 6061 IdentTLoc, // ident_t *<loc> 6062 ThreadId, // i32 <gtid> 6063 Lock // kmp_critical_name *&<lock> 6064 }; 6065 auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps]( 6066 CodeGenFunction &CGF, PrePostActionTy &Action) { 6067 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6068 auto IPriv = Privates.begin(); 6069 auto ILHS = LHSExprs.begin(); 6070 auto IRHS = RHSExprs.begin(); 6071 for (const Expr *E : ReductionOps) { 6072 RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 6073 cast<DeclRefExpr>(*IRHS)); 6074 ++IPriv; 6075 ++ILHS; 6076 ++IRHS; 6077 } 6078 }; 6079 RegionCodeGenTy RCG(CodeGen); 6080 CommonActionTy Action( 6081 nullptr, llvm::None, 6082 createRuntimeFunction(WithNowait ? OMPRTL__kmpc_end_reduce_nowait 6083 : OMPRTL__kmpc_end_reduce), 6084 EndArgs); 6085 RCG.setAction(Action); 6086 RCG(CGF); 6087 6088 CGF.EmitBranch(DefaultBB); 6089 6090 // 7. Build case 2: 6091 // ... 6092 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 6093 // ... 6094 // break; 6095 llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2"); 6096 SwInst->addCase(CGF.Builder.getInt32(2), Case2BB); 6097 CGF.EmitBlock(Case2BB); 6098 6099 auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps]( 6100 CodeGenFunction &CGF, PrePostActionTy &Action) { 6101 auto ILHS = LHSExprs.begin(); 6102 auto IRHS = RHSExprs.begin(); 6103 auto IPriv = Privates.begin(); 6104 for (const Expr *E : ReductionOps) { 6105 const Expr *XExpr = nullptr; 6106 const Expr *EExpr = nullptr; 6107 const Expr *UpExpr = nullptr; 6108 BinaryOperatorKind BO = BO_Comma; 6109 if (const auto *BO = dyn_cast<BinaryOperator>(E)) { 6110 if (BO->getOpcode() == BO_Assign) { 6111 XExpr = BO->getLHS(); 6112 UpExpr = BO->getRHS(); 6113 } 6114 } 6115 // Try to emit update expression as a simple atomic. 6116 const Expr *RHSExpr = UpExpr; 6117 if (RHSExpr) { 6118 // Analyze RHS part of the whole expression. 6119 if (const auto *ACO = dyn_cast<AbstractConditionalOperator>( 6120 RHSExpr->IgnoreParenImpCasts())) { 6121 // If this is a conditional operator, analyze its condition for 6122 // min/max reduction operator. 6123 RHSExpr = ACO->getCond(); 6124 } 6125 if (const auto *BORHS = 6126 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) { 6127 EExpr = BORHS->getRHS(); 6128 BO = BORHS->getOpcode(); 6129 } 6130 } 6131 if (XExpr) { 6132 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 6133 auto &&AtomicRedGen = [BO, VD, 6134 Loc](CodeGenFunction &CGF, const Expr *XExpr, 6135 const Expr *EExpr, const Expr *UpExpr) { 6136 LValue X = CGF.EmitLValue(XExpr); 6137 RValue E; 6138 if (EExpr) 6139 E = CGF.EmitAnyExpr(EExpr); 6140 CGF.EmitOMPAtomicSimpleUpdateExpr( 6141 X, E, BO, /*IsXLHSInRHSPart=*/true, 6142 llvm::AtomicOrdering::Monotonic, Loc, 6143 [&CGF, UpExpr, VD, Loc](RValue XRValue) { 6144 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 6145 PrivateScope.addPrivate( 6146 VD, [&CGF, VD, XRValue, Loc]() { 6147 Address LHSTemp = CGF.CreateMemTemp(VD->getType()); 6148 CGF.emitOMPSimpleStore( 6149 CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue, 6150 VD->getType().getNonReferenceType(), Loc); 6151 return LHSTemp; 6152 }); 6153 (void)PrivateScope.Privatize(); 6154 return CGF.EmitAnyExpr(UpExpr); 6155 }); 6156 }; 6157 if ((*IPriv)->getType()->isArrayType()) { 6158 // Emit atomic reduction for array section. 6159 const auto *RHSVar = 6160 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 6161 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar, 6162 AtomicRedGen, XExpr, EExpr, UpExpr); 6163 } else { 6164 // Emit atomic reduction for array subscript or single variable. 6165 AtomicRedGen(CGF, XExpr, EExpr, UpExpr); 6166 } 6167 } else { 6168 // Emit as a critical region. 6169 auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *, 6170 const Expr *, const Expr *) { 6171 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6172 std::string Name = RT.getName({"atomic_reduction"}); 6173 RT.emitCriticalRegion( 6174 CGF, Name, 6175 [=](CodeGenFunction &CGF, PrePostActionTy &Action) { 6176 Action.Enter(CGF); 6177 emitReductionCombiner(CGF, E); 6178 }, 6179 Loc); 6180 }; 6181 if ((*IPriv)->getType()->isArrayType()) { 6182 const auto *LHSVar = 6183 cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 6184 const auto *RHSVar = 6185 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 6186 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar, 6187 CritRedGen); 6188 } else { 6189 CritRedGen(CGF, nullptr, nullptr, nullptr); 6190 } 6191 } 6192 ++ILHS; 6193 ++IRHS; 6194 ++IPriv; 6195 } 6196 }; 6197 RegionCodeGenTy AtomicRCG(AtomicCodeGen); 6198 if (!WithNowait) { 6199 // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>); 6200 llvm::Value *EndArgs[] = { 6201 IdentTLoc, // ident_t *<loc> 6202 ThreadId, // i32 <gtid> 6203 Lock // kmp_critical_name *&<lock> 6204 }; 6205 CommonActionTy Action(nullptr, llvm::None, 6206 createRuntimeFunction(OMPRTL__kmpc_end_reduce), 6207 EndArgs); 6208 AtomicRCG.setAction(Action); 6209 AtomicRCG(CGF); 6210 } else { 6211 AtomicRCG(CGF); 6212 } 6213 6214 CGF.EmitBranch(DefaultBB); 6215 CGF.EmitBlock(DefaultBB, /*IsFinished=*/true); 6216 } 6217 6218 /// Generates unique name for artificial threadprivate variables. 6219 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>" 6220 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix, 6221 const Expr *Ref) { 6222 SmallString<256> Buffer; 6223 llvm::raw_svector_ostream Out(Buffer); 6224 const clang::DeclRefExpr *DE; 6225 const VarDecl *D = ::getBaseDecl(Ref, DE); 6226 if (!D) 6227 D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl()); 6228 D = D->getCanonicalDecl(); 6229 std::string Name = CGM.getOpenMPRuntime().getName( 6230 {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)}); 6231 Out << Prefix << Name << "_" 6232 << D->getCanonicalDecl()->getBeginLoc().getRawEncoding(); 6233 return std::string(Out.str()); 6234 } 6235 6236 /// Emits reduction initializer function: 6237 /// \code 6238 /// void @.red_init(void* %arg) { 6239 /// %0 = bitcast void* %arg to <type>* 6240 /// store <type> <init>, <type>* %0 6241 /// ret void 6242 /// } 6243 /// \endcode 6244 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM, 6245 SourceLocation Loc, 6246 ReductionCodeGen &RCG, unsigned N) { 6247 ASTContext &C = CGM.getContext(); 6248 FunctionArgList Args; 6249 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6250 ImplicitParamDecl::Other); 6251 Args.emplace_back(&Param); 6252 const auto &FnInfo = 6253 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6254 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6255 std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""}); 6256 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6257 Name, &CGM.getModule()); 6258 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6259 Fn->setDoesNotRecurse(); 6260 CodeGenFunction CGF(CGM); 6261 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6262 Address PrivateAddr = CGF.EmitLoadOfPointer( 6263 CGF.GetAddrOfLocalVar(&Param), 6264 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6265 llvm::Value *Size = nullptr; 6266 // If the size of the reduction item is non-constant, load it from global 6267 // threadprivate variable. 6268 if (RCG.getSizes(N).second) { 6269 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6270 CGF, CGM.getContext().getSizeType(), 6271 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6272 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6273 CGM.getContext().getSizeType(), Loc); 6274 } 6275 RCG.emitAggregateType(CGF, N, Size); 6276 LValue SharedLVal; 6277 // If initializer uses initializer from declare reduction construct, emit a 6278 // pointer to the address of the original reduction item (reuired by reduction 6279 // initializer) 6280 if (RCG.usesReductionInitializer(N)) { 6281 Address SharedAddr = 6282 CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6283 CGF, CGM.getContext().VoidPtrTy, 6284 generateUniqueName(CGM, "reduction", RCG.getRefExpr(N))); 6285 SharedAddr = CGF.EmitLoadOfPointer( 6286 SharedAddr, 6287 CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr()); 6288 SharedLVal = CGF.MakeAddrLValue(SharedAddr, CGM.getContext().VoidPtrTy); 6289 } else { 6290 SharedLVal = CGF.MakeNaturalAlignAddrLValue( 6291 llvm::ConstantPointerNull::get(CGM.VoidPtrTy), 6292 CGM.getContext().VoidPtrTy); 6293 } 6294 // Emit the initializer: 6295 // %0 = bitcast void* %arg to <type>* 6296 // store <type> <init>, <type>* %0 6297 RCG.emitInitialization(CGF, N, PrivateAddr, SharedLVal, 6298 [](CodeGenFunction &) { return false; }); 6299 CGF.FinishFunction(); 6300 return Fn; 6301 } 6302 6303 /// Emits reduction combiner function: 6304 /// \code 6305 /// void @.red_comb(void* %arg0, void* %arg1) { 6306 /// %lhs = bitcast void* %arg0 to <type>* 6307 /// %rhs = bitcast void* %arg1 to <type>* 6308 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs) 6309 /// store <type> %2, <type>* %lhs 6310 /// ret void 6311 /// } 6312 /// \endcode 6313 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM, 6314 SourceLocation Loc, 6315 ReductionCodeGen &RCG, unsigned N, 6316 const Expr *ReductionOp, 6317 const Expr *LHS, const Expr *RHS, 6318 const Expr *PrivateRef) { 6319 ASTContext &C = CGM.getContext(); 6320 const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl()); 6321 const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl()); 6322 FunctionArgList Args; 6323 ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 6324 C.VoidPtrTy, ImplicitParamDecl::Other); 6325 ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6326 ImplicitParamDecl::Other); 6327 Args.emplace_back(&ParamInOut); 6328 Args.emplace_back(&ParamIn); 6329 const auto &FnInfo = 6330 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6331 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6332 std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""}); 6333 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6334 Name, &CGM.getModule()); 6335 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6336 Fn->setDoesNotRecurse(); 6337 CodeGenFunction CGF(CGM); 6338 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6339 llvm::Value *Size = nullptr; 6340 // If the size of the reduction item is non-constant, load it from global 6341 // threadprivate variable. 6342 if (RCG.getSizes(N).second) { 6343 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6344 CGF, CGM.getContext().getSizeType(), 6345 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6346 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6347 CGM.getContext().getSizeType(), Loc); 6348 } 6349 RCG.emitAggregateType(CGF, N, Size); 6350 // Remap lhs and rhs variables to the addresses of the function arguments. 6351 // %lhs = bitcast void* %arg0 to <type>* 6352 // %rhs = bitcast void* %arg1 to <type>* 6353 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 6354 PrivateScope.addPrivate(LHSVD, [&C, &CGF, &ParamInOut, LHSVD]() { 6355 // Pull out the pointer to the variable. 6356 Address PtrAddr = CGF.EmitLoadOfPointer( 6357 CGF.GetAddrOfLocalVar(&ParamInOut), 6358 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6359 return CGF.Builder.CreateElementBitCast( 6360 PtrAddr, CGF.ConvertTypeForMem(LHSVD->getType())); 6361 }); 6362 PrivateScope.addPrivate(RHSVD, [&C, &CGF, &ParamIn, RHSVD]() { 6363 // Pull out the pointer to the variable. 6364 Address PtrAddr = CGF.EmitLoadOfPointer( 6365 CGF.GetAddrOfLocalVar(&ParamIn), 6366 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6367 return CGF.Builder.CreateElementBitCast( 6368 PtrAddr, CGF.ConvertTypeForMem(RHSVD->getType())); 6369 }); 6370 PrivateScope.Privatize(); 6371 // Emit the combiner body: 6372 // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs) 6373 // store <type> %2, <type>* %lhs 6374 CGM.getOpenMPRuntime().emitSingleReductionCombiner( 6375 CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS), 6376 cast<DeclRefExpr>(RHS)); 6377 CGF.FinishFunction(); 6378 return Fn; 6379 } 6380 6381 /// Emits reduction finalizer function: 6382 /// \code 6383 /// void @.red_fini(void* %arg) { 6384 /// %0 = bitcast void* %arg to <type>* 6385 /// <destroy>(<type>* %0) 6386 /// ret void 6387 /// } 6388 /// \endcode 6389 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM, 6390 SourceLocation Loc, 6391 ReductionCodeGen &RCG, unsigned N) { 6392 if (!RCG.needCleanups(N)) 6393 return nullptr; 6394 ASTContext &C = CGM.getContext(); 6395 FunctionArgList Args; 6396 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6397 ImplicitParamDecl::Other); 6398 Args.emplace_back(&Param); 6399 const auto &FnInfo = 6400 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6401 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6402 std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""}); 6403 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6404 Name, &CGM.getModule()); 6405 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6406 Fn->setDoesNotRecurse(); 6407 CodeGenFunction CGF(CGM); 6408 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6409 Address PrivateAddr = CGF.EmitLoadOfPointer( 6410 CGF.GetAddrOfLocalVar(&Param), 6411 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6412 llvm::Value *Size = nullptr; 6413 // If the size of the reduction item is non-constant, load it from global 6414 // threadprivate variable. 6415 if (RCG.getSizes(N).second) { 6416 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6417 CGF, CGM.getContext().getSizeType(), 6418 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6419 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6420 CGM.getContext().getSizeType(), Loc); 6421 } 6422 RCG.emitAggregateType(CGF, N, Size); 6423 // Emit the finalizer body: 6424 // <destroy>(<type>* %0) 6425 RCG.emitCleanups(CGF, N, PrivateAddr); 6426 CGF.FinishFunction(Loc); 6427 return Fn; 6428 } 6429 6430 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit( 6431 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 6432 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 6433 if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty()) 6434 return nullptr; 6435 6436 // Build typedef struct: 6437 // kmp_task_red_input { 6438 // void *reduce_shar; // shared reduction item 6439 // size_t reduce_size; // size of data item 6440 // void *reduce_init; // data initialization routine 6441 // void *reduce_fini; // data finalization routine 6442 // void *reduce_comb; // data combiner routine 6443 // kmp_task_red_flags_t flags; // flags for additional info from compiler 6444 // } kmp_task_red_input_t; 6445 ASTContext &C = CGM.getContext(); 6446 RecordDecl *RD = C.buildImplicitRecord("kmp_task_red_input_t"); 6447 RD->startDefinition(); 6448 const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6449 const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType()); 6450 const FieldDecl *InitFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6451 const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6452 const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6453 const FieldDecl *FlagsFD = addFieldToRecordDecl( 6454 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false)); 6455 RD->completeDefinition(); 6456 QualType RDType = C.getRecordType(RD); 6457 unsigned Size = Data.ReductionVars.size(); 6458 llvm::APInt ArraySize(/*numBits=*/64, Size); 6459 QualType ArrayRDType = C.getConstantArrayType( 6460 RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 6461 // kmp_task_red_input_t .rd_input.[Size]; 6462 Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input."); 6463 ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionCopies, 6464 Data.ReductionOps); 6465 for (unsigned Cnt = 0; Cnt < Size; ++Cnt) { 6466 // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt]; 6467 llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0), 6468 llvm::ConstantInt::get(CGM.SizeTy, Cnt)}; 6469 llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP( 6470 TaskRedInput.getPointer(), Idxs, 6471 /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc, 6472 ".rd_input.gep."); 6473 LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType); 6474 // ElemLVal.reduce_shar = &Shareds[Cnt]; 6475 LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD); 6476 RCG.emitSharedLValue(CGF, Cnt); 6477 llvm::Value *CastedShared = 6478 CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer(CGF)); 6479 CGF.EmitStoreOfScalar(CastedShared, SharedLVal); 6480 RCG.emitAggregateType(CGF, Cnt); 6481 llvm::Value *SizeValInChars; 6482 llvm::Value *SizeVal; 6483 std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt); 6484 // We use delayed creation/initialization for VLAs, array sections and 6485 // custom reduction initializations. It is required because runtime does not 6486 // provide the way to pass the sizes of VLAs/array sections to 6487 // initializer/combiner/finalizer functions and does not pass the pointer to 6488 // original reduction item to the initializer. Instead threadprivate global 6489 // variables are used to store these values and use them in the functions. 6490 bool DelayedCreation = !!SizeVal; 6491 SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy, 6492 /*isSigned=*/false); 6493 LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD); 6494 CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal); 6495 // ElemLVal.reduce_init = init; 6496 LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD); 6497 llvm::Value *InitAddr = 6498 CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt)); 6499 CGF.EmitStoreOfScalar(InitAddr, InitLVal); 6500 DelayedCreation = DelayedCreation || RCG.usesReductionInitializer(Cnt); 6501 // ElemLVal.reduce_fini = fini; 6502 LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD); 6503 llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt); 6504 llvm::Value *FiniAddr = Fini 6505 ? CGF.EmitCastToVoidPtr(Fini) 6506 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy); 6507 CGF.EmitStoreOfScalar(FiniAddr, FiniLVal); 6508 // ElemLVal.reduce_comb = comb; 6509 LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD); 6510 llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction( 6511 CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt], 6512 RHSExprs[Cnt], Data.ReductionCopies[Cnt])); 6513 CGF.EmitStoreOfScalar(CombAddr, CombLVal); 6514 // ElemLVal.flags = 0; 6515 LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD); 6516 if (DelayedCreation) { 6517 CGF.EmitStoreOfScalar( 6518 llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true), 6519 FlagsLVal); 6520 } else 6521 CGF.EmitNullInitialization(FlagsLVal.getAddress(CGF), 6522 FlagsLVal.getType()); 6523 } 6524 // Build call void *__kmpc_task_reduction_init(int gtid, int num_data, void 6525 // *data); 6526 llvm::Value *Args[] = { 6527 CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy, 6528 /*isSigned=*/true), 6529 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true), 6530 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(), 6531 CGM.VoidPtrTy)}; 6532 return CGF.EmitRuntimeCall( 6533 createRuntimeFunction(OMPRTL__kmpc_task_reduction_init), Args); 6534 } 6535 6536 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 6537 SourceLocation Loc, 6538 ReductionCodeGen &RCG, 6539 unsigned N) { 6540 auto Sizes = RCG.getSizes(N); 6541 // Emit threadprivate global variable if the type is non-constant 6542 // (Sizes.second = nullptr). 6543 if (Sizes.second) { 6544 llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy, 6545 /*isSigned=*/false); 6546 Address SizeAddr = getAddrOfArtificialThreadPrivate( 6547 CGF, CGM.getContext().getSizeType(), 6548 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6549 CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false); 6550 } 6551 // Store address of the original reduction item if custom initializer is used. 6552 if (RCG.usesReductionInitializer(N)) { 6553 Address SharedAddr = getAddrOfArtificialThreadPrivate( 6554 CGF, CGM.getContext().VoidPtrTy, 6555 generateUniqueName(CGM, "reduction", RCG.getRefExpr(N))); 6556 CGF.Builder.CreateStore( 6557 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6558 RCG.getSharedLValue(N).getPointer(CGF), CGM.VoidPtrTy), 6559 SharedAddr, /*IsVolatile=*/false); 6560 } 6561 } 6562 6563 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF, 6564 SourceLocation Loc, 6565 llvm::Value *ReductionsPtr, 6566 LValue SharedLVal) { 6567 // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 6568 // *d); 6569 llvm::Value *Args[] = {CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), 6570 CGM.IntTy, 6571 /*isSigned=*/true), 6572 ReductionsPtr, 6573 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6574 SharedLVal.getPointer(CGF), CGM.VoidPtrTy)}; 6575 return Address( 6576 CGF.EmitRuntimeCall( 6577 createRuntimeFunction(OMPRTL__kmpc_task_reduction_get_th_data), Args), 6578 SharedLVal.getAlignment()); 6579 } 6580 6581 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 6582 SourceLocation Loc) { 6583 if (!CGF.HaveInsertPoint()) 6584 return; 6585 6586 llvm::OpenMPIRBuilder *OMPBuilder = CGF.CGM.getOpenMPIRBuilder(); 6587 if (OMPBuilder) { 6588 OMPBuilder->CreateTaskwait(CGF.Builder); 6589 } else { 6590 // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 6591 // global_tid); 6592 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 6593 // Ignore return result until untied tasks are supported. 6594 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskwait), Args); 6595 } 6596 6597 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 6598 Region->emitUntiedSwitch(CGF); 6599 } 6600 6601 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF, 6602 OpenMPDirectiveKind InnerKind, 6603 const RegionCodeGenTy &CodeGen, 6604 bool HasCancel) { 6605 if (!CGF.HaveInsertPoint()) 6606 return; 6607 InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel); 6608 CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr); 6609 } 6610 6611 namespace { 6612 enum RTCancelKind { 6613 CancelNoreq = 0, 6614 CancelParallel = 1, 6615 CancelLoop = 2, 6616 CancelSections = 3, 6617 CancelTaskgroup = 4 6618 }; 6619 } // anonymous namespace 6620 6621 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) { 6622 RTCancelKind CancelKind = CancelNoreq; 6623 if (CancelRegion == OMPD_parallel) 6624 CancelKind = CancelParallel; 6625 else if (CancelRegion == OMPD_for) 6626 CancelKind = CancelLoop; 6627 else if (CancelRegion == OMPD_sections) 6628 CancelKind = CancelSections; 6629 else { 6630 assert(CancelRegion == OMPD_taskgroup); 6631 CancelKind = CancelTaskgroup; 6632 } 6633 return CancelKind; 6634 } 6635 6636 void CGOpenMPRuntime::emitCancellationPointCall( 6637 CodeGenFunction &CGF, SourceLocation Loc, 6638 OpenMPDirectiveKind CancelRegion) { 6639 if (!CGF.HaveInsertPoint()) 6640 return; 6641 // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 6642 // global_tid, kmp_int32 cncl_kind); 6643 if (auto *OMPRegionInfo = 6644 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6645 // For 'cancellation point taskgroup', the task region info may not have a 6646 // cancel. This may instead happen in another adjacent task. 6647 if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) { 6648 llvm::Value *Args[] = { 6649 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 6650 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6651 // Ignore return result until untied tasks are supported. 6652 llvm::Value *Result = CGF.EmitRuntimeCall( 6653 createRuntimeFunction(OMPRTL__kmpc_cancellationpoint), Args); 6654 // if (__kmpc_cancellationpoint()) { 6655 // exit from construct; 6656 // } 6657 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6658 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6659 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6660 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6661 CGF.EmitBlock(ExitBB); 6662 // exit from construct; 6663 CodeGenFunction::JumpDest CancelDest = 6664 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6665 CGF.EmitBranchThroughCleanup(CancelDest); 6666 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6667 } 6668 } 6669 } 6670 6671 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc, 6672 const Expr *IfCond, 6673 OpenMPDirectiveKind CancelRegion) { 6674 if (!CGF.HaveInsertPoint()) 6675 return; 6676 // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 6677 // kmp_int32 cncl_kind); 6678 if (auto *OMPRegionInfo = 6679 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6680 auto &&ThenGen = [Loc, CancelRegion, OMPRegionInfo](CodeGenFunction &CGF, 6681 PrePostActionTy &) { 6682 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6683 llvm::Value *Args[] = { 6684 RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc), 6685 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6686 // Ignore return result until untied tasks are supported. 6687 llvm::Value *Result = CGF.EmitRuntimeCall( 6688 RT.createRuntimeFunction(OMPRTL__kmpc_cancel), Args); 6689 // if (__kmpc_cancel()) { 6690 // exit from construct; 6691 // } 6692 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6693 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6694 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6695 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6696 CGF.EmitBlock(ExitBB); 6697 // exit from construct; 6698 CodeGenFunction::JumpDest CancelDest = 6699 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6700 CGF.EmitBranchThroughCleanup(CancelDest); 6701 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6702 }; 6703 if (IfCond) { 6704 emitIfClause(CGF, IfCond, ThenGen, 6705 [](CodeGenFunction &, PrePostActionTy &) {}); 6706 } else { 6707 RegionCodeGenTy ThenRCG(ThenGen); 6708 ThenRCG(CGF); 6709 } 6710 } 6711 } 6712 6713 void CGOpenMPRuntime::emitTargetOutlinedFunction( 6714 const OMPExecutableDirective &D, StringRef ParentName, 6715 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6716 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6717 assert(!ParentName.empty() && "Invalid target region parent name!"); 6718 HasEmittedTargetRegion = true; 6719 emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID, 6720 IsOffloadEntry, CodeGen); 6721 } 6722 6723 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper( 6724 const OMPExecutableDirective &D, StringRef ParentName, 6725 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6726 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6727 // Create a unique name for the entry function using the source location 6728 // information of the current target region. The name will be something like: 6729 // 6730 // __omp_offloading_DD_FFFF_PP_lBB 6731 // 6732 // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the 6733 // mangled name of the function that encloses the target region and BB is the 6734 // line number of the target region. 6735 6736 unsigned DeviceID; 6737 unsigned FileID; 6738 unsigned Line; 6739 getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID, 6740 Line); 6741 SmallString<64> EntryFnName; 6742 { 6743 llvm::raw_svector_ostream OS(EntryFnName); 6744 OS << "__omp_offloading" << llvm::format("_%x", DeviceID) 6745 << llvm::format("_%x_", FileID) << ParentName << "_l" << Line; 6746 } 6747 6748 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 6749 6750 CodeGenFunction CGF(CGM, true); 6751 CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName); 6752 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6753 6754 OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS, D.getBeginLoc()); 6755 6756 // If this target outline function is not an offload entry, we don't need to 6757 // register it. 6758 if (!IsOffloadEntry) 6759 return; 6760 6761 // The target region ID is used by the runtime library to identify the current 6762 // target region, so it only has to be unique and not necessarily point to 6763 // anything. It could be the pointer to the outlined function that implements 6764 // the target region, but we aren't using that so that the compiler doesn't 6765 // need to keep that, and could therefore inline the host function if proven 6766 // worthwhile during optimization. In the other hand, if emitting code for the 6767 // device, the ID has to be the function address so that it can retrieved from 6768 // the offloading entry and launched by the runtime library. We also mark the 6769 // outlined function to have external linkage in case we are emitting code for 6770 // the device, because these functions will be entry points to the device. 6771 6772 if (CGM.getLangOpts().OpenMPIsDevice) { 6773 OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy); 6774 OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 6775 OutlinedFn->setDSOLocal(false); 6776 } else { 6777 std::string Name = getName({EntryFnName, "region_id"}); 6778 OutlinedFnID = new llvm::GlobalVariable( 6779 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 6780 llvm::GlobalValue::WeakAnyLinkage, 6781 llvm::Constant::getNullValue(CGM.Int8Ty), Name); 6782 } 6783 6784 // Register the information for the entry associated with this target region. 6785 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 6786 DeviceID, FileID, ParentName, Line, OutlinedFn, OutlinedFnID, 6787 OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion); 6788 } 6789 6790 /// Checks if the expression is constant or does not have non-trivial function 6791 /// calls. 6792 static bool isTrivial(ASTContext &Ctx, const Expr * E) { 6793 // We can skip constant expressions. 6794 // We can skip expressions with trivial calls or simple expressions. 6795 return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) || 6796 !E->hasNonTrivialCall(Ctx)) && 6797 !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true); 6798 } 6799 6800 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx, 6801 const Stmt *Body) { 6802 const Stmt *Child = Body->IgnoreContainers(); 6803 while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) { 6804 Child = nullptr; 6805 for (const Stmt *S : C->body()) { 6806 if (const auto *E = dyn_cast<Expr>(S)) { 6807 if (isTrivial(Ctx, E)) 6808 continue; 6809 } 6810 // Some of the statements can be ignored. 6811 if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) || 6812 isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S)) 6813 continue; 6814 // Analyze declarations. 6815 if (const auto *DS = dyn_cast<DeclStmt>(S)) { 6816 if (llvm::all_of(DS->decls(), [&Ctx](const Decl *D) { 6817 if (isa<EmptyDecl>(D) || isa<DeclContext>(D) || 6818 isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) || 6819 isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) || 6820 isa<UsingDirectiveDecl>(D) || 6821 isa<OMPDeclareReductionDecl>(D) || 6822 isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D)) 6823 return true; 6824 const auto *VD = dyn_cast<VarDecl>(D); 6825 if (!VD) 6826 return false; 6827 return VD->isConstexpr() || 6828 ((VD->getType().isTrivialType(Ctx) || 6829 VD->getType()->isReferenceType()) && 6830 (!VD->hasInit() || isTrivial(Ctx, VD->getInit()))); 6831 })) 6832 continue; 6833 } 6834 // Found multiple children - cannot get the one child only. 6835 if (Child) 6836 return nullptr; 6837 Child = S; 6838 } 6839 if (Child) 6840 Child = Child->IgnoreContainers(); 6841 } 6842 return Child; 6843 } 6844 6845 /// Emit the number of teams for a target directive. Inspect the num_teams 6846 /// clause associated with a teams construct combined or closely nested 6847 /// with the target directive. 6848 /// 6849 /// Emit a team of size one for directives such as 'target parallel' that 6850 /// have no associated teams construct. 6851 /// 6852 /// Otherwise, return nullptr. 6853 static llvm::Value * 6854 emitNumTeamsForTargetDirective(CodeGenFunction &CGF, 6855 const OMPExecutableDirective &D) { 6856 assert(!CGF.getLangOpts().OpenMPIsDevice && 6857 "Clauses associated with the teams directive expected to be emitted " 6858 "only for the host!"); 6859 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6860 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 6861 "Expected target-based executable directive."); 6862 CGBuilderTy &Bld = CGF.Builder; 6863 switch (DirectiveKind) { 6864 case OMPD_target: { 6865 const auto *CS = D.getInnermostCapturedStmt(); 6866 const auto *Body = 6867 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 6868 const Stmt *ChildStmt = 6869 CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body); 6870 if (const auto *NestedDir = 6871 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 6872 if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) { 6873 if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) { 6874 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6875 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6876 const Expr *NumTeams = 6877 NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6878 llvm::Value *NumTeamsVal = 6879 CGF.EmitScalarExpr(NumTeams, 6880 /*IgnoreResultAssign*/ true); 6881 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6882 /*isSigned=*/true); 6883 } 6884 return Bld.getInt32(0); 6885 } 6886 if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) || 6887 isOpenMPSimdDirective(NestedDir->getDirectiveKind())) 6888 return Bld.getInt32(1); 6889 return Bld.getInt32(0); 6890 } 6891 return nullptr; 6892 } 6893 case OMPD_target_teams: 6894 case OMPD_target_teams_distribute: 6895 case OMPD_target_teams_distribute_simd: 6896 case OMPD_target_teams_distribute_parallel_for: 6897 case OMPD_target_teams_distribute_parallel_for_simd: { 6898 if (D.hasClausesOfKind<OMPNumTeamsClause>()) { 6899 CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF); 6900 const Expr *NumTeams = 6901 D.getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6902 llvm::Value *NumTeamsVal = 6903 CGF.EmitScalarExpr(NumTeams, 6904 /*IgnoreResultAssign*/ true); 6905 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6906 /*isSigned=*/true); 6907 } 6908 return Bld.getInt32(0); 6909 } 6910 case OMPD_target_parallel: 6911 case OMPD_target_parallel_for: 6912 case OMPD_target_parallel_for_simd: 6913 case OMPD_target_simd: 6914 return Bld.getInt32(1); 6915 case OMPD_parallel: 6916 case OMPD_for: 6917 case OMPD_parallel_for: 6918 case OMPD_parallel_master: 6919 case OMPD_parallel_sections: 6920 case OMPD_for_simd: 6921 case OMPD_parallel_for_simd: 6922 case OMPD_cancel: 6923 case OMPD_cancellation_point: 6924 case OMPD_ordered: 6925 case OMPD_threadprivate: 6926 case OMPD_allocate: 6927 case OMPD_task: 6928 case OMPD_simd: 6929 case OMPD_sections: 6930 case OMPD_section: 6931 case OMPD_single: 6932 case OMPD_master: 6933 case OMPD_critical: 6934 case OMPD_taskyield: 6935 case OMPD_barrier: 6936 case OMPD_taskwait: 6937 case OMPD_taskgroup: 6938 case OMPD_atomic: 6939 case OMPD_flush: 6940 case OMPD_depobj: 6941 case OMPD_scan: 6942 case OMPD_teams: 6943 case OMPD_target_data: 6944 case OMPD_target_exit_data: 6945 case OMPD_target_enter_data: 6946 case OMPD_distribute: 6947 case OMPD_distribute_simd: 6948 case OMPD_distribute_parallel_for: 6949 case OMPD_distribute_parallel_for_simd: 6950 case OMPD_teams_distribute: 6951 case OMPD_teams_distribute_simd: 6952 case OMPD_teams_distribute_parallel_for: 6953 case OMPD_teams_distribute_parallel_for_simd: 6954 case OMPD_target_update: 6955 case OMPD_declare_simd: 6956 case OMPD_declare_variant: 6957 case OMPD_begin_declare_variant: 6958 case OMPD_end_declare_variant: 6959 case OMPD_declare_target: 6960 case OMPD_end_declare_target: 6961 case OMPD_declare_reduction: 6962 case OMPD_declare_mapper: 6963 case OMPD_taskloop: 6964 case OMPD_taskloop_simd: 6965 case OMPD_master_taskloop: 6966 case OMPD_master_taskloop_simd: 6967 case OMPD_parallel_master_taskloop: 6968 case OMPD_parallel_master_taskloop_simd: 6969 case OMPD_requires: 6970 case OMPD_unknown: 6971 break; 6972 } 6973 llvm_unreachable("Unexpected directive kind."); 6974 } 6975 6976 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS, 6977 llvm::Value *DefaultThreadLimitVal) { 6978 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6979 CGF.getContext(), CS->getCapturedStmt()); 6980 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6981 if (isOpenMPParallelDirective(Dir->getDirectiveKind())) { 6982 llvm::Value *NumThreads = nullptr; 6983 llvm::Value *CondVal = nullptr; 6984 // Handle if clause. If if clause present, the number of threads is 6985 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 6986 if (Dir->hasClausesOfKind<OMPIfClause>()) { 6987 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6988 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6989 const OMPIfClause *IfClause = nullptr; 6990 for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) { 6991 if (C->getNameModifier() == OMPD_unknown || 6992 C->getNameModifier() == OMPD_parallel) { 6993 IfClause = C; 6994 break; 6995 } 6996 } 6997 if (IfClause) { 6998 const Expr *Cond = IfClause->getCondition(); 6999 bool Result; 7000 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 7001 if (!Result) 7002 return CGF.Builder.getInt32(1); 7003 } else { 7004 CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange()); 7005 if (const auto *PreInit = 7006 cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) { 7007 for (const auto *I : PreInit->decls()) { 7008 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 7009 CGF.EmitVarDecl(cast<VarDecl>(*I)); 7010 } else { 7011 CodeGenFunction::AutoVarEmission Emission = 7012 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 7013 CGF.EmitAutoVarCleanups(Emission); 7014 } 7015 } 7016 } 7017 CondVal = CGF.EvaluateExprAsBool(Cond); 7018 } 7019 } 7020 } 7021 // Check the value of num_threads clause iff if clause was not specified 7022 // or is not evaluated to false. 7023 if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) { 7024 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 7025 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 7026 const auto *NumThreadsClause = 7027 Dir->getSingleClause<OMPNumThreadsClause>(); 7028 CodeGenFunction::LexicalScope Scope( 7029 CGF, NumThreadsClause->getNumThreads()->getSourceRange()); 7030 if (const auto *PreInit = 7031 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) { 7032 for (const auto *I : PreInit->decls()) { 7033 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 7034 CGF.EmitVarDecl(cast<VarDecl>(*I)); 7035 } else { 7036 CodeGenFunction::AutoVarEmission Emission = 7037 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 7038 CGF.EmitAutoVarCleanups(Emission); 7039 } 7040 } 7041 } 7042 NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads()); 7043 NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, 7044 /*isSigned=*/false); 7045 if (DefaultThreadLimitVal) 7046 NumThreads = CGF.Builder.CreateSelect( 7047 CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads), 7048 DefaultThreadLimitVal, NumThreads); 7049 } else { 7050 NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal 7051 : CGF.Builder.getInt32(0); 7052 } 7053 // Process condition of the if clause. 7054 if (CondVal) { 7055 NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads, 7056 CGF.Builder.getInt32(1)); 7057 } 7058 return NumThreads; 7059 } 7060 if (isOpenMPSimdDirective(Dir->getDirectiveKind())) 7061 return CGF.Builder.getInt32(1); 7062 return DefaultThreadLimitVal; 7063 } 7064 return DefaultThreadLimitVal ? DefaultThreadLimitVal 7065 : CGF.Builder.getInt32(0); 7066 } 7067 7068 /// Emit the number of threads for a target directive. Inspect the 7069 /// thread_limit clause associated with a teams construct combined or closely 7070 /// nested with the target directive. 7071 /// 7072 /// Emit the num_threads clause for directives such as 'target parallel' that 7073 /// have no associated teams construct. 7074 /// 7075 /// Otherwise, return nullptr. 7076 static llvm::Value * 7077 emitNumThreadsForTargetDirective(CodeGenFunction &CGF, 7078 const OMPExecutableDirective &D) { 7079 assert(!CGF.getLangOpts().OpenMPIsDevice && 7080 "Clauses associated with the teams directive expected to be emitted " 7081 "only for the host!"); 7082 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 7083 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 7084 "Expected target-based executable directive."); 7085 CGBuilderTy &Bld = CGF.Builder; 7086 llvm::Value *ThreadLimitVal = nullptr; 7087 llvm::Value *NumThreadsVal = nullptr; 7088 switch (DirectiveKind) { 7089 case OMPD_target: { 7090 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 7091 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7092 return NumThreads; 7093 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7094 CGF.getContext(), CS->getCapturedStmt()); 7095 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 7096 if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) { 7097 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 7098 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 7099 const auto *ThreadLimitClause = 7100 Dir->getSingleClause<OMPThreadLimitClause>(); 7101 CodeGenFunction::LexicalScope Scope( 7102 CGF, ThreadLimitClause->getThreadLimit()->getSourceRange()); 7103 if (const auto *PreInit = 7104 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) { 7105 for (const auto *I : PreInit->decls()) { 7106 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 7107 CGF.EmitVarDecl(cast<VarDecl>(*I)); 7108 } else { 7109 CodeGenFunction::AutoVarEmission Emission = 7110 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 7111 CGF.EmitAutoVarCleanups(Emission); 7112 } 7113 } 7114 } 7115 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7116 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7117 ThreadLimitVal = 7118 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7119 } 7120 if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) && 7121 !isOpenMPDistributeDirective(Dir->getDirectiveKind())) { 7122 CS = Dir->getInnermostCapturedStmt(); 7123 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7124 CGF.getContext(), CS->getCapturedStmt()); 7125 Dir = dyn_cast_or_null<OMPExecutableDirective>(Child); 7126 } 7127 if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) && 7128 !isOpenMPSimdDirective(Dir->getDirectiveKind())) { 7129 CS = Dir->getInnermostCapturedStmt(); 7130 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7131 return NumThreads; 7132 } 7133 if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind())) 7134 return Bld.getInt32(1); 7135 } 7136 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 7137 } 7138 case OMPD_target_teams: { 7139 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7140 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7141 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7142 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7143 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7144 ThreadLimitVal = 7145 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7146 } 7147 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 7148 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7149 return NumThreads; 7150 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7151 CGF.getContext(), CS->getCapturedStmt()); 7152 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 7153 if (Dir->getDirectiveKind() == OMPD_distribute) { 7154 CS = Dir->getInnermostCapturedStmt(); 7155 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7156 return NumThreads; 7157 } 7158 } 7159 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 7160 } 7161 case OMPD_target_teams_distribute: 7162 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7163 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7164 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7165 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7166 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7167 ThreadLimitVal = 7168 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7169 } 7170 return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal); 7171 case OMPD_target_parallel: 7172 case OMPD_target_parallel_for: 7173 case OMPD_target_parallel_for_simd: 7174 case OMPD_target_teams_distribute_parallel_for: 7175 case OMPD_target_teams_distribute_parallel_for_simd: { 7176 llvm::Value *CondVal = nullptr; 7177 // Handle if clause. If if clause present, the number of threads is 7178 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 7179 if (D.hasClausesOfKind<OMPIfClause>()) { 7180 const OMPIfClause *IfClause = nullptr; 7181 for (const auto *C : D.getClausesOfKind<OMPIfClause>()) { 7182 if (C->getNameModifier() == OMPD_unknown || 7183 C->getNameModifier() == OMPD_parallel) { 7184 IfClause = C; 7185 break; 7186 } 7187 } 7188 if (IfClause) { 7189 const Expr *Cond = IfClause->getCondition(); 7190 bool Result; 7191 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 7192 if (!Result) 7193 return Bld.getInt32(1); 7194 } else { 7195 CodeGenFunction::RunCleanupsScope Scope(CGF); 7196 CondVal = CGF.EvaluateExprAsBool(Cond); 7197 } 7198 } 7199 } 7200 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7201 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7202 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7203 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7204 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7205 ThreadLimitVal = 7206 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7207 } 7208 if (D.hasClausesOfKind<OMPNumThreadsClause>()) { 7209 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF); 7210 const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>(); 7211 llvm::Value *NumThreads = CGF.EmitScalarExpr( 7212 NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true); 7213 NumThreadsVal = 7214 Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false); 7215 ThreadLimitVal = ThreadLimitVal 7216 ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal, 7217 ThreadLimitVal), 7218 NumThreadsVal, ThreadLimitVal) 7219 : NumThreadsVal; 7220 } 7221 if (!ThreadLimitVal) 7222 ThreadLimitVal = Bld.getInt32(0); 7223 if (CondVal) 7224 return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1)); 7225 return ThreadLimitVal; 7226 } 7227 case OMPD_target_teams_distribute_simd: 7228 case OMPD_target_simd: 7229 return Bld.getInt32(1); 7230 case OMPD_parallel: 7231 case OMPD_for: 7232 case OMPD_parallel_for: 7233 case OMPD_parallel_master: 7234 case OMPD_parallel_sections: 7235 case OMPD_for_simd: 7236 case OMPD_parallel_for_simd: 7237 case OMPD_cancel: 7238 case OMPD_cancellation_point: 7239 case OMPD_ordered: 7240 case OMPD_threadprivate: 7241 case OMPD_allocate: 7242 case OMPD_task: 7243 case OMPD_simd: 7244 case OMPD_sections: 7245 case OMPD_section: 7246 case OMPD_single: 7247 case OMPD_master: 7248 case OMPD_critical: 7249 case OMPD_taskyield: 7250 case OMPD_barrier: 7251 case OMPD_taskwait: 7252 case OMPD_taskgroup: 7253 case OMPD_atomic: 7254 case OMPD_flush: 7255 case OMPD_depobj: 7256 case OMPD_scan: 7257 case OMPD_teams: 7258 case OMPD_target_data: 7259 case OMPD_target_exit_data: 7260 case OMPD_target_enter_data: 7261 case OMPD_distribute: 7262 case OMPD_distribute_simd: 7263 case OMPD_distribute_parallel_for: 7264 case OMPD_distribute_parallel_for_simd: 7265 case OMPD_teams_distribute: 7266 case OMPD_teams_distribute_simd: 7267 case OMPD_teams_distribute_parallel_for: 7268 case OMPD_teams_distribute_parallel_for_simd: 7269 case OMPD_target_update: 7270 case OMPD_declare_simd: 7271 case OMPD_declare_variant: 7272 case OMPD_begin_declare_variant: 7273 case OMPD_end_declare_variant: 7274 case OMPD_declare_target: 7275 case OMPD_end_declare_target: 7276 case OMPD_declare_reduction: 7277 case OMPD_declare_mapper: 7278 case OMPD_taskloop: 7279 case OMPD_taskloop_simd: 7280 case OMPD_master_taskloop: 7281 case OMPD_master_taskloop_simd: 7282 case OMPD_parallel_master_taskloop: 7283 case OMPD_parallel_master_taskloop_simd: 7284 case OMPD_requires: 7285 case OMPD_unknown: 7286 break; 7287 } 7288 llvm_unreachable("Unsupported directive kind."); 7289 } 7290 7291 namespace { 7292 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 7293 7294 // Utility to handle information from clauses associated with a given 7295 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause). 7296 // It provides a convenient interface to obtain the information and generate 7297 // code for that information. 7298 class MappableExprsHandler { 7299 public: 7300 /// Values for bit flags used to specify the mapping type for 7301 /// offloading. 7302 enum OpenMPOffloadMappingFlags : uint64_t { 7303 /// No flags 7304 OMP_MAP_NONE = 0x0, 7305 /// Allocate memory on the device and move data from host to device. 7306 OMP_MAP_TO = 0x01, 7307 /// Allocate memory on the device and move data from device to host. 7308 OMP_MAP_FROM = 0x02, 7309 /// Always perform the requested mapping action on the element, even 7310 /// if it was already mapped before. 7311 OMP_MAP_ALWAYS = 0x04, 7312 /// Delete the element from the device environment, ignoring the 7313 /// current reference count associated with the element. 7314 OMP_MAP_DELETE = 0x08, 7315 /// The element being mapped is a pointer-pointee pair; both the 7316 /// pointer and the pointee should be mapped. 7317 OMP_MAP_PTR_AND_OBJ = 0x10, 7318 /// This flags signals that the base address of an entry should be 7319 /// passed to the target kernel as an argument. 7320 OMP_MAP_TARGET_PARAM = 0x20, 7321 /// Signal that the runtime library has to return the device pointer 7322 /// in the current position for the data being mapped. Used when we have the 7323 /// use_device_ptr clause. 7324 OMP_MAP_RETURN_PARAM = 0x40, 7325 /// This flag signals that the reference being passed is a pointer to 7326 /// private data. 7327 OMP_MAP_PRIVATE = 0x80, 7328 /// Pass the element to the device by value. 7329 OMP_MAP_LITERAL = 0x100, 7330 /// Implicit map 7331 OMP_MAP_IMPLICIT = 0x200, 7332 /// Close is a hint to the runtime to allocate memory close to 7333 /// the target device. 7334 OMP_MAP_CLOSE = 0x400, 7335 /// The 16 MSBs of the flags indicate whether the entry is member of some 7336 /// struct/class. 7337 OMP_MAP_MEMBER_OF = 0xffff000000000000, 7338 LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF), 7339 }; 7340 7341 /// Get the offset of the OMP_MAP_MEMBER_OF field. 7342 static unsigned getFlagMemberOffset() { 7343 unsigned Offset = 0; 7344 for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1); 7345 Remain = Remain >> 1) 7346 Offset++; 7347 return Offset; 7348 } 7349 7350 /// Class that associates information with a base pointer to be passed to the 7351 /// runtime library. 7352 class BasePointerInfo { 7353 /// The base pointer. 7354 llvm::Value *Ptr = nullptr; 7355 /// The base declaration that refers to this device pointer, or null if 7356 /// there is none. 7357 const ValueDecl *DevPtrDecl = nullptr; 7358 7359 public: 7360 BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr) 7361 : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {} 7362 llvm::Value *operator*() const { return Ptr; } 7363 const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; } 7364 void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; } 7365 }; 7366 7367 using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>; 7368 using MapValuesArrayTy = SmallVector<llvm::Value *, 4>; 7369 using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>; 7370 7371 /// Map between a struct and the its lowest & highest elements which have been 7372 /// mapped. 7373 /// [ValueDecl *] --> {LE(FieldIndex, Pointer), 7374 /// HE(FieldIndex, Pointer)} 7375 struct StructRangeInfoTy { 7376 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = { 7377 0, Address::invalid()}; 7378 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = { 7379 0, Address::invalid()}; 7380 Address Base = Address::invalid(); 7381 }; 7382 7383 private: 7384 /// Kind that defines how a device pointer has to be returned. 7385 struct MapInfo { 7386 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 7387 OpenMPMapClauseKind MapType = OMPC_MAP_unknown; 7388 ArrayRef<OpenMPMapModifierKind> MapModifiers; 7389 bool ReturnDevicePointer = false; 7390 bool IsImplicit = false; 7391 7392 MapInfo() = default; 7393 MapInfo( 7394 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7395 OpenMPMapClauseKind MapType, 7396 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7397 bool ReturnDevicePointer, bool IsImplicit) 7398 : Components(Components), MapType(MapType), MapModifiers(MapModifiers), 7399 ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit) {} 7400 }; 7401 7402 /// If use_device_ptr is used on a pointer which is a struct member and there 7403 /// is no map information about it, then emission of that entry is deferred 7404 /// until the whole struct has been processed. 7405 struct DeferredDevicePtrEntryTy { 7406 const Expr *IE = nullptr; 7407 const ValueDecl *VD = nullptr; 7408 7409 DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD) 7410 : IE(IE), VD(VD) {} 7411 }; 7412 7413 /// The target directive from where the mappable clauses were extracted. It 7414 /// is either a executable directive or a user-defined mapper directive. 7415 llvm::PointerUnion<const OMPExecutableDirective *, 7416 const OMPDeclareMapperDecl *> 7417 CurDir; 7418 7419 /// Function the directive is being generated for. 7420 CodeGenFunction &CGF; 7421 7422 /// Set of all first private variables in the current directive. 7423 /// bool data is set to true if the variable is implicitly marked as 7424 /// firstprivate, false otherwise. 7425 llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls; 7426 7427 /// Map between device pointer declarations and their expression components. 7428 /// The key value for declarations in 'this' is null. 7429 llvm::DenseMap< 7430 const ValueDecl *, 7431 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>> 7432 DevPointersMap; 7433 7434 llvm::Value *getExprTypeSize(const Expr *E) const { 7435 QualType ExprTy = E->getType().getCanonicalType(); 7436 7437 // Reference types are ignored for mapping purposes. 7438 if (const auto *RefTy = ExprTy->getAs<ReferenceType>()) 7439 ExprTy = RefTy->getPointeeType().getCanonicalType(); 7440 7441 // Given that an array section is considered a built-in type, we need to 7442 // do the calculation based on the length of the section instead of relying 7443 // on CGF.getTypeSize(E->getType()). 7444 if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) { 7445 QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType( 7446 OAE->getBase()->IgnoreParenImpCasts()) 7447 .getCanonicalType(); 7448 7449 // If there is no length associated with the expression and lower bound is 7450 // not specified too, that means we are using the whole length of the 7451 // base. 7452 if (!OAE->getLength() && OAE->getColonLoc().isValid() && 7453 !OAE->getLowerBound()) 7454 return CGF.getTypeSize(BaseTy); 7455 7456 llvm::Value *ElemSize; 7457 if (const auto *PTy = BaseTy->getAs<PointerType>()) { 7458 ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType()); 7459 } else { 7460 const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr()); 7461 assert(ATy && "Expecting array type if not a pointer type."); 7462 ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType()); 7463 } 7464 7465 // If we don't have a length at this point, that is because we have an 7466 // array section with a single element. 7467 if (!OAE->getLength() && OAE->getColonLoc().isInvalid()) 7468 return ElemSize; 7469 7470 if (const Expr *LenExpr = OAE->getLength()) { 7471 llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr); 7472 LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(), 7473 CGF.getContext().getSizeType(), 7474 LenExpr->getExprLoc()); 7475 return CGF.Builder.CreateNUWMul(LengthVal, ElemSize); 7476 } 7477 assert(!OAE->getLength() && OAE->getColonLoc().isValid() && 7478 OAE->getLowerBound() && "expected array_section[lb:]."); 7479 // Size = sizetype - lb * elemtype; 7480 llvm::Value *LengthVal = CGF.getTypeSize(BaseTy); 7481 llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound()); 7482 LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(), 7483 CGF.getContext().getSizeType(), 7484 OAE->getLowerBound()->getExprLoc()); 7485 LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize); 7486 llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal); 7487 llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal); 7488 LengthVal = CGF.Builder.CreateSelect( 7489 Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0)); 7490 return LengthVal; 7491 } 7492 return CGF.getTypeSize(ExprTy); 7493 } 7494 7495 /// Return the corresponding bits for a given map clause modifier. Add 7496 /// a flag marking the map as a pointer if requested. Add a flag marking the 7497 /// map as the first one of a series of maps that relate to the same map 7498 /// expression. 7499 OpenMPOffloadMappingFlags getMapTypeBits( 7500 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers, 7501 bool IsImplicit, bool AddPtrFlag, bool AddIsTargetParamFlag) const { 7502 OpenMPOffloadMappingFlags Bits = 7503 IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE; 7504 switch (MapType) { 7505 case OMPC_MAP_alloc: 7506 case OMPC_MAP_release: 7507 // alloc and release is the default behavior in the runtime library, i.e. 7508 // if we don't pass any bits alloc/release that is what the runtime is 7509 // going to do. Therefore, we don't need to signal anything for these two 7510 // type modifiers. 7511 break; 7512 case OMPC_MAP_to: 7513 Bits |= OMP_MAP_TO; 7514 break; 7515 case OMPC_MAP_from: 7516 Bits |= OMP_MAP_FROM; 7517 break; 7518 case OMPC_MAP_tofrom: 7519 Bits |= OMP_MAP_TO | OMP_MAP_FROM; 7520 break; 7521 case OMPC_MAP_delete: 7522 Bits |= OMP_MAP_DELETE; 7523 break; 7524 case OMPC_MAP_unknown: 7525 llvm_unreachable("Unexpected map type!"); 7526 } 7527 if (AddPtrFlag) 7528 Bits |= OMP_MAP_PTR_AND_OBJ; 7529 if (AddIsTargetParamFlag) 7530 Bits |= OMP_MAP_TARGET_PARAM; 7531 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_always) 7532 != MapModifiers.end()) 7533 Bits |= OMP_MAP_ALWAYS; 7534 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_close) 7535 != MapModifiers.end()) 7536 Bits |= OMP_MAP_CLOSE; 7537 return Bits; 7538 } 7539 7540 /// Return true if the provided expression is a final array section. A 7541 /// final array section, is one whose length can't be proved to be one. 7542 bool isFinalArraySectionExpression(const Expr *E) const { 7543 const auto *OASE = dyn_cast<OMPArraySectionExpr>(E); 7544 7545 // It is not an array section and therefore not a unity-size one. 7546 if (!OASE) 7547 return false; 7548 7549 // An array section with no colon always refer to a single element. 7550 if (OASE->getColonLoc().isInvalid()) 7551 return false; 7552 7553 const Expr *Length = OASE->getLength(); 7554 7555 // If we don't have a length we have to check if the array has size 1 7556 // for this dimension. Also, we should always expect a length if the 7557 // base type is pointer. 7558 if (!Length) { 7559 QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType( 7560 OASE->getBase()->IgnoreParenImpCasts()) 7561 .getCanonicalType(); 7562 if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr())) 7563 return ATy->getSize().getSExtValue() != 1; 7564 // If we don't have a constant dimension length, we have to consider 7565 // the current section as having any size, so it is not necessarily 7566 // unitary. If it happen to be unity size, that's user fault. 7567 return true; 7568 } 7569 7570 // Check if the length evaluates to 1. 7571 Expr::EvalResult Result; 7572 if (!Length->EvaluateAsInt(Result, CGF.getContext())) 7573 return true; // Can have more that size 1. 7574 7575 llvm::APSInt ConstLength = Result.Val.getInt(); 7576 return ConstLength.getSExtValue() != 1; 7577 } 7578 7579 /// Generate the base pointers, section pointers, sizes and map type 7580 /// bits for the provided map type, map modifier, and expression components. 7581 /// \a IsFirstComponent should be set to true if the provided set of 7582 /// components is the first associated with a capture. 7583 void generateInfoForComponentList( 7584 OpenMPMapClauseKind MapType, 7585 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7586 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7587 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 7588 MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types, 7589 StructRangeInfoTy &PartialStruct, bool IsFirstComponentList, 7590 bool IsImplicit, 7591 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 7592 OverlappedElements = llvm::None) const { 7593 // The following summarizes what has to be generated for each map and the 7594 // types below. The generated information is expressed in this order: 7595 // base pointer, section pointer, size, flags 7596 // (to add to the ones that come from the map type and modifier). 7597 // 7598 // double d; 7599 // int i[100]; 7600 // float *p; 7601 // 7602 // struct S1 { 7603 // int i; 7604 // float f[50]; 7605 // } 7606 // struct S2 { 7607 // int i; 7608 // float f[50]; 7609 // S1 s; 7610 // double *p; 7611 // struct S2 *ps; 7612 // } 7613 // S2 s; 7614 // S2 *ps; 7615 // 7616 // map(d) 7617 // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM 7618 // 7619 // map(i) 7620 // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM 7621 // 7622 // map(i[1:23]) 7623 // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM 7624 // 7625 // map(p) 7626 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM 7627 // 7628 // map(p[1:24]) 7629 // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM 7630 // 7631 // map(s) 7632 // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM 7633 // 7634 // map(s.i) 7635 // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM 7636 // 7637 // map(s.s.f) 7638 // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7639 // 7640 // map(s.p) 7641 // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM 7642 // 7643 // map(to: s.p[:22]) 7644 // &s, &(s.p), sizeof(double*), TARGET_PARAM (*) 7645 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**) 7646 // &(s.p), &(s.p[0]), 22*sizeof(double), 7647 // MEMBER_OF(1) | PTR_AND_OBJ | TO (***) 7648 // (*) alloc space for struct members, only this is a target parameter 7649 // (**) map the pointer (nothing to be mapped in this example) (the compiler 7650 // optimizes this entry out, same in the examples below) 7651 // (***) map the pointee (map: to) 7652 // 7653 // map(s.ps) 7654 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7655 // 7656 // map(from: s.ps->s.i) 7657 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7658 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7659 // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7660 // 7661 // map(to: s.ps->ps) 7662 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7663 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7664 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | TO 7665 // 7666 // map(s.ps->ps->ps) 7667 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7668 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7669 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7670 // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7671 // 7672 // map(to: s.ps->ps->s.f[:22]) 7673 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7674 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7675 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7676 // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7677 // 7678 // map(ps) 7679 // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM 7680 // 7681 // map(ps->i) 7682 // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM 7683 // 7684 // map(ps->s.f) 7685 // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7686 // 7687 // map(from: ps->p) 7688 // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM 7689 // 7690 // map(to: ps->p[:22]) 7691 // ps, &(ps->p), sizeof(double*), TARGET_PARAM 7692 // ps, &(ps->p), sizeof(double*), MEMBER_OF(1) 7693 // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO 7694 // 7695 // map(ps->ps) 7696 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7697 // 7698 // map(from: ps->ps->s.i) 7699 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7700 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7701 // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7702 // 7703 // map(from: ps->ps->ps) 7704 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7705 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7706 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7707 // 7708 // map(ps->ps->ps->ps) 7709 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7710 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7711 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7712 // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7713 // 7714 // map(to: ps->ps->ps->s.f[:22]) 7715 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7716 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7717 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7718 // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7719 // 7720 // map(to: s.f[:22]) map(from: s.p[:33]) 7721 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) + 7722 // sizeof(double*) (**), TARGET_PARAM 7723 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO 7724 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) 7725 // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7726 // (*) allocate contiguous space needed to fit all mapped members even if 7727 // we allocate space for members not mapped (in this example, 7728 // s.f[22..49] and s.s are not mapped, yet we must allocate space for 7729 // them as well because they fall between &s.f[0] and &s.p) 7730 // 7731 // map(from: s.f[:22]) map(to: ps->p[:33]) 7732 // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM 7733 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7734 // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*) 7735 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO 7736 // (*) the struct this entry pertains to is the 2nd element in the list of 7737 // arguments, hence MEMBER_OF(2) 7738 // 7739 // map(from: s.f[:22], s.s) map(to: ps->p[:33]) 7740 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM 7741 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM 7742 // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM 7743 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7744 // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*) 7745 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO 7746 // (*) the struct this entry pertains to is the 4th element in the list 7747 // of arguments, hence MEMBER_OF(4) 7748 7749 // Track if the map information being generated is the first for a capture. 7750 bool IsCaptureFirstInfo = IsFirstComponentList; 7751 // When the variable is on a declare target link or in a to clause with 7752 // unified memory, a reference is needed to hold the host/device address 7753 // of the variable. 7754 bool RequiresReference = false; 7755 7756 // Scan the components from the base to the complete expression. 7757 auto CI = Components.rbegin(); 7758 auto CE = Components.rend(); 7759 auto I = CI; 7760 7761 // Track if the map information being generated is the first for a list of 7762 // components. 7763 bool IsExpressionFirstInfo = true; 7764 Address BP = Address::invalid(); 7765 const Expr *AssocExpr = I->getAssociatedExpression(); 7766 const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr); 7767 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 7768 7769 if (isa<MemberExpr>(AssocExpr)) { 7770 // The base is the 'this' pointer. The content of the pointer is going 7771 // to be the base of the field being mapped. 7772 BP = CGF.LoadCXXThisAddress(); 7773 } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) || 7774 (OASE && 7775 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) { 7776 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF); 7777 } else { 7778 // The base is the reference to the variable. 7779 // BP = &Var. 7780 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(CGF); 7781 if (const auto *VD = 7782 dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) { 7783 if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 7784 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) { 7785 if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 7786 (*Res == OMPDeclareTargetDeclAttr::MT_To && 7787 CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) { 7788 RequiresReference = true; 7789 BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 7790 } 7791 } 7792 } 7793 7794 // If the variable is a pointer and is being dereferenced (i.e. is not 7795 // the last component), the base has to be the pointer itself, not its 7796 // reference. References are ignored for mapping purposes. 7797 QualType Ty = 7798 I->getAssociatedDeclaration()->getType().getNonReferenceType(); 7799 if (Ty->isAnyPointerType() && std::next(I) != CE) { 7800 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>()); 7801 7802 // We do not need to generate individual map information for the 7803 // pointer, it can be associated with the combined storage. 7804 ++I; 7805 } 7806 } 7807 7808 // Track whether a component of the list should be marked as MEMBER_OF some 7809 // combined entry (for partial structs). Only the first PTR_AND_OBJ entry 7810 // in a component list should be marked as MEMBER_OF, all subsequent entries 7811 // do not belong to the base struct. E.g. 7812 // struct S2 s; 7813 // s.ps->ps->ps->f[:] 7814 // (1) (2) (3) (4) 7815 // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a 7816 // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3) 7817 // is the pointee of ps(2) which is not member of struct s, so it should not 7818 // be marked as such (it is still PTR_AND_OBJ). 7819 // The variable is initialized to false so that PTR_AND_OBJ entries which 7820 // are not struct members are not considered (e.g. array of pointers to 7821 // data). 7822 bool ShouldBeMemberOf = false; 7823 7824 // Variable keeping track of whether or not we have encountered a component 7825 // in the component list which is a member expression. Useful when we have a 7826 // pointer or a final array section, in which case it is the previous 7827 // component in the list which tells us whether we have a member expression. 7828 // E.g. X.f[:] 7829 // While processing the final array section "[:]" it is "f" which tells us 7830 // whether we are dealing with a member of a declared struct. 7831 const MemberExpr *EncounteredME = nullptr; 7832 7833 for (; I != CE; ++I) { 7834 // If the current component is member of a struct (parent struct) mark it. 7835 if (!EncounteredME) { 7836 EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression()); 7837 // If we encounter a PTR_AND_OBJ entry from now on it should be marked 7838 // as MEMBER_OF the parent struct. 7839 if (EncounteredME) 7840 ShouldBeMemberOf = true; 7841 } 7842 7843 auto Next = std::next(I); 7844 7845 // We need to generate the addresses and sizes if this is the last 7846 // component, if the component is a pointer or if it is an array section 7847 // whose length can't be proved to be one. If this is a pointer, it 7848 // becomes the base address for the following components. 7849 7850 // A final array section, is one whose length can't be proved to be one. 7851 bool IsFinalArraySection = 7852 isFinalArraySectionExpression(I->getAssociatedExpression()); 7853 7854 // Get information on whether the element is a pointer. Have to do a 7855 // special treatment for array sections given that they are built-in 7856 // types. 7857 const auto *OASE = 7858 dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression()); 7859 const auto *UO = dyn_cast<UnaryOperator>(I->getAssociatedExpression()); 7860 const auto *BO = dyn_cast<BinaryOperator>(I->getAssociatedExpression()); 7861 bool IsPointer = 7862 (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE) 7863 .getCanonicalType() 7864 ->isAnyPointerType()) || 7865 I->getAssociatedExpression()->getType()->isAnyPointerType(); 7866 bool IsNonDerefPointer = IsPointer && !UO && !BO; 7867 7868 if (Next == CE || IsNonDerefPointer || IsFinalArraySection) { 7869 // If this is not the last component, we expect the pointer to be 7870 // associated with an array expression or member expression. 7871 assert((Next == CE || 7872 isa<MemberExpr>(Next->getAssociatedExpression()) || 7873 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) || 7874 isa<OMPArraySectionExpr>(Next->getAssociatedExpression()) || 7875 isa<UnaryOperator>(Next->getAssociatedExpression()) || 7876 isa<BinaryOperator>(Next->getAssociatedExpression())) && 7877 "Unexpected expression"); 7878 7879 Address LB = CGF.EmitOMPSharedLValue(I->getAssociatedExpression()) 7880 .getAddress(CGF); 7881 7882 // If this component is a pointer inside the base struct then we don't 7883 // need to create any entry for it - it will be combined with the object 7884 // it is pointing to into a single PTR_AND_OBJ entry. 7885 bool IsMemberPointer = 7886 IsPointer && EncounteredME && 7887 (dyn_cast<MemberExpr>(I->getAssociatedExpression()) == 7888 EncounteredME); 7889 if (!OverlappedElements.empty()) { 7890 // Handle base element with the info for overlapped elements. 7891 assert(!PartialStruct.Base.isValid() && "The base element is set."); 7892 assert(Next == CE && 7893 "Expected last element for the overlapped elements."); 7894 assert(!IsPointer && 7895 "Unexpected base element with the pointer type."); 7896 // Mark the whole struct as the struct that requires allocation on the 7897 // device. 7898 PartialStruct.LowestElem = {0, LB}; 7899 CharUnits TypeSize = CGF.getContext().getTypeSizeInChars( 7900 I->getAssociatedExpression()->getType()); 7901 Address HB = CGF.Builder.CreateConstGEP( 7902 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(LB, 7903 CGF.VoidPtrTy), 7904 TypeSize.getQuantity() - 1); 7905 PartialStruct.HighestElem = { 7906 std::numeric_limits<decltype( 7907 PartialStruct.HighestElem.first)>::max(), 7908 HB}; 7909 PartialStruct.Base = BP; 7910 // Emit data for non-overlapped data. 7911 OpenMPOffloadMappingFlags Flags = 7912 OMP_MAP_MEMBER_OF | 7913 getMapTypeBits(MapType, MapModifiers, IsImplicit, 7914 /*AddPtrFlag=*/false, 7915 /*AddIsTargetParamFlag=*/false); 7916 LB = BP; 7917 llvm::Value *Size = nullptr; 7918 // Do bitcopy of all non-overlapped structure elements. 7919 for (OMPClauseMappableExprCommon::MappableExprComponentListRef 7920 Component : OverlappedElements) { 7921 Address ComponentLB = Address::invalid(); 7922 for (const OMPClauseMappableExprCommon::MappableComponent &MC : 7923 Component) { 7924 if (MC.getAssociatedDeclaration()) { 7925 ComponentLB = 7926 CGF.EmitOMPSharedLValue(MC.getAssociatedExpression()) 7927 .getAddress(CGF); 7928 Size = CGF.Builder.CreatePtrDiff( 7929 CGF.EmitCastToVoidPtr(ComponentLB.getPointer()), 7930 CGF.EmitCastToVoidPtr(LB.getPointer())); 7931 break; 7932 } 7933 } 7934 BasePointers.push_back(BP.getPointer()); 7935 Pointers.push_back(LB.getPointer()); 7936 Sizes.push_back(CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, 7937 /*isSigned=*/true)); 7938 Types.push_back(Flags); 7939 LB = CGF.Builder.CreateConstGEP(ComponentLB, 1); 7940 } 7941 BasePointers.push_back(BP.getPointer()); 7942 Pointers.push_back(LB.getPointer()); 7943 Size = CGF.Builder.CreatePtrDiff( 7944 CGF.EmitCastToVoidPtr( 7945 CGF.Builder.CreateConstGEP(HB, 1).getPointer()), 7946 CGF.EmitCastToVoidPtr(LB.getPointer())); 7947 Sizes.push_back( 7948 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 7949 Types.push_back(Flags); 7950 break; 7951 } 7952 llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression()); 7953 if (!IsMemberPointer) { 7954 BasePointers.push_back(BP.getPointer()); 7955 Pointers.push_back(LB.getPointer()); 7956 Sizes.push_back( 7957 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 7958 7959 // We need to add a pointer flag for each map that comes from the 7960 // same expression except for the first one. We also need to signal 7961 // this map is the first one that relates with the current capture 7962 // (there is a set of entries for each capture). 7963 OpenMPOffloadMappingFlags Flags = getMapTypeBits( 7964 MapType, MapModifiers, IsImplicit, 7965 !IsExpressionFirstInfo || RequiresReference, 7966 IsCaptureFirstInfo && !RequiresReference); 7967 7968 if (!IsExpressionFirstInfo) { 7969 // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well, 7970 // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags. 7971 if (IsPointer) 7972 Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS | 7973 OMP_MAP_DELETE | OMP_MAP_CLOSE); 7974 7975 if (ShouldBeMemberOf) { 7976 // Set placeholder value MEMBER_OF=FFFF to indicate that the flag 7977 // should be later updated with the correct value of MEMBER_OF. 7978 Flags |= OMP_MAP_MEMBER_OF; 7979 // From now on, all subsequent PTR_AND_OBJ entries should not be 7980 // marked as MEMBER_OF. 7981 ShouldBeMemberOf = false; 7982 } 7983 } 7984 7985 Types.push_back(Flags); 7986 } 7987 7988 // If we have encountered a member expression so far, keep track of the 7989 // mapped member. If the parent is "*this", then the value declaration 7990 // is nullptr. 7991 if (EncounteredME) { 7992 const auto *FD = cast<FieldDecl>(EncounteredME->getMemberDecl()); 7993 unsigned FieldIndex = FD->getFieldIndex(); 7994 7995 // Update info about the lowest and highest elements for this struct 7996 if (!PartialStruct.Base.isValid()) { 7997 PartialStruct.LowestElem = {FieldIndex, LB}; 7998 PartialStruct.HighestElem = {FieldIndex, LB}; 7999 PartialStruct.Base = BP; 8000 } else if (FieldIndex < PartialStruct.LowestElem.first) { 8001 PartialStruct.LowestElem = {FieldIndex, LB}; 8002 } else if (FieldIndex > PartialStruct.HighestElem.first) { 8003 PartialStruct.HighestElem = {FieldIndex, LB}; 8004 } 8005 } 8006 8007 // If we have a final array section, we are done with this expression. 8008 if (IsFinalArraySection) 8009 break; 8010 8011 // The pointer becomes the base for the next element. 8012 if (Next != CE) 8013 BP = LB; 8014 8015 IsExpressionFirstInfo = false; 8016 IsCaptureFirstInfo = false; 8017 } 8018 } 8019 } 8020 8021 /// Return the adjusted map modifiers if the declaration a capture refers to 8022 /// appears in a first-private clause. This is expected to be used only with 8023 /// directives that start with 'target'. 8024 MappableExprsHandler::OpenMPOffloadMappingFlags 8025 getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const { 8026 assert(Cap.capturesVariable() && "Expected capture by reference only!"); 8027 8028 // A first private variable captured by reference will use only the 8029 // 'private ptr' and 'map to' flag. Return the right flags if the captured 8030 // declaration is known as first-private in this handler. 8031 if (FirstPrivateDecls.count(Cap.getCapturedVar())) { 8032 if (Cap.getCapturedVar()->getType().isConstant(CGF.getContext()) && 8033 Cap.getCaptureKind() == CapturedStmt::VCK_ByRef) 8034 return MappableExprsHandler::OMP_MAP_ALWAYS | 8035 MappableExprsHandler::OMP_MAP_TO; 8036 if (Cap.getCapturedVar()->getType()->isAnyPointerType()) 8037 return MappableExprsHandler::OMP_MAP_TO | 8038 MappableExprsHandler::OMP_MAP_PTR_AND_OBJ; 8039 return MappableExprsHandler::OMP_MAP_PRIVATE | 8040 MappableExprsHandler::OMP_MAP_TO; 8041 } 8042 return MappableExprsHandler::OMP_MAP_TO | 8043 MappableExprsHandler::OMP_MAP_FROM; 8044 } 8045 8046 static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) { 8047 // Rotate by getFlagMemberOffset() bits. 8048 return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1) 8049 << getFlagMemberOffset()); 8050 } 8051 8052 static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags, 8053 OpenMPOffloadMappingFlags MemberOfFlag) { 8054 // If the entry is PTR_AND_OBJ but has not been marked with the special 8055 // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be 8056 // marked as MEMBER_OF. 8057 if ((Flags & OMP_MAP_PTR_AND_OBJ) && 8058 ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF)) 8059 return; 8060 8061 // Reset the placeholder value to prepare the flag for the assignment of the 8062 // proper MEMBER_OF value. 8063 Flags &= ~OMP_MAP_MEMBER_OF; 8064 Flags |= MemberOfFlag; 8065 } 8066 8067 void getPlainLayout(const CXXRecordDecl *RD, 8068 llvm::SmallVectorImpl<const FieldDecl *> &Layout, 8069 bool AsBase) const { 8070 const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD); 8071 8072 llvm::StructType *St = 8073 AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType(); 8074 8075 unsigned NumElements = St->getNumElements(); 8076 llvm::SmallVector< 8077 llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4> 8078 RecordLayout(NumElements); 8079 8080 // Fill bases. 8081 for (const auto &I : RD->bases()) { 8082 if (I.isVirtual()) 8083 continue; 8084 const auto *Base = I.getType()->getAsCXXRecordDecl(); 8085 // Ignore empty bases. 8086 if (Base->isEmpty() || CGF.getContext() 8087 .getASTRecordLayout(Base) 8088 .getNonVirtualSize() 8089 .isZero()) 8090 continue; 8091 8092 unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base); 8093 RecordLayout[FieldIndex] = Base; 8094 } 8095 // Fill in virtual bases. 8096 for (const auto &I : RD->vbases()) { 8097 const auto *Base = I.getType()->getAsCXXRecordDecl(); 8098 // Ignore empty bases. 8099 if (Base->isEmpty()) 8100 continue; 8101 unsigned FieldIndex = RL.getVirtualBaseIndex(Base); 8102 if (RecordLayout[FieldIndex]) 8103 continue; 8104 RecordLayout[FieldIndex] = Base; 8105 } 8106 // Fill in all the fields. 8107 assert(!RD->isUnion() && "Unexpected union."); 8108 for (const auto *Field : RD->fields()) { 8109 // Fill in non-bitfields. (Bitfields always use a zero pattern, which we 8110 // will fill in later.) 8111 if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) { 8112 unsigned FieldIndex = RL.getLLVMFieldNo(Field); 8113 RecordLayout[FieldIndex] = Field; 8114 } 8115 } 8116 for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *> 8117 &Data : RecordLayout) { 8118 if (Data.isNull()) 8119 continue; 8120 if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>()) 8121 getPlainLayout(Base, Layout, /*AsBase=*/true); 8122 else 8123 Layout.push_back(Data.get<const FieldDecl *>()); 8124 } 8125 } 8126 8127 public: 8128 MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF) 8129 : CurDir(&Dir), CGF(CGF) { 8130 // Extract firstprivate clause information. 8131 for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>()) 8132 for (const auto *D : C->varlists()) 8133 FirstPrivateDecls.try_emplace( 8134 cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit()); 8135 // Extract device pointer clause information. 8136 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>()) 8137 for (auto L : C->component_lists()) 8138 DevPointersMap[L.first].push_back(L.second); 8139 } 8140 8141 /// Constructor for the declare mapper directive. 8142 MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF) 8143 : CurDir(&Dir), CGF(CGF) {} 8144 8145 /// Generate code for the combined entry if we have a partially mapped struct 8146 /// and take care of the mapping flags of the arguments corresponding to 8147 /// individual struct members. 8148 void emitCombinedEntry(MapBaseValuesArrayTy &BasePointers, 8149 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 8150 MapFlagsArrayTy &Types, MapFlagsArrayTy &CurTypes, 8151 const StructRangeInfoTy &PartialStruct) const { 8152 // Base is the base of the struct 8153 BasePointers.push_back(PartialStruct.Base.getPointer()); 8154 // Pointer is the address of the lowest element 8155 llvm::Value *LB = PartialStruct.LowestElem.second.getPointer(); 8156 Pointers.push_back(LB); 8157 // Size is (addr of {highest+1} element) - (addr of lowest element) 8158 llvm::Value *HB = PartialStruct.HighestElem.second.getPointer(); 8159 llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(HB, /*Idx0=*/1); 8160 llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy); 8161 llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy); 8162 llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr); 8163 llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty, 8164 /*isSigned=*/false); 8165 Sizes.push_back(Size); 8166 // Map type is always TARGET_PARAM 8167 Types.push_back(OMP_MAP_TARGET_PARAM); 8168 // Remove TARGET_PARAM flag from the first element 8169 (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM; 8170 8171 // All other current entries will be MEMBER_OF the combined entry 8172 // (except for PTR_AND_OBJ entries which do not have a placeholder value 8173 // 0xFFFF in the MEMBER_OF field). 8174 OpenMPOffloadMappingFlags MemberOfFlag = 8175 getMemberOfFlag(BasePointers.size() - 1); 8176 for (auto &M : CurTypes) 8177 setCorrectMemberOfFlag(M, MemberOfFlag); 8178 } 8179 8180 /// Generate all the base pointers, section pointers, sizes and map 8181 /// types for the extracted mappable expressions. Also, for each item that 8182 /// relates with a device pointer, a pair of the relevant declaration and 8183 /// index where it occurs is appended to the device pointers info array. 8184 void generateAllInfo(MapBaseValuesArrayTy &BasePointers, 8185 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 8186 MapFlagsArrayTy &Types) const { 8187 // We have to process the component lists that relate with the same 8188 // declaration in a single chunk so that we can generate the map flags 8189 // correctly. Therefore, we organize all lists in a map. 8190 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 8191 8192 // Helper function to fill the information map for the different supported 8193 // clauses. 8194 auto &&InfoGen = [&Info]( 8195 const ValueDecl *D, 8196 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 8197 OpenMPMapClauseKind MapType, 8198 ArrayRef<OpenMPMapModifierKind> MapModifiers, 8199 bool ReturnDevicePointer, bool IsImplicit) { 8200 const ValueDecl *VD = 8201 D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr; 8202 Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer, 8203 IsImplicit); 8204 }; 8205 8206 assert(CurDir.is<const OMPExecutableDirective *>() && 8207 "Expect a executable directive"); 8208 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8209 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) 8210 for (const auto L : C->component_lists()) { 8211 InfoGen(L.first, L.second, C->getMapType(), C->getMapTypeModifiers(), 8212 /*ReturnDevicePointer=*/false, C->isImplicit()); 8213 } 8214 for (const auto *C : CurExecDir->getClausesOfKind<OMPToClause>()) 8215 for (const auto L : C->component_lists()) { 8216 InfoGen(L.first, L.second, OMPC_MAP_to, llvm::None, 8217 /*ReturnDevicePointer=*/false, C->isImplicit()); 8218 } 8219 for (const auto *C : CurExecDir->getClausesOfKind<OMPFromClause>()) 8220 for (const auto L : C->component_lists()) { 8221 InfoGen(L.first, L.second, OMPC_MAP_from, llvm::None, 8222 /*ReturnDevicePointer=*/false, C->isImplicit()); 8223 } 8224 8225 // Look at the use_device_ptr clause information and mark the existing map 8226 // entries as such. If there is no map information for an entry in the 8227 // use_device_ptr list, we create one with map type 'alloc' and zero size 8228 // section. It is the user fault if that was not mapped before. If there is 8229 // no map information and the pointer is a struct member, then we defer the 8230 // emission of that entry until the whole struct has been processed. 8231 llvm::MapVector<const ValueDecl *, SmallVector<DeferredDevicePtrEntryTy, 4>> 8232 DeferredInfo; 8233 8234 for (const auto *C : 8235 CurExecDir->getClausesOfKind<OMPUseDevicePtrClause>()) { 8236 for (const auto L : C->component_lists()) { 8237 assert(!L.second.empty() && "Not expecting empty list of components!"); 8238 const ValueDecl *VD = L.second.back().getAssociatedDeclaration(); 8239 VD = cast<ValueDecl>(VD->getCanonicalDecl()); 8240 const Expr *IE = L.second.back().getAssociatedExpression(); 8241 // If the first component is a member expression, we have to look into 8242 // 'this', which maps to null in the map of map information. Otherwise 8243 // look directly for the information. 8244 auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD); 8245 8246 // We potentially have map information for this declaration already. 8247 // Look for the first set of components that refer to it. 8248 if (It != Info.end()) { 8249 auto CI = std::find_if( 8250 It->second.begin(), It->second.end(), [VD](const MapInfo &MI) { 8251 return MI.Components.back().getAssociatedDeclaration() == VD; 8252 }); 8253 // If we found a map entry, signal that the pointer has to be returned 8254 // and move on to the next declaration. 8255 if (CI != It->second.end()) { 8256 CI->ReturnDevicePointer = true; 8257 continue; 8258 } 8259 } 8260 8261 // We didn't find any match in our map information - generate a zero 8262 // size array section - if the pointer is a struct member we defer this 8263 // action until the whole struct has been processed. 8264 if (isa<MemberExpr>(IE)) { 8265 // Insert the pointer into Info to be processed by 8266 // generateInfoForComponentList. Because it is a member pointer 8267 // without a pointee, no entry will be generated for it, therefore 8268 // we need to generate one after the whole struct has been processed. 8269 // Nonetheless, generateInfoForComponentList must be called to take 8270 // the pointer into account for the calculation of the range of the 8271 // partial struct. 8272 InfoGen(nullptr, L.second, OMPC_MAP_unknown, llvm::None, 8273 /*ReturnDevicePointer=*/false, C->isImplicit()); 8274 DeferredInfo[nullptr].emplace_back(IE, VD); 8275 } else { 8276 llvm::Value *Ptr = 8277 CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc()); 8278 BasePointers.emplace_back(Ptr, VD); 8279 Pointers.push_back(Ptr); 8280 Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty)); 8281 Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_TARGET_PARAM); 8282 } 8283 } 8284 } 8285 8286 for (const auto &M : Info) { 8287 // We need to know when we generate information for the first component 8288 // associated with a capture, because the mapping flags depend on it. 8289 bool IsFirstComponentList = true; 8290 8291 // Temporary versions of arrays 8292 MapBaseValuesArrayTy CurBasePointers; 8293 MapValuesArrayTy CurPointers; 8294 MapValuesArrayTy CurSizes; 8295 MapFlagsArrayTy CurTypes; 8296 StructRangeInfoTy PartialStruct; 8297 8298 for (const MapInfo &L : M.second) { 8299 assert(!L.Components.empty() && 8300 "Not expecting declaration with no component lists."); 8301 8302 // Remember the current base pointer index. 8303 unsigned CurrentBasePointersIdx = CurBasePointers.size(); 8304 generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components, 8305 CurBasePointers, CurPointers, CurSizes, 8306 CurTypes, PartialStruct, 8307 IsFirstComponentList, L.IsImplicit); 8308 8309 // If this entry relates with a device pointer, set the relevant 8310 // declaration and add the 'return pointer' flag. 8311 if (L.ReturnDevicePointer) { 8312 assert(CurBasePointers.size() > CurrentBasePointersIdx && 8313 "Unexpected number of mapped base pointers."); 8314 8315 const ValueDecl *RelevantVD = 8316 L.Components.back().getAssociatedDeclaration(); 8317 assert(RelevantVD && 8318 "No relevant declaration related with device pointer??"); 8319 8320 CurBasePointers[CurrentBasePointersIdx].setDevicePtrDecl(RelevantVD); 8321 CurTypes[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM; 8322 } 8323 IsFirstComponentList = false; 8324 } 8325 8326 // Append any pending zero-length pointers which are struct members and 8327 // used with use_device_ptr. 8328 auto CI = DeferredInfo.find(M.first); 8329 if (CI != DeferredInfo.end()) { 8330 for (const DeferredDevicePtrEntryTy &L : CI->second) { 8331 llvm::Value *BasePtr = this->CGF.EmitLValue(L.IE).getPointer(CGF); 8332 llvm::Value *Ptr = this->CGF.EmitLoadOfScalar( 8333 this->CGF.EmitLValue(L.IE), L.IE->getExprLoc()); 8334 CurBasePointers.emplace_back(BasePtr, L.VD); 8335 CurPointers.push_back(Ptr); 8336 CurSizes.push_back(llvm::Constant::getNullValue(this->CGF.Int64Ty)); 8337 // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the placeholder 8338 // value MEMBER_OF=FFFF so that the entry is later updated with the 8339 // correct value of MEMBER_OF. 8340 CurTypes.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM | 8341 OMP_MAP_MEMBER_OF); 8342 } 8343 } 8344 8345 // If there is an entry in PartialStruct it means we have a struct with 8346 // individual members mapped. Emit an extra combined entry. 8347 if (PartialStruct.Base.isValid()) 8348 emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes, 8349 PartialStruct); 8350 8351 // We need to append the results of this capture to what we already have. 8352 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 8353 Pointers.append(CurPointers.begin(), CurPointers.end()); 8354 Sizes.append(CurSizes.begin(), CurSizes.end()); 8355 Types.append(CurTypes.begin(), CurTypes.end()); 8356 } 8357 } 8358 8359 /// Generate all the base pointers, section pointers, sizes and map types for 8360 /// the extracted map clauses of user-defined mapper. 8361 void generateAllInfoForMapper(MapBaseValuesArrayTy &BasePointers, 8362 MapValuesArrayTy &Pointers, 8363 MapValuesArrayTy &Sizes, 8364 MapFlagsArrayTy &Types) const { 8365 assert(CurDir.is<const OMPDeclareMapperDecl *>() && 8366 "Expect a declare mapper directive"); 8367 const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>(); 8368 // We have to process the component lists that relate with the same 8369 // declaration in a single chunk so that we can generate the map flags 8370 // correctly. Therefore, we organize all lists in a map. 8371 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 8372 8373 // Helper function to fill the information map for the different supported 8374 // clauses. 8375 auto &&InfoGen = [&Info]( 8376 const ValueDecl *D, 8377 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 8378 OpenMPMapClauseKind MapType, 8379 ArrayRef<OpenMPMapModifierKind> MapModifiers, 8380 bool ReturnDevicePointer, bool IsImplicit) { 8381 const ValueDecl *VD = 8382 D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr; 8383 Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer, 8384 IsImplicit); 8385 }; 8386 8387 for (const auto *C : CurMapperDir->clauselists()) { 8388 const auto *MC = cast<OMPMapClause>(C); 8389 for (const auto L : MC->component_lists()) { 8390 InfoGen(L.first, L.second, MC->getMapType(), MC->getMapTypeModifiers(), 8391 /*ReturnDevicePointer=*/false, MC->isImplicit()); 8392 } 8393 } 8394 8395 for (const auto &M : Info) { 8396 // We need to know when we generate information for the first component 8397 // associated with a capture, because the mapping flags depend on it. 8398 bool IsFirstComponentList = true; 8399 8400 // Temporary versions of arrays 8401 MapBaseValuesArrayTy CurBasePointers; 8402 MapValuesArrayTy CurPointers; 8403 MapValuesArrayTy CurSizes; 8404 MapFlagsArrayTy CurTypes; 8405 StructRangeInfoTy PartialStruct; 8406 8407 for (const MapInfo &L : M.second) { 8408 assert(!L.Components.empty() && 8409 "Not expecting declaration with no component lists."); 8410 generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components, 8411 CurBasePointers, CurPointers, CurSizes, 8412 CurTypes, PartialStruct, 8413 IsFirstComponentList, L.IsImplicit); 8414 IsFirstComponentList = false; 8415 } 8416 8417 // If there is an entry in PartialStruct it means we have a struct with 8418 // individual members mapped. Emit an extra combined entry. 8419 if (PartialStruct.Base.isValid()) 8420 emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes, 8421 PartialStruct); 8422 8423 // We need to append the results of this capture to what we already have. 8424 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 8425 Pointers.append(CurPointers.begin(), CurPointers.end()); 8426 Sizes.append(CurSizes.begin(), CurSizes.end()); 8427 Types.append(CurTypes.begin(), CurTypes.end()); 8428 } 8429 } 8430 8431 /// Emit capture info for lambdas for variables captured by reference. 8432 void generateInfoForLambdaCaptures( 8433 const ValueDecl *VD, llvm::Value *Arg, MapBaseValuesArrayTy &BasePointers, 8434 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 8435 MapFlagsArrayTy &Types, 8436 llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const { 8437 const auto *RD = VD->getType() 8438 .getCanonicalType() 8439 .getNonReferenceType() 8440 ->getAsCXXRecordDecl(); 8441 if (!RD || !RD->isLambda()) 8442 return; 8443 Address VDAddr = Address(Arg, CGF.getContext().getDeclAlign(VD)); 8444 LValue VDLVal = CGF.MakeAddrLValue( 8445 VDAddr, VD->getType().getCanonicalType().getNonReferenceType()); 8446 llvm::DenseMap<const VarDecl *, FieldDecl *> Captures; 8447 FieldDecl *ThisCapture = nullptr; 8448 RD->getCaptureFields(Captures, ThisCapture); 8449 if (ThisCapture) { 8450 LValue ThisLVal = 8451 CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture); 8452 LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture); 8453 LambdaPointers.try_emplace(ThisLVal.getPointer(CGF), 8454 VDLVal.getPointer(CGF)); 8455 BasePointers.push_back(ThisLVal.getPointer(CGF)); 8456 Pointers.push_back(ThisLValVal.getPointer(CGF)); 8457 Sizes.push_back( 8458 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy), 8459 CGF.Int64Ty, /*isSigned=*/true)); 8460 Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8461 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 8462 } 8463 for (const LambdaCapture &LC : RD->captures()) { 8464 if (!LC.capturesVariable()) 8465 continue; 8466 const VarDecl *VD = LC.getCapturedVar(); 8467 if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType()) 8468 continue; 8469 auto It = Captures.find(VD); 8470 assert(It != Captures.end() && "Found lambda capture without field."); 8471 LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second); 8472 if (LC.getCaptureKind() == LCK_ByRef) { 8473 LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second); 8474 LambdaPointers.try_emplace(VarLVal.getPointer(CGF), 8475 VDLVal.getPointer(CGF)); 8476 BasePointers.push_back(VarLVal.getPointer(CGF)); 8477 Pointers.push_back(VarLValVal.getPointer(CGF)); 8478 Sizes.push_back(CGF.Builder.CreateIntCast( 8479 CGF.getTypeSize( 8480 VD->getType().getCanonicalType().getNonReferenceType()), 8481 CGF.Int64Ty, /*isSigned=*/true)); 8482 } else { 8483 RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation()); 8484 LambdaPointers.try_emplace(VarLVal.getPointer(CGF), 8485 VDLVal.getPointer(CGF)); 8486 BasePointers.push_back(VarLVal.getPointer(CGF)); 8487 Pointers.push_back(VarRVal.getScalarVal()); 8488 Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0)); 8489 } 8490 Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8491 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 8492 } 8493 } 8494 8495 /// Set correct indices for lambdas captures. 8496 void adjustMemberOfForLambdaCaptures( 8497 const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers, 8498 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 8499 MapFlagsArrayTy &Types) const { 8500 for (unsigned I = 0, E = Types.size(); I < E; ++I) { 8501 // Set correct member_of idx for all implicit lambda captures. 8502 if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8503 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT)) 8504 continue; 8505 llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]); 8506 assert(BasePtr && "Unable to find base lambda address."); 8507 int TgtIdx = -1; 8508 for (unsigned J = I; J > 0; --J) { 8509 unsigned Idx = J - 1; 8510 if (Pointers[Idx] != BasePtr) 8511 continue; 8512 TgtIdx = Idx; 8513 break; 8514 } 8515 assert(TgtIdx != -1 && "Unable to find parent lambda."); 8516 // All other current entries will be MEMBER_OF the combined entry 8517 // (except for PTR_AND_OBJ entries which do not have a placeholder value 8518 // 0xFFFF in the MEMBER_OF field). 8519 OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx); 8520 setCorrectMemberOfFlag(Types[I], MemberOfFlag); 8521 } 8522 } 8523 8524 /// Generate the base pointers, section pointers, sizes and map types 8525 /// associated to a given capture. 8526 void generateInfoForCapture(const CapturedStmt::Capture *Cap, 8527 llvm::Value *Arg, 8528 MapBaseValuesArrayTy &BasePointers, 8529 MapValuesArrayTy &Pointers, 8530 MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types, 8531 StructRangeInfoTy &PartialStruct) const { 8532 assert(!Cap->capturesVariableArrayType() && 8533 "Not expecting to generate map info for a variable array type!"); 8534 8535 // We need to know when we generating information for the first component 8536 const ValueDecl *VD = Cap->capturesThis() 8537 ? nullptr 8538 : Cap->getCapturedVar()->getCanonicalDecl(); 8539 8540 // If this declaration appears in a is_device_ptr clause we just have to 8541 // pass the pointer by value. If it is a reference to a declaration, we just 8542 // pass its value. 8543 if (DevPointersMap.count(VD)) { 8544 BasePointers.emplace_back(Arg, VD); 8545 Pointers.push_back(Arg); 8546 Sizes.push_back( 8547 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy), 8548 CGF.Int64Ty, /*isSigned=*/true)); 8549 Types.push_back(OMP_MAP_LITERAL | OMP_MAP_TARGET_PARAM); 8550 return; 8551 } 8552 8553 using MapData = 8554 std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef, 8555 OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool>; 8556 SmallVector<MapData, 4> DeclComponentLists; 8557 assert(CurDir.is<const OMPExecutableDirective *>() && 8558 "Expect a executable directive"); 8559 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8560 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) { 8561 for (const auto L : C->decl_component_lists(VD)) { 8562 assert(L.first == VD && 8563 "We got information for the wrong declaration??"); 8564 assert(!L.second.empty() && 8565 "Not expecting declaration with no component lists."); 8566 DeclComponentLists.emplace_back(L.second, C->getMapType(), 8567 C->getMapTypeModifiers(), 8568 C->isImplicit()); 8569 } 8570 } 8571 8572 // Find overlapping elements (including the offset from the base element). 8573 llvm::SmallDenseMap< 8574 const MapData *, 8575 llvm::SmallVector< 8576 OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>, 8577 4> 8578 OverlappedData; 8579 size_t Count = 0; 8580 for (const MapData &L : DeclComponentLists) { 8581 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8582 OpenMPMapClauseKind MapType; 8583 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8584 bool IsImplicit; 8585 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8586 ++Count; 8587 for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) { 8588 OMPClauseMappableExprCommon::MappableExprComponentListRef Components1; 8589 std::tie(Components1, MapType, MapModifiers, IsImplicit) = L1; 8590 auto CI = Components.rbegin(); 8591 auto CE = Components.rend(); 8592 auto SI = Components1.rbegin(); 8593 auto SE = Components1.rend(); 8594 for (; CI != CE && SI != SE; ++CI, ++SI) { 8595 if (CI->getAssociatedExpression()->getStmtClass() != 8596 SI->getAssociatedExpression()->getStmtClass()) 8597 break; 8598 // Are we dealing with different variables/fields? 8599 if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration()) 8600 break; 8601 } 8602 // Found overlapping if, at least for one component, reached the head of 8603 // the components list. 8604 if (CI == CE || SI == SE) { 8605 assert((CI != CE || SI != SE) && 8606 "Unexpected full match of the mapping components."); 8607 const MapData &BaseData = CI == CE ? L : L1; 8608 OMPClauseMappableExprCommon::MappableExprComponentListRef SubData = 8609 SI == SE ? Components : Components1; 8610 auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData); 8611 OverlappedElements.getSecond().push_back(SubData); 8612 } 8613 } 8614 } 8615 // Sort the overlapped elements for each item. 8616 llvm::SmallVector<const FieldDecl *, 4> Layout; 8617 if (!OverlappedData.empty()) { 8618 if (const auto *CRD = 8619 VD->getType().getCanonicalType()->getAsCXXRecordDecl()) 8620 getPlainLayout(CRD, Layout, /*AsBase=*/false); 8621 else { 8622 const auto *RD = VD->getType().getCanonicalType()->getAsRecordDecl(); 8623 Layout.append(RD->field_begin(), RD->field_end()); 8624 } 8625 } 8626 for (auto &Pair : OverlappedData) { 8627 llvm::sort( 8628 Pair.getSecond(), 8629 [&Layout]( 8630 OMPClauseMappableExprCommon::MappableExprComponentListRef First, 8631 OMPClauseMappableExprCommon::MappableExprComponentListRef 8632 Second) { 8633 auto CI = First.rbegin(); 8634 auto CE = First.rend(); 8635 auto SI = Second.rbegin(); 8636 auto SE = Second.rend(); 8637 for (; CI != CE && SI != SE; ++CI, ++SI) { 8638 if (CI->getAssociatedExpression()->getStmtClass() != 8639 SI->getAssociatedExpression()->getStmtClass()) 8640 break; 8641 // Are we dealing with different variables/fields? 8642 if (CI->getAssociatedDeclaration() != 8643 SI->getAssociatedDeclaration()) 8644 break; 8645 } 8646 8647 // Lists contain the same elements. 8648 if (CI == CE && SI == SE) 8649 return false; 8650 8651 // List with less elements is less than list with more elements. 8652 if (CI == CE || SI == SE) 8653 return CI == CE; 8654 8655 const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration()); 8656 const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration()); 8657 if (FD1->getParent() == FD2->getParent()) 8658 return FD1->getFieldIndex() < FD2->getFieldIndex(); 8659 const auto It = 8660 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) { 8661 return FD == FD1 || FD == FD2; 8662 }); 8663 return *It == FD1; 8664 }); 8665 } 8666 8667 // Associated with a capture, because the mapping flags depend on it. 8668 // Go through all of the elements with the overlapped elements. 8669 for (const auto &Pair : OverlappedData) { 8670 const MapData &L = *Pair.getFirst(); 8671 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8672 OpenMPMapClauseKind MapType; 8673 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8674 bool IsImplicit; 8675 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8676 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 8677 OverlappedComponents = Pair.getSecond(); 8678 bool IsFirstComponentList = true; 8679 generateInfoForComponentList(MapType, MapModifiers, Components, 8680 BasePointers, Pointers, Sizes, Types, 8681 PartialStruct, IsFirstComponentList, 8682 IsImplicit, OverlappedComponents); 8683 } 8684 // Go through other elements without overlapped elements. 8685 bool IsFirstComponentList = OverlappedData.empty(); 8686 for (const MapData &L : DeclComponentLists) { 8687 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8688 OpenMPMapClauseKind MapType; 8689 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8690 bool IsImplicit; 8691 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8692 auto It = OverlappedData.find(&L); 8693 if (It == OverlappedData.end()) 8694 generateInfoForComponentList(MapType, MapModifiers, Components, 8695 BasePointers, Pointers, Sizes, Types, 8696 PartialStruct, IsFirstComponentList, 8697 IsImplicit); 8698 IsFirstComponentList = false; 8699 } 8700 } 8701 8702 /// Generate the base pointers, section pointers, sizes and map types 8703 /// associated with the declare target link variables. 8704 void generateInfoForDeclareTargetLink(MapBaseValuesArrayTy &BasePointers, 8705 MapValuesArrayTy &Pointers, 8706 MapValuesArrayTy &Sizes, 8707 MapFlagsArrayTy &Types) const { 8708 assert(CurDir.is<const OMPExecutableDirective *>() && 8709 "Expect a executable directive"); 8710 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8711 // Map other list items in the map clause which are not captured variables 8712 // but "declare target link" global variables. 8713 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) { 8714 for (const auto L : C->component_lists()) { 8715 if (!L.first) 8716 continue; 8717 const auto *VD = dyn_cast<VarDecl>(L.first); 8718 if (!VD) 8719 continue; 8720 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 8721 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 8722 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() || 8723 !Res || *Res != OMPDeclareTargetDeclAttr::MT_Link) 8724 continue; 8725 StructRangeInfoTy PartialStruct; 8726 generateInfoForComponentList( 8727 C->getMapType(), C->getMapTypeModifiers(), L.second, BasePointers, 8728 Pointers, Sizes, Types, PartialStruct, 8729 /*IsFirstComponentList=*/true, C->isImplicit()); 8730 assert(!PartialStruct.Base.isValid() && 8731 "No partial structs for declare target link expected."); 8732 } 8733 } 8734 } 8735 8736 /// Generate the default map information for a given capture \a CI, 8737 /// record field declaration \a RI and captured value \a CV. 8738 void generateDefaultMapInfo(const CapturedStmt::Capture &CI, 8739 const FieldDecl &RI, llvm::Value *CV, 8740 MapBaseValuesArrayTy &CurBasePointers, 8741 MapValuesArrayTy &CurPointers, 8742 MapValuesArrayTy &CurSizes, 8743 MapFlagsArrayTy &CurMapTypes) const { 8744 bool IsImplicit = true; 8745 // Do the default mapping. 8746 if (CI.capturesThis()) { 8747 CurBasePointers.push_back(CV); 8748 CurPointers.push_back(CV); 8749 const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr()); 8750 CurSizes.push_back( 8751 CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()), 8752 CGF.Int64Ty, /*isSigned=*/true)); 8753 // Default map type. 8754 CurMapTypes.push_back(OMP_MAP_TO | OMP_MAP_FROM); 8755 } else if (CI.capturesVariableByCopy()) { 8756 CurBasePointers.push_back(CV); 8757 CurPointers.push_back(CV); 8758 if (!RI.getType()->isAnyPointerType()) { 8759 // We have to signal to the runtime captures passed by value that are 8760 // not pointers. 8761 CurMapTypes.push_back(OMP_MAP_LITERAL); 8762 CurSizes.push_back(CGF.Builder.CreateIntCast( 8763 CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true)); 8764 } else { 8765 // Pointers are implicitly mapped with a zero size and no flags 8766 // (other than first map that is added for all implicit maps). 8767 CurMapTypes.push_back(OMP_MAP_NONE); 8768 CurSizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty)); 8769 } 8770 const VarDecl *VD = CI.getCapturedVar(); 8771 auto I = FirstPrivateDecls.find(VD); 8772 if (I != FirstPrivateDecls.end()) 8773 IsImplicit = I->getSecond(); 8774 } else { 8775 assert(CI.capturesVariable() && "Expected captured reference."); 8776 const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr()); 8777 QualType ElementType = PtrTy->getPointeeType(); 8778 CurSizes.push_back(CGF.Builder.CreateIntCast( 8779 CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true)); 8780 // The default map type for a scalar/complex type is 'to' because by 8781 // default the value doesn't have to be retrieved. For an aggregate 8782 // type, the default is 'tofrom'. 8783 CurMapTypes.push_back(getMapModifiersForPrivateClauses(CI)); 8784 const VarDecl *VD = CI.getCapturedVar(); 8785 auto I = FirstPrivateDecls.find(VD); 8786 if (I != FirstPrivateDecls.end() && 8787 VD->getType().isConstant(CGF.getContext())) { 8788 llvm::Constant *Addr = 8789 CGF.CGM.getOpenMPRuntime().registerTargetFirstprivateCopy(CGF, VD); 8790 // Copy the value of the original variable to the new global copy. 8791 CGF.Builder.CreateMemCpy( 8792 CGF.MakeNaturalAlignAddrLValue(Addr, ElementType).getAddress(CGF), 8793 Address(CV, CGF.getContext().getTypeAlignInChars(ElementType)), 8794 CurSizes.back(), /*IsVolatile=*/false); 8795 // Use new global variable as the base pointers. 8796 CurBasePointers.push_back(Addr); 8797 CurPointers.push_back(Addr); 8798 } else { 8799 CurBasePointers.push_back(CV); 8800 if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) { 8801 Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue( 8802 CV, ElementType, CGF.getContext().getDeclAlign(VD), 8803 AlignmentSource::Decl)); 8804 CurPointers.push_back(PtrAddr.getPointer()); 8805 } else { 8806 CurPointers.push_back(CV); 8807 } 8808 } 8809 if (I != FirstPrivateDecls.end()) 8810 IsImplicit = I->getSecond(); 8811 } 8812 // Every default map produces a single argument which is a target parameter. 8813 CurMapTypes.back() |= OMP_MAP_TARGET_PARAM; 8814 8815 // Add flag stating this is an implicit map. 8816 if (IsImplicit) 8817 CurMapTypes.back() |= OMP_MAP_IMPLICIT; 8818 } 8819 }; 8820 } // anonymous namespace 8821 8822 /// Emit the arrays used to pass the captures and map information to the 8823 /// offloading runtime library. If there is no map or capture information, 8824 /// return nullptr by reference. 8825 static void 8826 emitOffloadingArrays(CodeGenFunction &CGF, 8827 MappableExprsHandler::MapBaseValuesArrayTy &BasePointers, 8828 MappableExprsHandler::MapValuesArrayTy &Pointers, 8829 MappableExprsHandler::MapValuesArrayTy &Sizes, 8830 MappableExprsHandler::MapFlagsArrayTy &MapTypes, 8831 CGOpenMPRuntime::TargetDataInfo &Info) { 8832 CodeGenModule &CGM = CGF.CGM; 8833 ASTContext &Ctx = CGF.getContext(); 8834 8835 // Reset the array information. 8836 Info.clearArrayInfo(); 8837 Info.NumberOfPtrs = BasePointers.size(); 8838 8839 if (Info.NumberOfPtrs) { 8840 // Detect if we have any capture size requiring runtime evaluation of the 8841 // size so that a constant array could be eventually used. 8842 bool hasRuntimeEvaluationCaptureSize = false; 8843 for (llvm::Value *S : Sizes) 8844 if (!isa<llvm::Constant>(S)) { 8845 hasRuntimeEvaluationCaptureSize = true; 8846 break; 8847 } 8848 8849 llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true); 8850 QualType PointerArrayType = Ctx.getConstantArrayType( 8851 Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal, 8852 /*IndexTypeQuals=*/0); 8853 8854 Info.BasePointersArray = 8855 CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer(); 8856 Info.PointersArray = 8857 CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer(); 8858 8859 // If we don't have any VLA types or other types that require runtime 8860 // evaluation, we can use a constant array for the map sizes, otherwise we 8861 // need to fill up the arrays as we do for the pointers. 8862 QualType Int64Ty = 8863 Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 8864 if (hasRuntimeEvaluationCaptureSize) { 8865 QualType SizeArrayType = Ctx.getConstantArrayType( 8866 Int64Ty, PointerNumAP, nullptr, ArrayType::Normal, 8867 /*IndexTypeQuals=*/0); 8868 Info.SizesArray = 8869 CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer(); 8870 } else { 8871 // We expect all the sizes to be constant, so we collect them to create 8872 // a constant array. 8873 SmallVector<llvm::Constant *, 16> ConstSizes; 8874 for (llvm::Value *S : Sizes) 8875 ConstSizes.push_back(cast<llvm::Constant>(S)); 8876 8877 auto *SizesArrayInit = llvm::ConstantArray::get( 8878 llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes); 8879 std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"}); 8880 auto *SizesArrayGbl = new llvm::GlobalVariable( 8881 CGM.getModule(), SizesArrayInit->getType(), 8882 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 8883 SizesArrayInit, Name); 8884 SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 8885 Info.SizesArray = SizesArrayGbl; 8886 } 8887 8888 // The map types are always constant so we don't need to generate code to 8889 // fill arrays. Instead, we create an array constant. 8890 SmallVector<uint64_t, 4> Mapping(MapTypes.size(), 0); 8891 llvm::copy(MapTypes, Mapping.begin()); 8892 llvm::Constant *MapTypesArrayInit = 8893 llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping); 8894 std::string MaptypesName = 8895 CGM.getOpenMPRuntime().getName({"offload_maptypes"}); 8896 auto *MapTypesArrayGbl = new llvm::GlobalVariable( 8897 CGM.getModule(), MapTypesArrayInit->getType(), 8898 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 8899 MapTypesArrayInit, MaptypesName); 8900 MapTypesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 8901 Info.MapTypesArray = MapTypesArrayGbl; 8902 8903 for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) { 8904 llvm::Value *BPVal = *BasePointers[I]; 8905 llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32( 8906 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8907 Info.BasePointersArray, 0, I); 8908 BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8909 BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0)); 8910 Address BPAddr(BP, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 8911 CGF.Builder.CreateStore(BPVal, BPAddr); 8912 8913 if (Info.requiresDevicePointerInfo()) 8914 if (const ValueDecl *DevVD = BasePointers[I].getDevicePtrDecl()) 8915 Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr); 8916 8917 llvm::Value *PVal = Pointers[I]; 8918 llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32( 8919 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8920 Info.PointersArray, 0, I); 8921 P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8922 P, PVal->getType()->getPointerTo(/*AddrSpace=*/0)); 8923 Address PAddr(P, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 8924 CGF.Builder.CreateStore(PVal, PAddr); 8925 8926 if (hasRuntimeEvaluationCaptureSize) { 8927 llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32( 8928 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 8929 Info.SizesArray, 8930 /*Idx0=*/0, 8931 /*Idx1=*/I); 8932 Address SAddr(S, Ctx.getTypeAlignInChars(Int64Ty)); 8933 CGF.Builder.CreateStore( 8934 CGF.Builder.CreateIntCast(Sizes[I], CGM.Int64Ty, /*isSigned=*/true), 8935 SAddr); 8936 } 8937 } 8938 } 8939 } 8940 8941 /// Emit the arguments to be passed to the runtime library based on the 8942 /// arrays of pointers, sizes and map types. 8943 static void emitOffloadingArraysArgument( 8944 CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg, 8945 llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg, 8946 llvm::Value *&MapTypesArrayArg, CGOpenMPRuntime::TargetDataInfo &Info) { 8947 CodeGenModule &CGM = CGF.CGM; 8948 if (Info.NumberOfPtrs) { 8949 BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8950 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8951 Info.BasePointersArray, 8952 /*Idx0=*/0, /*Idx1=*/0); 8953 PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8954 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8955 Info.PointersArray, 8956 /*Idx0=*/0, 8957 /*Idx1=*/0); 8958 SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8959 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray, 8960 /*Idx0=*/0, /*Idx1=*/0); 8961 MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8962 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 8963 Info.MapTypesArray, 8964 /*Idx0=*/0, 8965 /*Idx1=*/0); 8966 } else { 8967 BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 8968 PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 8969 SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 8970 MapTypesArrayArg = 8971 llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 8972 } 8973 } 8974 8975 /// Check for inner distribute directive. 8976 static const OMPExecutableDirective * 8977 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) { 8978 const auto *CS = D.getInnermostCapturedStmt(); 8979 const auto *Body = 8980 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 8981 const Stmt *ChildStmt = 8982 CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 8983 8984 if (const auto *NestedDir = 8985 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 8986 OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind(); 8987 switch (D.getDirectiveKind()) { 8988 case OMPD_target: 8989 if (isOpenMPDistributeDirective(DKind)) 8990 return NestedDir; 8991 if (DKind == OMPD_teams) { 8992 Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers( 8993 /*IgnoreCaptured=*/true); 8994 if (!Body) 8995 return nullptr; 8996 ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 8997 if (const auto *NND = 8998 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 8999 DKind = NND->getDirectiveKind(); 9000 if (isOpenMPDistributeDirective(DKind)) 9001 return NND; 9002 } 9003 } 9004 return nullptr; 9005 case OMPD_target_teams: 9006 if (isOpenMPDistributeDirective(DKind)) 9007 return NestedDir; 9008 return nullptr; 9009 case OMPD_target_parallel: 9010 case OMPD_target_simd: 9011 case OMPD_target_parallel_for: 9012 case OMPD_target_parallel_for_simd: 9013 return nullptr; 9014 case OMPD_target_teams_distribute: 9015 case OMPD_target_teams_distribute_simd: 9016 case OMPD_target_teams_distribute_parallel_for: 9017 case OMPD_target_teams_distribute_parallel_for_simd: 9018 case OMPD_parallel: 9019 case OMPD_for: 9020 case OMPD_parallel_for: 9021 case OMPD_parallel_master: 9022 case OMPD_parallel_sections: 9023 case OMPD_for_simd: 9024 case OMPD_parallel_for_simd: 9025 case OMPD_cancel: 9026 case OMPD_cancellation_point: 9027 case OMPD_ordered: 9028 case OMPD_threadprivate: 9029 case OMPD_allocate: 9030 case OMPD_task: 9031 case OMPD_simd: 9032 case OMPD_sections: 9033 case OMPD_section: 9034 case OMPD_single: 9035 case OMPD_master: 9036 case OMPD_critical: 9037 case OMPD_taskyield: 9038 case OMPD_barrier: 9039 case OMPD_taskwait: 9040 case OMPD_taskgroup: 9041 case OMPD_atomic: 9042 case OMPD_flush: 9043 case OMPD_depobj: 9044 case OMPD_scan: 9045 case OMPD_teams: 9046 case OMPD_target_data: 9047 case OMPD_target_exit_data: 9048 case OMPD_target_enter_data: 9049 case OMPD_distribute: 9050 case OMPD_distribute_simd: 9051 case OMPD_distribute_parallel_for: 9052 case OMPD_distribute_parallel_for_simd: 9053 case OMPD_teams_distribute: 9054 case OMPD_teams_distribute_simd: 9055 case OMPD_teams_distribute_parallel_for: 9056 case OMPD_teams_distribute_parallel_for_simd: 9057 case OMPD_target_update: 9058 case OMPD_declare_simd: 9059 case OMPD_declare_variant: 9060 case OMPD_begin_declare_variant: 9061 case OMPD_end_declare_variant: 9062 case OMPD_declare_target: 9063 case OMPD_end_declare_target: 9064 case OMPD_declare_reduction: 9065 case OMPD_declare_mapper: 9066 case OMPD_taskloop: 9067 case OMPD_taskloop_simd: 9068 case OMPD_master_taskloop: 9069 case OMPD_master_taskloop_simd: 9070 case OMPD_parallel_master_taskloop: 9071 case OMPD_parallel_master_taskloop_simd: 9072 case OMPD_requires: 9073 case OMPD_unknown: 9074 llvm_unreachable("Unexpected directive."); 9075 } 9076 } 9077 9078 return nullptr; 9079 } 9080 9081 /// Emit the user-defined mapper function. The code generation follows the 9082 /// pattern in the example below. 9083 /// \code 9084 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle, 9085 /// void *base, void *begin, 9086 /// int64_t size, int64_t type) { 9087 /// // Allocate space for an array section first. 9088 /// if (size > 1 && !maptype.IsDelete) 9089 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 9090 /// size*sizeof(Ty), clearToFrom(type)); 9091 /// // Map members. 9092 /// for (unsigned i = 0; i < size; i++) { 9093 /// // For each component specified by this mapper: 9094 /// for (auto c : all_components) { 9095 /// if (c.hasMapper()) 9096 /// (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size, 9097 /// c.arg_type); 9098 /// else 9099 /// __tgt_push_mapper_component(rt_mapper_handle, c.arg_base, 9100 /// c.arg_begin, c.arg_size, c.arg_type); 9101 /// } 9102 /// } 9103 /// // Delete the array section. 9104 /// if (size > 1 && maptype.IsDelete) 9105 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 9106 /// size*sizeof(Ty), clearToFrom(type)); 9107 /// } 9108 /// \endcode 9109 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D, 9110 CodeGenFunction *CGF) { 9111 if (UDMMap.count(D) > 0) 9112 return; 9113 ASTContext &C = CGM.getContext(); 9114 QualType Ty = D->getType(); 9115 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 9116 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 9117 auto *MapperVarDecl = 9118 cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl()); 9119 SourceLocation Loc = D->getLocation(); 9120 CharUnits ElementSize = C.getTypeSizeInChars(Ty); 9121 9122 // Prepare mapper function arguments and attributes. 9123 ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 9124 C.VoidPtrTy, ImplicitParamDecl::Other); 9125 ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 9126 ImplicitParamDecl::Other); 9127 ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 9128 C.VoidPtrTy, ImplicitParamDecl::Other); 9129 ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 9130 ImplicitParamDecl::Other); 9131 ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 9132 ImplicitParamDecl::Other); 9133 FunctionArgList Args; 9134 Args.push_back(&HandleArg); 9135 Args.push_back(&BaseArg); 9136 Args.push_back(&BeginArg); 9137 Args.push_back(&SizeArg); 9138 Args.push_back(&TypeArg); 9139 const CGFunctionInfo &FnInfo = 9140 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 9141 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 9142 SmallString<64> TyStr; 9143 llvm::raw_svector_ostream Out(TyStr); 9144 CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out); 9145 std::string Name = getName({"omp_mapper", TyStr, D->getName()}); 9146 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 9147 Name, &CGM.getModule()); 9148 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 9149 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 9150 // Start the mapper function code generation. 9151 CodeGenFunction MapperCGF(CGM); 9152 MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 9153 // Compute the starting and end addreses of array elements. 9154 llvm::Value *Size = MapperCGF.EmitLoadOfScalar( 9155 MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false, 9156 C.getPointerType(Int64Ty), Loc); 9157 llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast( 9158 MapperCGF.GetAddrOfLocalVar(&BeginArg).getPointer(), 9159 CGM.getTypes().ConvertTypeForMem(C.getPointerType(PtrTy))); 9160 llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(PtrBegin, Size); 9161 llvm::Value *MapType = MapperCGF.EmitLoadOfScalar( 9162 MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false, 9163 C.getPointerType(Int64Ty), Loc); 9164 // Prepare common arguments for array initiation and deletion. 9165 llvm::Value *Handle = MapperCGF.EmitLoadOfScalar( 9166 MapperCGF.GetAddrOfLocalVar(&HandleArg), 9167 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9168 llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar( 9169 MapperCGF.GetAddrOfLocalVar(&BaseArg), 9170 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9171 llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar( 9172 MapperCGF.GetAddrOfLocalVar(&BeginArg), 9173 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9174 9175 // Emit array initiation if this is an array section and \p MapType indicates 9176 // that memory allocation is required. 9177 llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head"); 9178 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 9179 ElementSize, HeadBB, /*IsInit=*/true); 9180 9181 // Emit a for loop to iterate through SizeArg of elements and map all of them. 9182 9183 // Emit the loop header block. 9184 MapperCGF.EmitBlock(HeadBB); 9185 llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body"); 9186 llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done"); 9187 // Evaluate whether the initial condition is satisfied. 9188 llvm::Value *IsEmpty = 9189 MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty"); 9190 MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 9191 llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock(); 9192 9193 // Emit the loop body block. 9194 MapperCGF.EmitBlock(BodyBB); 9195 llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI( 9196 PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent"); 9197 PtrPHI->addIncoming(PtrBegin, EntryBB); 9198 Address PtrCurrent = 9199 Address(PtrPHI, MapperCGF.GetAddrOfLocalVar(&BeginArg) 9200 .getAlignment() 9201 .alignmentOfArrayElement(ElementSize)); 9202 // Privatize the declared variable of mapper to be the current array element. 9203 CodeGenFunction::OMPPrivateScope Scope(MapperCGF); 9204 Scope.addPrivate(MapperVarDecl, [&MapperCGF, PtrCurrent, PtrTy]() { 9205 return MapperCGF 9206 .EmitLoadOfPointerLValue(PtrCurrent, PtrTy->castAs<PointerType>()) 9207 .getAddress(MapperCGF); 9208 }); 9209 (void)Scope.Privatize(); 9210 9211 // Get map clause information. Fill up the arrays with all mapped variables. 9212 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 9213 MappableExprsHandler::MapValuesArrayTy Pointers; 9214 MappableExprsHandler::MapValuesArrayTy Sizes; 9215 MappableExprsHandler::MapFlagsArrayTy MapTypes; 9216 MappableExprsHandler MEHandler(*D, MapperCGF); 9217 MEHandler.generateAllInfoForMapper(BasePointers, Pointers, Sizes, MapTypes); 9218 9219 // Call the runtime API __tgt_mapper_num_components to get the number of 9220 // pre-existing components. 9221 llvm::Value *OffloadingArgs[] = {Handle}; 9222 llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall( 9223 createRuntimeFunction(OMPRTL__tgt_mapper_num_components), OffloadingArgs); 9224 llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl( 9225 PreviousSize, 9226 MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset())); 9227 9228 // Fill up the runtime mapper handle for all components. 9229 for (unsigned I = 0; I < BasePointers.size(); ++I) { 9230 llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast( 9231 *BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 9232 llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast( 9233 Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 9234 llvm::Value *CurSizeArg = Sizes[I]; 9235 9236 // Extract the MEMBER_OF field from the map type. 9237 llvm::BasicBlock *MemberBB = MapperCGF.createBasicBlock("omp.member"); 9238 MapperCGF.EmitBlock(MemberBB); 9239 llvm::Value *OriMapType = MapperCGF.Builder.getInt64(MapTypes[I]); 9240 llvm::Value *Member = MapperCGF.Builder.CreateAnd( 9241 OriMapType, 9242 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_MEMBER_OF)); 9243 llvm::BasicBlock *MemberCombineBB = 9244 MapperCGF.createBasicBlock("omp.member.combine"); 9245 llvm::BasicBlock *TypeBB = MapperCGF.createBasicBlock("omp.type"); 9246 llvm::Value *IsMember = MapperCGF.Builder.CreateIsNull(Member); 9247 MapperCGF.Builder.CreateCondBr(IsMember, TypeBB, MemberCombineBB); 9248 // Add the number of pre-existing components to the MEMBER_OF field if it 9249 // is valid. 9250 MapperCGF.EmitBlock(MemberCombineBB); 9251 llvm::Value *CombinedMember = 9252 MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize); 9253 // Do nothing if it is not a member of previous components. 9254 MapperCGF.EmitBlock(TypeBB); 9255 llvm::PHINode *MemberMapType = 9256 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.membermaptype"); 9257 MemberMapType->addIncoming(OriMapType, MemberBB); 9258 MemberMapType->addIncoming(CombinedMember, MemberCombineBB); 9259 9260 // Combine the map type inherited from user-defined mapper with that 9261 // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM 9262 // bits of the \a MapType, which is the input argument of the mapper 9263 // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM 9264 // bits of MemberMapType. 9265 // [OpenMP 5.0], 1.2.6. map-type decay. 9266 // | alloc | to | from | tofrom | release | delete 9267 // ---------------------------------------------------------- 9268 // alloc | alloc | alloc | alloc | alloc | release | delete 9269 // to | alloc | to | alloc | to | release | delete 9270 // from | alloc | alloc | from | from | release | delete 9271 // tofrom | alloc | to | from | tofrom | release | delete 9272 llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd( 9273 MapType, 9274 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO | 9275 MappableExprsHandler::OMP_MAP_FROM)); 9276 llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc"); 9277 llvm::BasicBlock *AllocElseBB = 9278 MapperCGF.createBasicBlock("omp.type.alloc.else"); 9279 llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to"); 9280 llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else"); 9281 llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from"); 9282 llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end"); 9283 llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom); 9284 MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB); 9285 // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM. 9286 MapperCGF.EmitBlock(AllocBB); 9287 llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd( 9288 MemberMapType, 9289 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 9290 MappableExprsHandler::OMP_MAP_FROM))); 9291 MapperCGF.Builder.CreateBr(EndBB); 9292 MapperCGF.EmitBlock(AllocElseBB); 9293 llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ( 9294 LeftToFrom, 9295 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO)); 9296 MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB); 9297 // In case of to, clear OMP_MAP_FROM. 9298 MapperCGF.EmitBlock(ToBB); 9299 llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd( 9300 MemberMapType, 9301 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM)); 9302 MapperCGF.Builder.CreateBr(EndBB); 9303 MapperCGF.EmitBlock(ToElseBB); 9304 llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ( 9305 LeftToFrom, 9306 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM)); 9307 MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB); 9308 // In case of from, clear OMP_MAP_TO. 9309 MapperCGF.EmitBlock(FromBB); 9310 llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd( 9311 MemberMapType, 9312 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO)); 9313 // In case of tofrom, do nothing. 9314 MapperCGF.EmitBlock(EndBB); 9315 llvm::PHINode *CurMapType = 9316 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype"); 9317 CurMapType->addIncoming(AllocMapType, AllocBB); 9318 CurMapType->addIncoming(ToMapType, ToBB); 9319 CurMapType->addIncoming(FromMapType, FromBB); 9320 CurMapType->addIncoming(MemberMapType, ToElseBB); 9321 9322 // TODO: call the corresponding mapper function if a user-defined mapper is 9323 // associated with this map clause. 9324 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 9325 // data structure. 9326 llvm::Value *OffloadingArgs[] = {Handle, CurBaseArg, CurBeginArg, 9327 CurSizeArg, CurMapType}; 9328 MapperCGF.EmitRuntimeCall( 9329 createRuntimeFunction(OMPRTL__tgt_push_mapper_component), 9330 OffloadingArgs); 9331 } 9332 9333 // Update the pointer to point to the next element that needs to be mapped, 9334 // and check whether we have mapped all elements. 9335 llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32( 9336 PtrPHI, /*Idx0=*/1, "omp.arraymap.next"); 9337 PtrPHI->addIncoming(PtrNext, BodyBB); 9338 llvm::Value *IsDone = 9339 MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone"); 9340 llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit"); 9341 MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB); 9342 9343 MapperCGF.EmitBlock(ExitBB); 9344 // Emit array deletion if this is an array section and \p MapType indicates 9345 // that deletion is required. 9346 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 9347 ElementSize, DoneBB, /*IsInit=*/false); 9348 9349 // Emit the function exit block. 9350 MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true); 9351 MapperCGF.FinishFunction(); 9352 UDMMap.try_emplace(D, Fn); 9353 if (CGF) { 9354 auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn); 9355 Decls.second.push_back(D); 9356 } 9357 } 9358 9359 /// Emit the array initialization or deletion portion for user-defined mapper 9360 /// code generation. First, it evaluates whether an array section is mapped and 9361 /// whether the \a MapType instructs to delete this section. If \a IsInit is 9362 /// true, and \a MapType indicates to not delete this array, array 9363 /// initialization code is generated. If \a IsInit is false, and \a MapType 9364 /// indicates to not this array, array deletion code is generated. 9365 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel( 9366 CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base, 9367 llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType, 9368 CharUnits ElementSize, llvm::BasicBlock *ExitBB, bool IsInit) { 9369 StringRef Prefix = IsInit ? ".init" : ".del"; 9370 9371 // Evaluate if this is an array section. 9372 llvm::BasicBlock *IsDeleteBB = 9373 MapperCGF.createBasicBlock(getName({"omp.array", Prefix, ".evaldelete"})); 9374 llvm::BasicBlock *BodyBB = 9375 MapperCGF.createBasicBlock(getName({"omp.array", Prefix})); 9376 llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGE( 9377 Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray"); 9378 MapperCGF.Builder.CreateCondBr(IsArray, IsDeleteBB, ExitBB); 9379 9380 // Evaluate if we are going to delete this section. 9381 MapperCGF.EmitBlock(IsDeleteBB); 9382 llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd( 9383 MapType, 9384 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE)); 9385 llvm::Value *DeleteCond; 9386 if (IsInit) { 9387 DeleteCond = MapperCGF.Builder.CreateIsNull( 9388 DeleteBit, getName({"omp.array", Prefix, ".delete"})); 9389 } else { 9390 DeleteCond = MapperCGF.Builder.CreateIsNotNull( 9391 DeleteBit, getName({"omp.array", Prefix, ".delete"})); 9392 } 9393 MapperCGF.Builder.CreateCondBr(DeleteCond, BodyBB, ExitBB); 9394 9395 MapperCGF.EmitBlock(BodyBB); 9396 // Get the array size by multiplying element size and element number (i.e., \p 9397 // Size). 9398 llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul( 9399 Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity())); 9400 // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves 9401 // memory allocation/deletion purpose only. 9402 llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd( 9403 MapType, 9404 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 9405 MappableExprsHandler::OMP_MAP_FROM))); 9406 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 9407 // data structure. 9408 llvm::Value *OffloadingArgs[] = {Handle, Base, Begin, ArraySize, MapTypeArg}; 9409 MapperCGF.EmitRuntimeCall( 9410 createRuntimeFunction(OMPRTL__tgt_push_mapper_component), OffloadingArgs); 9411 } 9412 9413 void CGOpenMPRuntime::emitTargetNumIterationsCall( 9414 CodeGenFunction &CGF, const OMPExecutableDirective &D, 9415 llvm::Value *DeviceID, 9416 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 9417 const OMPLoopDirective &D)> 9418 SizeEmitter) { 9419 OpenMPDirectiveKind Kind = D.getDirectiveKind(); 9420 const OMPExecutableDirective *TD = &D; 9421 // Get nested teams distribute kind directive, if any. 9422 if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind)) 9423 TD = getNestedDistributeDirective(CGM.getContext(), D); 9424 if (!TD) 9425 return; 9426 const auto *LD = cast<OMPLoopDirective>(TD); 9427 auto &&CodeGen = [LD, DeviceID, SizeEmitter, this](CodeGenFunction &CGF, 9428 PrePostActionTy &) { 9429 if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) { 9430 llvm::Value *Args[] = {DeviceID, NumIterations}; 9431 CGF.EmitRuntimeCall( 9432 createRuntimeFunction(OMPRTL__kmpc_push_target_tripcount), Args); 9433 } 9434 }; 9435 emitInlinedDirective(CGF, OMPD_unknown, CodeGen); 9436 } 9437 9438 void CGOpenMPRuntime::emitTargetCall( 9439 CodeGenFunction &CGF, const OMPExecutableDirective &D, 9440 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 9441 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device, 9442 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 9443 const OMPLoopDirective &D)> 9444 SizeEmitter) { 9445 if (!CGF.HaveInsertPoint()) 9446 return; 9447 9448 assert(OutlinedFn && "Invalid outlined function!"); 9449 9450 const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>(); 9451 llvm::SmallVector<llvm::Value *, 16> CapturedVars; 9452 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 9453 auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF, 9454 PrePostActionTy &) { 9455 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9456 }; 9457 emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen); 9458 9459 CodeGenFunction::OMPTargetDataInfo InputInfo; 9460 llvm::Value *MapTypesArray = nullptr; 9461 // Fill up the pointer arrays and transfer execution to the device. 9462 auto &&ThenGen = [this, Device, OutlinedFn, OutlinedFnID, &D, &InputInfo, 9463 &MapTypesArray, &CS, RequiresOuterTask, &CapturedVars, 9464 SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) { 9465 if (Device.getInt() == OMPC_DEVICE_ancestor) { 9466 // Reverse offloading is not supported, so just execute on the host. 9467 if (RequiresOuterTask) { 9468 CapturedVars.clear(); 9469 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9470 } 9471 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9472 return; 9473 } 9474 9475 // On top of the arrays that were filled up, the target offloading call 9476 // takes as arguments the device id as well as the host pointer. The host 9477 // pointer is used by the runtime library to identify the current target 9478 // region, so it only has to be unique and not necessarily point to 9479 // anything. It could be the pointer to the outlined function that 9480 // implements the target region, but we aren't using that so that the 9481 // compiler doesn't need to keep that, and could therefore inline the host 9482 // function if proven worthwhile during optimization. 9483 9484 // From this point on, we need to have an ID of the target region defined. 9485 assert(OutlinedFnID && "Invalid outlined function ID!"); 9486 9487 // Emit device ID if any. 9488 llvm::Value *DeviceID; 9489 if (Device.getPointer()) { 9490 assert((Device.getInt() == OMPC_DEVICE_unknown || 9491 Device.getInt() == OMPC_DEVICE_device_num) && 9492 "Expected device_num modifier."); 9493 llvm::Value *DevVal = CGF.EmitScalarExpr(Device.getPointer()); 9494 DeviceID = 9495 CGF.Builder.CreateIntCast(DevVal, CGF.Int64Ty, /*isSigned=*/true); 9496 } else { 9497 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 9498 } 9499 9500 // Emit the number of elements in the offloading arrays. 9501 llvm::Value *PointerNum = 9502 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 9503 9504 // Return value of the runtime offloading call. 9505 llvm::Value *Return; 9506 9507 llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D); 9508 llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D); 9509 9510 // Emit tripcount for the target loop-based directive. 9511 emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter); 9512 9513 bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 9514 // The target region is an outlined function launched by the runtime 9515 // via calls __tgt_target() or __tgt_target_teams(). 9516 // 9517 // __tgt_target() launches a target region with one team and one thread, 9518 // executing a serial region. This master thread may in turn launch 9519 // more threads within its team upon encountering a parallel region, 9520 // however, no additional teams can be launched on the device. 9521 // 9522 // __tgt_target_teams() launches a target region with one or more teams, 9523 // each with one or more threads. This call is required for target 9524 // constructs such as: 9525 // 'target teams' 9526 // 'target' / 'teams' 9527 // 'target teams distribute parallel for' 9528 // 'target parallel' 9529 // and so on. 9530 // 9531 // Note that on the host and CPU targets, the runtime implementation of 9532 // these calls simply call the outlined function without forking threads. 9533 // The outlined functions themselves have runtime calls to 9534 // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by 9535 // the compiler in emitTeamsCall() and emitParallelCall(). 9536 // 9537 // In contrast, on the NVPTX target, the implementation of 9538 // __tgt_target_teams() launches a GPU kernel with the requested number 9539 // of teams and threads so no additional calls to the runtime are required. 9540 if (NumTeams) { 9541 // If we have NumTeams defined this means that we have an enclosed teams 9542 // region. Therefore we also expect to have NumThreads defined. These two 9543 // values should be defined in the presence of a teams directive, 9544 // regardless of having any clauses associated. If the user is using teams 9545 // but no clauses, these two values will be the default that should be 9546 // passed to the runtime library - a 32-bit integer with the value zero. 9547 assert(NumThreads && "Thread limit expression should be available along " 9548 "with number of teams."); 9549 llvm::Value *OffloadingArgs[] = {DeviceID, 9550 OutlinedFnID, 9551 PointerNum, 9552 InputInfo.BasePointersArray.getPointer(), 9553 InputInfo.PointersArray.getPointer(), 9554 InputInfo.SizesArray.getPointer(), 9555 MapTypesArray, 9556 NumTeams, 9557 NumThreads}; 9558 Return = CGF.EmitRuntimeCall( 9559 createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_teams_nowait 9560 : OMPRTL__tgt_target_teams), 9561 OffloadingArgs); 9562 } else { 9563 llvm::Value *OffloadingArgs[] = {DeviceID, 9564 OutlinedFnID, 9565 PointerNum, 9566 InputInfo.BasePointersArray.getPointer(), 9567 InputInfo.PointersArray.getPointer(), 9568 InputInfo.SizesArray.getPointer(), 9569 MapTypesArray}; 9570 Return = CGF.EmitRuntimeCall( 9571 createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_nowait 9572 : OMPRTL__tgt_target), 9573 OffloadingArgs); 9574 } 9575 9576 // Check the error code and execute the host version if required. 9577 llvm::BasicBlock *OffloadFailedBlock = 9578 CGF.createBasicBlock("omp_offload.failed"); 9579 llvm::BasicBlock *OffloadContBlock = 9580 CGF.createBasicBlock("omp_offload.cont"); 9581 llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return); 9582 CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock); 9583 9584 CGF.EmitBlock(OffloadFailedBlock); 9585 if (RequiresOuterTask) { 9586 CapturedVars.clear(); 9587 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9588 } 9589 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9590 CGF.EmitBranch(OffloadContBlock); 9591 9592 CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true); 9593 }; 9594 9595 // Notify that the host version must be executed. 9596 auto &&ElseGen = [this, &D, OutlinedFn, &CS, &CapturedVars, 9597 RequiresOuterTask](CodeGenFunction &CGF, 9598 PrePostActionTy &) { 9599 if (RequiresOuterTask) { 9600 CapturedVars.clear(); 9601 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9602 } 9603 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9604 }; 9605 9606 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray, 9607 &CapturedVars, RequiresOuterTask, 9608 &CS](CodeGenFunction &CGF, PrePostActionTy &) { 9609 // Fill up the arrays with all the captured variables. 9610 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 9611 MappableExprsHandler::MapValuesArrayTy Pointers; 9612 MappableExprsHandler::MapValuesArrayTy Sizes; 9613 MappableExprsHandler::MapFlagsArrayTy MapTypes; 9614 9615 // Get mappable expression information. 9616 MappableExprsHandler MEHandler(D, CGF); 9617 llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers; 9618 9619 auto RI = CS.getCapturedRecordDecl()->field_begin(); 9620 auto CV = CapturedVars.begin(); 9621 for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(), 9622 CE = CS.capture_end(); 9623 CI != CE; ++CI, ++RI, ++CV) { 9624 MappableExprsHandler::MapBaseValuesArrayTy CurBasePointers; 9625 MappableExprsHandler::MapValuesArrayTy CurPointers; 9626 MappableExprsHandler::MapValuesArrayTy CurSizes; 9627 MappableExprsHandler::MapFlagsArrayTy CurMapTypes; 9628 MappableExprsHandler::StructRangeInfoTy PartialStruct; 9629 9630 // VLA sizes are passed to the outlined region by copy and do not have map 9631 // information associated. 9632 if (CI->capturesVariableArrayType()) { 9633 CurBasePointers.push_back(*CV); 9634 CurPointers.push_back(*CV); 9635 CurSizes.push_back(CGF.Builder.CreateIntCast( 9636 CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true)); 9637 // Copy to the device as an argument. No need to retrieve it. 9638 CurMapTypes.push_back(MappableExprsHandler::OMP_MAP_LITERAL | 9639 MappableExprsHandler::OMP_MAP_TARGET_PARAM | 9640 MappableExprsHandler::OMP_MAP_IMPLICIT); 9641 } else { 9642 // If we have any information in the map clause, we use it, otherwise we 9643 // just do a default mapping. 9644 MEHandler.generateInfoForCapture(CI, *CV, CurBasePointers, CurPointers, 9645 CurSizes, CurMapTypes, PartialStruct); 9646 if (CurBasePointers.empty()) 9647 MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurBasePointers, 9648 CurPointers, CurSizes, CurMapTypes); 9649 // Generate correct mapping for variables captured by reference in 9650 // lambdas. 9651 if (CI->capturesVariable()) 9652 MEHandler.generateInfoForLambdaCaptures( 9653 CI->getCapturedVar(), *CV, CurBasePointers, CurPointers, CurSizes, 9654 CurMapTypes, LambdaPointers); 9655 } 9656 // We expect to have at least an element of information for this capture. 9657 assert(!CurBasePointers.empty() && 9658 "Non-existing map pointer for capture!"); 9659 assert(CurBasePointers.size() == CurPointers.size() && 9660 CurBasePointers.size() == CurSizes.size() && 9661 CurBasePointers.size() == CurMapTypes.size() && 9662 "Inconsistent map information sizes!"); 9663 9664 // If there is an entry in PartialStruct it means we have a struct with 9665 // individual members mapped. Emit an extra combined entry. 9666 if (PartialStruct.Base.isValid()) 9667 MEHandler.emitCombinedEntry(BasePointers, Pointers, Sizes, MapTypes, 9668 CurMapTypes, PartialStruct); 9669 9670 // We need to append the results of this capture to what we already have. 9671 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 9672 Pointers.append(CurPointers.begin(), CurPointers.end()); 9673 Sizes.append(CurSizes.begin(), CurSizes.end()); 9674 MapTypes.append(CurMapTypes.begin(), CurMapTypes.end()); 9675 } 9676 // Adjust MEMBER_OF flags for the lambdas captures. 9677 MEHandler.adjustMemberOfForLambdaCaptures(LambdaPointers, BasePointers, 9678 Pointers, MapTypes); 9679 // Map other list items in the map clause which are not captured variables 9680 // but "declare target link" global variables. 9681 MEHandler.generateInfoForDeclareTargetLink(BasePointers, Pointers, Sizes, 9682 MapTypes); 9683 9684 TargetDataInfo Info; 9685 // Fill up the arrays and create the arguments. 9686 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 9687 emitOffloadingArraysArgument(CGF, Info.BasePointersArray, 9688 Info.PointersArray, Info.SizesArray, 9689 Info.MapTypesArray, Info); 9690 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 9691 InputInfo.BasePointersArray = 9692 Address(Info.BasePointersArray, CGM.getPointerAlign()); 9693 InputInfo.PointersArray = 9694 Address(Info.PointersArray, CGM.getPointerAlign()); 9695 InputInfo.SizesArray = Address(Info.SizesArray, CGM.getPointerAlign()); 9696 MapTypesArray = Info.MapTypesArray; 9697 if (RequiresOuterTask) 9698 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 9699 else 9700 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 9701 }; 9702 9703 auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask]( 9704 CodeGenFunction &CGF, PrePostActionTy &) { 9705 if (RequiresOuterTask) { 9706 CodeGenFunction::OMPTargetDataInfo InputInfo; 9707 CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo); 9708 } else { 9709 emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen); 9710 } 9711 }; 9712 9713 // If we have a target function ID it means that we need to support 9714 // offloading, otherwise, just execute on the host. We need to execute on host 9715 // regardless of the conditional in the if clause if, e.g., the user do not 9716 // specify target triples. 9717 if (OutlinedFnID) { 9718 if (IfCond) { 9719 emitIfClause(CGF, IfCond, TargetThenGen, TargetElseGen); 9720 } else { 9721 RegionCodeGenTy ThenRCG(TargetThenGen); 9722 ThenRCG(CGF); 9723 } 9724 } else { 9725 RegionCodeGenTy ElseRCG(TargetElseGen); 9726 ElseRCG(CGF); 9727 } 9728 } 9729 9730 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S, 9731 StringRef ParentName) { 9732 if (!S) 9733 return; 9734 9735 // Codegen OMP target directives that offload compute to the device. 9736 bool RequiresDeviceCodegen = 9737 isa<OMPExecutableDirective>(S) && 9738 isOpenMPTargetExecutionDirective( 9739 cast<OMPExecutableDirective>(S)->getDirectiveKind()); 9740 9741 if (RequiresDeviceCodegen) { 9742 const auto &E = *cast<OMPExecutableDirective>(S); 9743 unsigned DeviceID; 9744 unsigned FileID; 9745 unsigned Line; 9746 getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID, 9747 FileID, Line); 9748 9749 // Is this a target region that should not be emitted as an entry point? If 9750 // so just signal we are done with this target region. 9751 if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID, 9752 ParentName, Line)) 9753 return; 9754 9755 switch (E.getDirectiveKind()) { 9756 case OMPD_target: 9757 CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName, 9758 cast<OMPTargetDirective>(E)); 9759 break; 9760 case OMPD_target_parallel: 9761 CodeGenFunction::EmitOMPTargetParallelDeviceFunction( 9762 CGM, ParentName, cast<OMPTargetParallelDirective>(E)); 9763 break; 9764 case OMPD_target_teams: 9765 CodeGenFunction::EmitOMPTargetTeamsDeviceFunction( 9766 CGM, ParentName, cast<OMPTargetTeamsDirective>(E)); 9767 break; 9768 case OMPD_target_teams_distribute: 9769 CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction( 9770 CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E)); 9771 break; 9772 case OMPD_target_teams_distribute_simd: 9773 CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction( 9774 CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E)); 9775 break; 9776 case OMPD_target_parallel_for: 9777 CodeGenFunction::EmitOMPTargetParallelForDeviceFunction( 9778 CGM, ParentName, cast<OMPTargetParallelForDirective>(E)); 9779 break; 9780 case OMPD_target_parallel_for_simd: 9781 CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction( 9782 CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E)); 9783 break; 9784 case OMPD_target_simd: 9785 CodeGenFunction::EmitOMPTargetSimdDeviceFunction( 9786 CGM, ParentName, cast<OMPTargetSimdDirective>(E)); 9787 break; 9788 case OMPD_target_teams_distribute_parallel_for: 9789 CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction( 9790 CGM, ParentName, 9791 cast<OMPTargetTeamsDistributeParallelForDirective>(E)); 9792 break; 9793 case OMPD_target_teams_distribute_parallel_for_simd: 9794 CodeGenFunction:: 9795 EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction( 9796 CGM, ParentName, 9797 cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E)); 9798 break; 9799 case OMPD_parallel: 9800 case OMPD_for: 9801 case OMPD_parallel_for: 9802 case OMPD_parallel_master: 9803 case OMPD_parallel_sections: 9804 case OMPD_for_simd: 9805 case OMPD_parallel_for_simd: 9806 case OMPD_cancel: 9807 case OMPD_cancellation_point: 9808 case OMPD_ordered: 9809 case OMPD_threadprivate: 9810 case OMPD_allocate: 9811 case OMPD_task: 9812 case OMPD_simd: 9813 case OMPD_sections: 9814 case OMPD_section: 9815 case OMPD_single: 9816 case OMPD_master: 9817 case OMPD_critical: 9818 case OMPD_taskyield: 9819 case OMPD_barrier: 9820 case OMPD_taskwait: 9821 case OMPD_taskgroup: 9822 case OMPD_atomic: 9823 case OMPD_flush: 9824 case OMPD_depobj: 9825 case OMPD_scan: 9826 case OMPD_teams: 9827 case OMPD_target_data: 9828 case OMPD_target_exit_data: 9829 case OMPD_target_enter_data: 9830 case OMPD_distribute: 9831 case OMPD_distribute_simd: 9832 case OMPD_distribute_parallel_for: 9833 case OMPD_distribute_parallel_for_simd: 9834 case OMPD_teams_distribute: 9835 case OMPD_teams_distribute_simd: 9836 case OMPD_teams_distribute_parallel_for: 9837 case OMPD_teams_distribute_parallel_for_simd: 9838 case OMPD_target_update: 9839 case OMPD_declare_simd: 9840 case OMPD_declare_variant: 9841 case OMPD_begin_declare_variant: 9842 case OMPD_end_declare_variant: 9843 case OMPD_declare_target: 9844 case OMPD_end_declare_target: 9845 case OMPD_declare_reduction: 9846 case OMPD_declare_mapper: 9847 case OMPD_taskloop: 9848 case OMPD_taskloop_simd: 9849 case OMPD_master_taskloop: 9850 case OMPD_master_taskloop_simd: 9851 case OMPD_parallel_master_taskloop: 9852 case OMPD_parallel_master_taskloop_simd: 9853 case OMPD_requires: 9854 case OMPD_unknown: 9855 llvm_unreachable("Unknown target directive for OpenMP device codegen."); 9856 } 9857 return; 9858 } 9859 9860 if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) { 9861 if (!E->hasAssociatedStmt() || !E->getAssociatedStmt()) 9862 return; 9863 9864 scanForTargetRegionsFunctions( 9865 E->getInnermostCapturedStmt()->getCapturedStmt(), ParentName); 9866 return; 9867 } 9868 9869 // If this is a lambda function, look into its body. 9870 if (const auto *L = dyn_cast<LambdaExpr>(S)) 9871 S = L->getBody(); 9872 9873 // Keep looking for target regions recursively. 9874 for (const Stmt *II : S->children()) 9875 scanForTargetRegionsFunctions(II, ParentName); 9876 } 9877 9878 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) { 9879 // If emitting code for the host, we do not process FD here. Instead we do 9880 // the normal code generation. 9881 if (!CGM.getLangOpts().OpenMPIsDevice) { 9882 if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl())) { 9883 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 9884 OMPDeclareTargetDeclAttr::getDeviceType(FD); 9885 // Do not emit device_type(nohost) functions for the host. 9886 if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_NoHost) 9887 return true; 9888 } 9889 return false; 9890 } 9891 9892 const ValueDecl *VD = cast<ValueDecl>(GD.getDecl()); 9893 // Try to detect target regions in the function. 9894 if (const auto *FD = dyn_cast<FunctionDecl>(VD)) { 9895 StringRef Name = CGM.getMangledName(GD); 9896 scanForTargetRegionsFunctions(FD->getBody(), Name); 9897 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 9898 OMPDeclareTargetDeclAttr::getDeviceType(FD); 9899 // Do not emit device_type(nohost) functions for the host. 9900 if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_Host) 9901 return true; 9902 } 9903 9904 // Do not to emit function if it is not marked as declare target. 9905 return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) && 9906 AlreadyEmittedTargetDecls.count(VD) == 0; 9907 } 9908 9909 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 9910 if (!CGM.getLangOpts().OpenMPIsDevice) 9911 return false; 9912 9913 // Check if there are Ctors/Dtors in this declaration and look for target 9914 // regions in it. We use the complete variant to produce the kernel name 9915 // mangling. 9916 QualType RDTy = cast<VarDecl>(GD.getDecl())->getType(); 9917 if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) { 9918 for (const CXXConstructorDecl *Ctor : RD->ctors()) { 9919 StringRef ParentName = 9920 CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete)); 9921 scanForTargetRegionsFunctions(Ctor->getBody(), ParentName); 9922 } 9923 if (const CXXDestructorDecl *Dtor = RD->getDestructor()) { 9924 StringRef ParentName = 9925 CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete)); 9926 scanForTargetRegionsFunctions(Dtor->getBody(), ParentName); 9927 } 9928 } 9929 9930 // Do not to emit variable if it is not marked as declare target. 9931 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 9932 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration( 9933 cast<VarDecl>(GD.getDecl())); 9934 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 9935 (*Res == OMPDeclareTargetDeclAttr::MT_To && 9936 HasRequiresUnifiedSharedMemory)) { 9937 DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl())); 9938 return true; 9939 } 9940 return false; 9941 } 9942 9943 llvm::Constant * 9944 CGOpenMPRuntime::registerTargetFirstprivateCopy(CodeGenFunction &CGF, 9945 const VarDecl *VD) { 9946 assert(VD->getType().isConstant(CGM.getContext()) && 9947 "Expected constant variable."); 9948 StringRef VarName; 9949 llvm::Constant *Addr; 9950 llvm::GlobalValue::LinkageTypes Linkage; 9951 QualType Ty = VD->getType(); 9952 SmallString<128> Buffer; 9953 { 9954 unsigned DeviceID; 9955 unsigned FileID; 9956 unsigned Line; 9957 getTargetEntryUniqueInfo(CGM.getContext(), VD->getLocation(), DeviceID, 9958 FileID, Line); 9959 llvm::raw_svector_ostream OS(Buffer); 9960 OS << "__omp_offloading_firstprivate_" << llvm::format("_%x", DeviceID) 9961 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 9962 VarName = OS.str(); 9963 } 9964 Linkage = llvm::GlobalValue::InternalLinkage; 9965 Addr = 9966 getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(Ty), VarName, 9967 getDefaultFirstprivateAddressSpace()); 9968 cast<llvm::GlobalValue>(Addr)->setLinkage(Linkage); 9969 CharUnits VarSize = CGM.getContext().getTypeSizeInChars(Ty); 9970 CGM.addCompilerUsedGlobal(cast<llvm::GlobalValue>(Addr)); 9971 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 9972 VarName, Addr, VarSize, 9973 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo, Linkage); 9974 return Addr; 9975 } 9976 9977 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD, 9978 llvm::Constant *Addr) { 9979 if (CGM.getLangOpts().OMPTargetTriples.empty() && 9980 !CGM.getLangOpts().OpenMPIsDevice) 9981 return; 9982 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 9983 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 9984 if (!Res) { 9985 if (CGM.getLangOpts().OpenMPIsDevice) { 9986 // Register non-target variables being emitted in device code (debug info 9987 // may cause this). 9988 StringRef VarName = CGM.getMangledName(VD); 9989 EmittedNonTargetVariables.try_emplace(VarName, Addr); 9990 } 9991 return; 9992 } 9993 // Register declare target variables. 9994 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags; 9995 StringRef VarName; 9996 CharUnits VarSize; 9997 llvm::GlobalValue::LinkageTypes Linkage; 9998 9999 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 10000 !HasRequiresUnifiedSharedMemory) { 10001 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 10002 VarName = CGM.getMangledName(VD); 10003 if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) { 10004 VarSize = CGM.getContext().getTypeSizeInChars(VD->getType()); 10005 assert(!VarSize.isZero() && "Expected non-zero size of the variable"); 10006 } else { 10007 VarSize = CharUnits::Zero(); 10008 } 10009 Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false); 10010 // Temp solution to prevent optimizations of the internal variables. 10011 if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) { 10012 std::string RefName = getName({VarName, "ref"}); 10013 if (!CGM.GetGlobalValue(RefName)) { 10014 llvm::Constant *AddrRef = 10015 getOrCreateInternalVariable(Addr->getType(), RefName); 10016 auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef); 10017 GVAddrRef->setConstant(/*Val=*/true); 10018 GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage); 10019 GVAddrRef->setInitializer(Addr); 10020 CGM.addCompilerUsedGlobal(GVAddrRef); 10021 } 10022 } 10023 } else { 10024 assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 10025 (*Res == OMPDeclareTargetDeclAttr::MT_To && 10026 HasRequiresUnifiedSharedMemory)) && 10027 "Declare target attribute must link or to with unified memory."); 10028 if (*Res == OMPDeclareTargetDeclAttr::MT_Link) 10029 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink; 10030 else 10031 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 10032 10033 if (CGM.getLangOpts().OpenMPIsDevice) { 10034 VarName = Addr->getName(); 10035 Addr = nullptr; 10036 } else { 10037 VarName = getAddrOfDeclareTargetVar(VD).getName(); 10038 Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer()); 10039 } 10040 VarSize = CGM.getPointerSize(); 10041 Linkage = llvm::GlobalValue::WeakAnyLinkage; 10042 } 10043 10044 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 10045 VarName, Addr, VarSize, Flags, Linkage); 10046 } 10047 10048 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) { 10049 if (isa<FunctionDecl>(GD.getDecl()) || 10050 isa<OMPDeclareReductionDecl>(GD.getDecl())) 10051 return emitTargetFunctions(GD); 10052 10053 return emitTargetGlobalVariable(GD); 10054 } 10055 10056 void CGOpenMPRuntime::emitDeferredTargetDecls() const { 10057 for (const VarDecl *VD : DeferredGlobalVariables) { 10058 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 10059 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 10060 if (!Res) 10061 continue; 10062 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 10063 !HasRequiresUnifiedSharedMemory) { 10064 CGM.EmitGlobal(VD); 10065 } else { 10066 assert((*Res == OMPDeclareTargetDeclAttr::MT_Link || 10067 (*Res == OMPDeclareTargetDeclAttr::MT_To && 10068 HasRequiresUnifiedSharedMemory)) && 10069 "Expected link clause or to clause with unified memory."); 10070 (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 10071 } 10072 } 10073 } 10074 10075 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas( 10076 CodeGenFunction &CGF, const OMPExecutableDirective &D) const { 10077 assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) && 10078 " Expected target-based directive."); 10079 } 10080 10081 void CGOpenMPRuntime::processRequiresDirective(const OMPRequiresDecl *D) { 10082 for (const OMPClause *Clause : D->clauselists()) { 10083 if (Clause->getClauseKind() == OMPC_unified_shared_memory) { 10084 HasRequiresUnifiedSharedMemory = true; 10085 } else if (const auto *AC = 10086 dyn_cast<OMPAtomicDefaultMemOrderClause>(Clause)) { 10087 switch (AC->getAtomicDefaultMemOrderKind()) { 10088 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_acq_rel: 10089 RequiresAtomicOrdering = llvm::AtomicOrdering::AcquireRelease; 10090 break; 10091 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_seq_cst: 10092 RequiresAtomicOrdering = llvm::AtomicOrdering::SequentiallyConsistent; 10093 break; 10094 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_relaxed: 10095 RequiresAtomicOrdering = llvm::AtomicOrdering::Monotonic; 10096 break; 10097 case OMPC_ATOMIC_DEFAULT_MEM_ORDER_unknown: 10098 break; 10099 } 10100 } 10101 } 10102 } 10103 10104 llvm::AtomicOrdering CGOpenMPRuntime::getDefaultMemoryOrdering() const { 10105 return RequiresAtomicOrdering; 10106 } 10107 10108 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD, 10109 LangAS &AS) { 10110 if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>()) 10111 return false; 10112 const auto *A = VD->getAttr<OMPAllocateDeclAttr>(); 10113 switch(A->getAllocatorType()) { 10114 case OMPAllocateDeclAttr::OMPDefaultMemAlloc: 10115 // Not supported, fallback to the default mem space. 10116 case OMPAllocateDeclAttr::OMPLargeCapMemAlloc: 10117 case OMPAllocateDeclAttr::OMPCGroupMemAlloc: 10118 case OMPAllocateDeclAttr::OMPHighBWMemAlloc: 10119 case OMPAllocateDeclAttr::OMPLowLatMemAlloc: 10120 case OMPAllocateDeclAttr::OMPThreadMemAlloc: 10121 case OMPAllocateDeclAttr::OMPConstMemAlloc: 10122 case OMPAllocateDeclAttr::OMPPTeamMemAlloc: 10123 AS = LangAS::Default; 10124 return true; 10125 case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc: 10126 llvm_unreachable("Expected predefined allocator for the variables with the " 10127 "static storage."); 10128 } 10129 return false; 10130 } 10131 10132 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const { 10133 return HasRequiresUnifiedSharedMemory; 10134 } 10135 10136 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII( 10137 CodeGenModule &CGM) 10138 : CGM(CGM) { 10139 if (CGM.getLangOpts().OpenMPIsDevice) { 10140 SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal; 10141 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false; 10142 } 10143 } 10144 10145 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() { 10146 if (CGM.getLangOpts().OpenMPIsDevice) 10147 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal; 10148 } 10149 10150 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) { 10151 if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal) 10152 return true; 10153 10154 const auto *D = cast<FunctionDecl>(GD.getDecl()); 10155 // Do not to emit function if it is marked as declare target as it was already 10156 // emitted. 10157 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) { 10158 if (D->hasBody() && AlreadyEmittedTargetDecls.count(D) == 0) { 10159 if (auto *F = dyn_cast_or_null<llvm::Function>( 10160 CGM.GetGlobalValue(CGM.getMangledName(GD)))) 10161 return !F->isDeclaration(); 10162 return false; 10163 } 10164 return true; 10165 } 10166 10167 return !AlreadyEmittedTargetDecls.insert(D).second; 10168 } 10169 10170 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() { 10171 // If we don't have entries or if we are emitting code for the device, we 10172 // don't need to do anything. 10173 if (CGM.getLangOpts().OMPTargetTriples.empty() || 10174 CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice || 10175 (OffloadEntriesInfoManager.empty() && 10176 !HasEmittedDeclareTargetRegion && 10177 !HasEmittedTargetRegion)) 10178 return nullptr; 10179 10180 // Create and register the function that handles the requires directives. 10181 ASTContext &C = CGM.getContext(); 10182 10183 llvm::Function *RequiresRegFn; 10184 { 10185 CodeGenFunction CGF(CGM); 10186 const auto &FI = CGM.getTypes().arrangeNullaryFunction(); 10187 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 10188 std::string ReqName = getName({"omp_offloading", "requires_reg"}); 10189 RequiresRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, ReqName, FI); 10190 CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {}); 10191 OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE; 10192 // TODO: check for other requires clauses. 10193 // The requires directive takes effect only when a target region is 10194 // present in the compilation unit. Otherwise it is ignored and not 10195 // passed to the runtime. This avoids the runtime from throwing an error 10196 // for mismatching requires clauses across compilation units that don't 10197 // contain at least 1 target region. 10198 assert((HasEmittedTargetRegion || 10199 HasEmittedDeclareTargetRegion || 10200 !OffloadEntriesInfoManager.empty()) && 10201 "Target or declare target region expected."); 10202 if (HasRequiresUnifiedSharedMemory) 10203 Flags = OMP_REQ_UNIFIED_SHARED_MEMORY; 10204 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_requires), 10205 llvm::ConstantInt::get(CGM.Int64Ty, Flags)); 10206 CGF.FinishFunction(); 10207 } 10208 return RequiresRegFn; 10209 } 10210 10211 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF, 10212 const OMPExecutableDirective &D, 10213 SourceLocation Loc, 10214 llvm::Function *OutlinedFn, 10215 ArrayRef<llvm::Value *> CapturedVars) { 10216 if (!CGF.HaveInsertPoint()) 10217 return; 10218 10219 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 10220 CodeGenFunction::RunCleanupsScope Scope(CGF); 10221 10222 // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn); 10223 llvm::Value *Args[] = { 10224 RTLoc, 10225 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 10226 CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())}; 10227 llvm::SmallVector<llvm::Value *, 16> RealArgs; 10228 RealArgs.append(std::begin(Args), std::end(Args)); 10229 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 10230 10231 llvm::FunctionCallee RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_teams); 10232 CGF.EmitRuntimeCall(RTLFn, RealArgs); 10233 } 10234 10235 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 10236 const Expr *NumTeams, 10237 const Expr *ThreadLimit, 10238 SourceLocation Loc) { 10239 if (!CGF.HaveInsertPoint()) 10240 return; 10241 10242 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 10243 10244 llvm::Value *NumTeamsVal = 10245 NumTeams 10246 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams), 10247 CGF.CGM.Int32Ty, /* isSigned = */ true) 10248 : CGF.Builder.getInt32(0); 10249 10250 llvm::Value *ThreadLimitVal = 10251 ThreadLimit 10252 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit), 10253 CGF.CGM.Int32Ty, /* isSigned = */ true) 10254 : CGF.Builder.getInt32(0); 10255 10256 // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit) 10257 llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal, 10258 ThreadLimitVal}; 10259 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_teams), 10260 PushNumTeamsArgs); 10261 } 10262 10263 void CGOpenMPRuntime::emitTargetDataCalls( 10264 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 10265 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 10266 if (!CGF.HaveInsertPoint()) 10267 return; 10268 10269 // Action used to replace the default codegen action and turn privatization 10270 // off. 10271 PrePostActionTy NoPrivAction; 10272 10273 // Generate the code for the opening of the data environment. Capture all the 10274 // arguments of the runtime call by reference because they are used in the 10275 // closing of the region. 10276 auto &&BeginThenGen = [this, &D, Device, &Info, 10277 &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) { 10278 // Fill up the arrays with all the mapped variables. 10279 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 10280 MappableExprsHandler::MapValuesArrayTy Pointers; 10281 MappableExprsHandler::MapValuesArrayTy Sizes; 10282 MappableExprsHandler::MapFlagsArrayTy MapTypes; 10283 10284 // Get map clause information. 10285 MappableExprsHandler MCHandler(D, CGF); 10286 MCHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes); 10287 10288 // Fill up the arrays and create the arguments. 10289 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 10290 10291 llvm::Value *BasePointersArrayArg = nullptr; 10292 llvm::Value *PointersArrayArg = nullptr; 10293 llvm::Value *SizesArrayArg = nullptr; 10294 llvm::Value *MapTypesArrayArg = nullptr; 10295 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 10296 SizesArrayArg, MapTypesArrayArg, Info); 10297 10298 // Emit device ID if any. 10299 llvm::Value *DeviceID = nullptr; 10300 if (Device) { 10301 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10302 CGF.Int64Ty, /*isSigned=*/true); 10303 } else { 10304 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10305 } 10306 10307 // Emit the number of elements in the offloading arrays. 10308 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 10309 10310 llvm::Value *OffloadingArgs[] = { 10311 DeviceID, PointerNum, BasePointersArrayArg, 10312 PointersArrayArg, SizesArrayArg, MapTypesArrayArg}; 10313 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_begin), 10314 OffloadingArgs); 10315 10316 // If device pointer privatization is required, emit the body of the region 10317 // here. It will have to be duplicated: with and without privatization. 10318 if (!Info.CaptureDeviceAddrMap.empty()) 10319 CodeGen(CGF); 10320 }; 10321 10322 // Generate code for the closing of the data region. 10323 auto &&EndThenGen = [this, Device, &Info](CodeGenFunction &CGF, 10324 PrePostActionTy &) { 10325 assert(Info.isValid() && "Invalid data environment closing arguments."); 10326 10327 llvm::Value *BasePointersArrayArg = nullptr; 10328 llvm::Value *PointersArrayArg = nullptr; 10329 llvm::Value *SizesArrayArg = nullptr; 10330 llvm::Value *MapTypesArrayArg = nullptr; 10331 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 10332 SizesArrayArg, MapTypesArrayArg, Info); 10333 10334 // Emit device ID if any. 10335 llvm::Value *DeviceID = nullptr; 10336 if (Device) { 10337 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10338 CGF.Int64Ty, /*isSigned=*/true); 10339 } else { 10340 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10341 } 10342 10343 // Emit the number of elements in the offloading arrays. 10344 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 10345 10346 llvm::Value *OffloadingArgs[] = { 10347 DeviceID, PointerNum, BasePointersArrayArg, 10348 PointersArrayArg, SizesArrayArg, MapTypesArrayArg}; 10349 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_end), 10350 OffloadingArgs); 10351 }; 10352 10353 // If we need device pointer privatization, we need to emit the body of the 10354 // region with no privatization in the 'else' branch of the conditional. 10355 // Otherwise, we don't have to do anything. 10356 auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF, 10357 PrePostActionTy &) { 10358 if (!Info.CaptureDeviceAddrMap.empty()) { 10359 CodeGen.setAction(NoPrivAction); 10360 CodeGen(CGF); 10361 } 10362 }; 10363 10364 // We don't have to do anything to close the region if the if clause evaluates 10365 // to false. 10366 auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {}; 10367 10368 if (IfCond) { 10369 emitIfClause(CGF, IfCond, BeginThenGen, BeginElseGen); 10370 } else { 10371 RegionCodeGenTy RCG(BeginThenGen); 10372 RCG(CGF); 10373 } 10374 10375 // If we don't require privatization of device pointers, we emit the body in 10376 // between the runtime calls. This avoids duplicating the body code. 10377 if (Info.CaptureDeviceAddrMap.empty()) { 10378 CodeGen.setAction(NoPrivAction); 10379 CodeGen(CGF); 10380 } 10381 10382 if (IfCond) { 10383 emitIfClause(CGF, IfCond, EndThenGen, EndElseGen); 10384 } else { 10385 RegionCodeGenTy RCG(EndThenGen); 10386 RCG(CGF); 10387 } 10388 } 10389 10390 void CGOpenMPRuntime::emitTargetDataStandAloneCall( 10391 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 10392 const Expr *Device) { 10393 if (!CGF.HaveInsertPoint()) 10394 return; 10395 10396 assert((isa<OMPTargetEnterDataDirective>(D) || 10397 isa<OMPTargetExitDataDirective>(D) || 10398 isa<OMPTargetUpdateDirective>(D)) && 10399 "Expecting either target enter, exit data, or update directives."); 10400 10401 CodeGenFunction::OMPTargetDataInfo InputInfo; 10402 llvm::Value *MapTypesArray = nullptr; 10403 // Generate the code for the opening of the data environment. 10404 auto &&ThenGen = [this, &D, Device, &InputInfo, 10405 &MapTypesArray](CodeGenFunction &CGF, PrePostActionTy &) { 10406 // Emit device ID if any. 10407 llvm::Value *DeviceID = nullptr; 10408 if (Device) { 10409 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10410 CGF.Int64Ty, /*isSigned=*/true); 10411 } else { 10412 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10413 } 10414 10415 // Emit the number of elements in the offloading arrays. 10416 llvm::Constant *PointerNum = 10417 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 10418 10419 llvm::Value *OffloadingArgs[] = {DeviceID, 10420 PointerNum, 10421 InputInfo.BasePointersArray.getPointer(), 10422 InputInfo.PointersArray.getPointer(), 10423 InputInfo.SizesArray.getPointer(), 10424 MapTypesArray}; 10425 10426 // Select the right runtime function call for each expected standalone 10427 // directive. 10428 const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 10429 OpenMPRTLFunction RTLFn; 10430 switch (D.getDirectiveKind()) { 10431 case OMPD_target_enter_data: 10432 RTLFn = HasNowait ? OMPRTL__tgt_target_data_begin_nowait 10433 : OMPRTL__tgt_target_data_begin; 10434 break; 10435 case OMPD_target_exit_data: 10436 RTLFn = HasNowait ? OMPRTL__tgt_target_data_end_nowait 10437 : OMPRTL__tgt_target_data_end; 10438 break; 10439 case OMPD_target_update: 10440 RTLFn = HasNowait ? OMPRTL__tgt_target_data_update_nowait 10441 : OMPRTL__tgt_target_data_update; 10442 break; 10443 case OMPD_parallel: 10444 case OMPD_for: 10445 case OMPD_parallel_for: 10446 case OMPD_parallel_master: 10447 case OMPD_parallel_sections: 10448 case OMPD_for_simd: 10449 case OMPD_parallel_for_simd: 10450 case OMPD_cancel: 10451 case OMPD_cancellation_point: 10452 case OMPD_ordered: 10453 case OMPD_threadprivate: 10454 case OMPD_allocate: 10455 case OMPD_task: 10456 case OMPD_simd: 10457 case OMPD_sections: 10458 case OMPD_section: 10459 case OMPD_single: 10460 case OMPD_master: 10461 case OMPD_critical: 10462 case OMPD_taskyield: 10463 case OMPD_barrier: 10464 case OMPD_taskwait: 10465 case OMPD_taskgroup: 10466 case OMPD_atomic: 10467 case OMPD_flush: 10468 case OMPD_depobj: 10469 case OMPD_scan: 10470 case OMPD_teams: 10471 case OMPD_target_data: 10472 case OMPD_distribute: 10473 case OMPD_distribute_simd: 10474 case OMPD_distribute_parallel_for: 10475 case OMPD_distribute_parallel_for_simd: 10476 case OMPD_teams_distribute: 10477 case OMPD_teams_distribute_simd: 10478 case OMPD_teams_distribute_parallel_for: 10479 case OMPD_teams_distribute_parallel_for_simd: 10480 case OMPD_declare_simd: 10481 case OMPD_declare_variant: 10482 case OMPD_begin_declare_variant: 10483 case OMPD_end_declare_variant: 10484 case OMPD_declare_target: 10485 case OMPD_end_declare_target: 10486 case OMPD_declare_reduction: 10487 case OMPD_declare_mapper: 10488 case OMPD_taskloop: 10489 case OMPD_taskloop_simd: 10490 case OMPD_master_taskloop: 10491 case OMPD_master_taskloop_simd: 10492 case OMPD_parallel_master_taskloop: 10493 case OMPD_parallel_master_taskloop_simd: 10494 case OMPD_target: 10495 case OMPD_target_simd: 10496 case OMPD_target_teams_distribute: 10497 case OMPD_target_teams_distribute_simd: 10498 case OMPD_target_teams_distribute_parallel_for: 10499 case OMPD_target_teams_distribute_parallel_for_simd: 10500 case OMPD_target_teams: 10501 case OMPD_target_parallel: 10502 case OMPD_target_parallel_for: 10503 case OMPD_target_parallel_for_simd: 10504 case OMPD_requires: 10505 case OMPD_unknown: 10506 llvm_unreachable("Unexpected standalone target data directive."); 10507 break; 10508 } 10509 CGF.EmitRuntimeCall(createRuntimeFunction(RTLFn), OffloadingArgs); 10510 }; 10511 10512 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray]( 10513 CodeGenFunction &CGF, PrePostActionTy &) { 10514 // Fill up the arrays with all the mapped variables. 10515 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 10516 MappableExprsHandler::MapValuesArrayTy Pointers; 10517 MappableExprsHandler::MapValuesArrayTy Sizes; 10518 MappableExprsHandler::MapFlagsArrayTy MapTypes; 10519 10520 // Get map clause information. 10521 MappableExprsHandler MEHandler(D, CGF); 10522 MEHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes); 10523 10524 TargetDataInfo Info; 10525 // Fill up the arrays and create the arguments. 10526 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 10527 emitOffloadingArraysArgument(CGF, Info.BasePointersArray, 10528 Info.PointersArray, Info.SizesArray, 10529 Info.MapTypesArray, Info); 10530 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 10531 InputInfo.BasePointersArray = 10532 Address(Info.BasePointersArray, CGM.getPointerAlign()); 10533 InputInfo.PointersArray = 10534 Address(Info.PointersArray, CGM.getPointerAlign()); 10535 InputInfo.SizesArray = 10536 Address(Info.SizesArray, CGM.getPointerAlign()); 10537 MapTypesArray = Info.MapTypesArray; 10538 if (D.hasClausesOfKind<OMPDependClause>()) 10539 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 10540 else 10541 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 10542 }; 10543 10544 if (IfCond) { 10545 emitIfClause(CGF, IfCond, TargetThenGen, 10546 [](CodeGenFunction &CGF, PrePostActionTy &) {}); 10547 } else { 10548 RegionCodeGenTy ThenRCG(TargetThenGen); 10549 ThenRCG(CGF); 10550 } 10551 } 10552 10553 namespace { 10554 /// Kind of parameter in a function with 'declare simd' directive. 10555 enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector }; 10556 /// Attribute set of the parameter. 10557 struct ParamAttrTy { 10558 ParamKindTy Kind = Vector; 10559 llvm::APSInt StrideOrArg; 10560 llvm::APSInt Alignment; 10561 }; 10562 } // namespace 10563 10564 static unsigned evaluateCDTSize(const FunctionDecl *FD, 10565 ArrayRef<ParamAttrTy> ParamAttrs) { 10566 // Every vector variant of a SIMD-enabled function has a vector length (VLEN). 10567 // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument 10568 // of that clause. The VLEN value must be power of 2. 10569 // In other case the notion of the function`s "characteristic data type" (CDT) 10570 // is used to compute the vector length. 10571 // CDT is defined in the following order: 10572 // a) For non-void function, the CDT is the return type. 10573 // b) If the function has any non-uniform, non-linear parameters, then the 10574 // CDT is the type of the first such parameter. 10575 // c) If the CDT determined by a) or b) above is struct, union, or class 10576 // type which is pass-by-value (except for the type that maps to the 10577 // built-in complex data type), the characteristic data type is int. 10578 // d) If none of the above three cases is applicable, the CDT is int. 10579 // The VLEN is then determined based on the CDT and the size of vector 10580 // register of that ISA for which current vector version is generated. The 10581 // VLEN is computed using the formula below: 10582 // VLEN = sizeof(vector_register) / sizeof(CDT), 10583 // where vector register size specified in section 3.2.1 Registers and the 10584 // Stack Frame of original AMD64 ABI document. 10585 QualType RetType = FD->getReturnType(); 10586 if (RetType.isNull()) 10587 return 0; 10588 ASTContext &C = FD->getASTContext(); 10589 QualType CDT; 10590 if (!RetType.isNull() && !RetType->isVoidType()) { 10591 CDT = RetType; 10592 } else { 10593 unsigned Offset = 0; 10594 if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) { 10595 if (ParamAttrs[Offset].Kind == Vector) 10596 CDT = C.getPointerType(C.getRecordType(MD->getParent())); 10597 ++Offset; 10598 } 10599 if (CDT.isNull()) { 10600 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 10601 if (ParamAttrs[I + Offset].Kind == Vector) { 10602 CDT = FD->getParamDecl(I)->getType(); 10603 break; 10604 } 10605 } 10606 } 10607 } 10608 if (CDT.isNull()) 10609 CDT = C.IntTy; 10610 CDT = CDT->getCanonicalTypeUnqualified(); 10611 if (CDT->isRecordType() || CDT->isUnionType()) 10612 CDT = C.IntTy; 10613 return C.getTypeSize(CDT); 10614 } 10615 10616 static void 10617 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn, 10618 const llvm::APSInt &VLENVal, 10619 ArrayRef<ParamAttrTy> ParamAttrs, 10620 OMPDeclareSimdDeclAttr::BranchStateTy State) { 10621 struct ISADataTy { 10622 char ISA; 10623 unsigned VecRegSize; 10624 }; 10625 ISADataTy ISAData[] = { 10626 { 10627 'b', 128 10628 }, // SSE 10629 { 10630 'c', 256 10631 }, // AVX 10632 { 10633 'd', 256 10634 }, // AVX2 10635 { 10636 'e', 512 10637 }, // AVX512 10638 }; 10639 llvm::SmallVector<char, 2> Masked; 10640 switch (State) { 10641 case OMPDeclareSimdDeclAttr::BS_Undefined: 10642 Masked.push_back('N'); 10643 Masked.push_back('M'); 10644 break; 10645 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10646 Masked.push_back('N'); 10647 break; 10648 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10649 Masked.push_back('M'); 10650 break; 10651 } 10652 for (char Mask : Masked) { 10653 for (const ISADataTy &Data : ISAData) { 10654 SmallString<256> Buffer; 10655 llvm::raw_svector_ostream Out(Buffer); 10656 Out << "_ZGV" << Data.ISA << Mask; 10657 if (!VLENVal) { 10658 unsigned NumElts = evaluateCDTSize(FD, ParamAttrs); 10659 assert(NumElts && "Non-zero simdlen/cdtsize expected"); 10660 Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts); 10661 } else { 10662 Out << VLENVal; 10663 } 10664 for (const ParamAttrTy &ParamAttr : ParamAttrs) { 10665 switch (ParamAttr.Kind){ 10666 case LinearWithVarStride: 10667 Out << 's' << ParamAttr.StrideOrArg; 10668 break; 10669 case Linear: 10670 Out << 'l'; 10671 if (!!ParamAttr.StrideOrArg) 10672 Out << ParamAttr.StrideOrArg; 10673 break; 10674 case Uniform: 10675 Out << 'u'; 10676 break; 10677 case Vector: 10678 Out << 'v'; 10679 break; 10680 } 10681 if (!!ParamAttr.Alignment) 10682 Out << 'a' << ParamAttr.Alignment; 10683 } 10684 Out << '_' << Fn->getName(); 10685 Fn->addFnAttr(Out.str()); 10686 } 10687 } 10688 } 10689 10690 // This are the Functions that are needed to mangle the name of the 10691 // vector functions generated by the compiler, according to the rules 10692 // defined in the "Vector Function ABI specifications for AArch64", 10693 // available at 10694 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi. 10695 10696 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI. 10697 /// 10698 /// TODO: Need to implement the behavior for reference marked with a 10699 /// var or no linear modifiers (1.b in the section). For this, we 10700 /// need to extend ParamKindTy to support the linear modifiers. 10701 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) { 10702 QT = QT.getCanonicalType(); 10703 10704 if (QT->isVoidType()) 10705 return false; 10706 10707 if (Kind == ParamKindTy::Uniform) 10708 return false; 10709 10710 if (Kind == ParamKindTy::Linear) 10711 return false; 10712 10713 // TODO: Handle linear references with modifiers 10714 10715 if (Kind == ParamKindTy::LinearWithVarStride) 10716 return false; 10717 10718 return true; 10719 } 10720 10721 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI. 10722 static bool getAArch64PBV(QualType QT, ASTContext &C) { 10723 QT = QT.getCanonicalType(); 10724 unsigned Size = C.getTypeSize(QT); 10725 10726 // Only scalars and complex within 16 bytes wide set PVB to true. 10727 if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128) 10728 return false; 10729 10730 if (QT->isFloatingType()) 10731 return true; 10732 10733 if (QT->isIntegerType()) 10734 return true; 10735 10736 if (QT->isPointerType()) 10737 return true; 10738 10739 // TODO: Add support for complex types (section 3.1.2, item 2). 10740 10741 return false; 10742 } 10743 10744 /// Computes the lane size (LS) of a return type or of an input parameter, 10745 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI. 10746 /// TODO: Add support for references, section 3.2.1, item 1. 10747 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) { 10748 if (getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) { 10749 QualType PTy = QT.getCanonicalType()->getPointeeType(); 10750 if (getAArch64PBV(PTy, C)) 10751 return C.getTypeSize(PTy); 10752 } 10753 if (getAArch64PBV(QT, C)) 10754 return C.getTypeSize(QT); 10755 10756 return C.getTypeSize(C.getUIntPtrType()); 10757 } 10758 10759 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the 10760 // signature of the scalar function, as defined in 3.2.2 of the 10761 // AAVFABI. 10762 static std::tuple<unsigned, unsigned, bool> 10763 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) { 10764 QualType RetType = FD->getReturnType().getCanonicalType(); 10765 10766 ASTContext &C = FD->getASTContext(); 10767 10768 bool OutputBecomesInput = false; 10769 10770 llvm::SmallVector<unsigned, 8> Sizes; 10771 if (!RetType->isVoidType()) { 10772 Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C)); 10773 if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {})) 10774 OutputBecomesInput = true; 10775 } 10776 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 10777 QualType QT = FD->getParamDecl(I)->getType().getCanonicalType(); 10778 Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C)); 10779 } 10780 10781 assert(!Sizes.empty() && "Unable to determine NDS and WDS."); 10782 // The LS of a function parameter / return value can only be a power 10783 // of 2, starting from 8 bits, up to 128. 10784 assert(std::all_of(Sizes.begin(), Sizes.end(), 10785 [](unsigned Size) { 10786 return Size == 8 || Size == 16 || Size == 32 || 10787 Size == 64 || Size == 128; 10788 }) && 10789 "Invalid size"); 10790 10791 return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)), 10792 *std::max_element(std::begin(Sizes), std::end(Sizes)), 10793 OutputBecomesInput); 10794 } 10795 10796 /// Mangle the parameter part of the vector function name according to 10797 /// their OpenMP classification. The mangling function is defined in 10798 /// section 3.5 of the AAVFABI. 10799 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) { 10800 SmallString<256> Buffer; 10801 llvm::raw_svector_ostream Out(Buffer); 10802 for (const auto &ParamAttr : ParamAttrs) { 10803 switch (ParamAttr.Kind) { 10804 case LinearWithVarStride: 10805 Out << "ls" << ParamAttr.StrideOrArg; 10806 break; 10807 case Linear: 10808 Out << 'l'; 10809 // Don't print the step value if it is not present or if it is 10810 // equal to 1. 10811 if (!!ParamAttr.StrideOrArg && ParamAttr.StrideOrArg != 1) 10812 Out << ParamAttr.StrideOrArg; 10813 break; 10814 case Uniform: 10815 Out << 'u'; 10816 break; 10817 case Vector: 10818 Out << 'v'; 10819 break; 10820 } 10821 10822 if (!!ParamAttr.Alignment) 10823 Out << 'a' << ParamAttr.Alignment; 10824 } 10825 10826 return std::string(Out.str()); 10827 } 10828 10829 // Function used to add the attribute. The parameter `VLEN` is 10830 // templated to allow the use of "x" when targeting scalable functions 10831 // for SVE. 10832 template <typename T> 10833 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix, 10834 char ISA, StringRef ParSeq, 10835 StringRef MangledName, bool OutputBecomesInput, 10836 llvm::Function *Fn) { 10837 SmallString<256> Buffer; 10838 llvm::raw_svector_ostream Out(Buffer); 10839 Out << Prefix << ISA << LMask << VLEN; 10840 if (OutputBecomesInput) 10841 Out << "v"; 10842 Out << ParSeq << "_" << MangledName; 10843 Fn->addFnAttr(Out.str()); 10844 } 10845 10846 // Helper function to generate the Advanced SIMD names depending on 10847 // the value of the NDS when simdlen is not present. 10848 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask, 10849 StringRef Prefix, char ISA, 10850 StringRef ParSeq, StringRef MangledName, 10851 bool OutputBecomesInput, 10852 llvm::Function *Fn) { 10853 switch (NDS) { 10854 case 8: 10855 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 10856 OutputBecomesInput, Fn); 10857 addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName, 10858 OutputBecomesInput, Fn); 10859 break; 10860 case 16: 10861 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 10862 OutputBecomesInput, Fn); 10863 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 10864 OutputBecomesInput, Fn); 10865 break; 10866 case 32: 10867 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 10868 OutputBecomesInput, Fn); 10869 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 10870 OutputBecomesInput, Fn); 10871 break; 10872 case 64: 10873 case 128: 10874 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 10875 OutputBecomesInput, Fn); 10876 break; 10877 default: 10878 llvm_unreachable("Scalar type is too wide."); 10879 } 10880 } 10881 10882 /// Emit vector function attributes for AArch64, as defined in the AAVFABI. 10883 static void emitAArch64DeclareSimdFunction( 10884 CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN, 10885 ArrayRef<ParamAttrTy> ParamAttrs, 10886 OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName, 10887 char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) { 10888 10889 // Get basic data for building the vector signature. 10890 const auto Data = getNDSWDS(FD, ParamAttrs); 10891 const unsigned NDS = std::get<0>(Data); 10892 const unsigned WDS = std::get<1>(Data); 10893 const bool OutputBecomesInput = std::get<2>(Data); 10894 10895 // Check the values provided via `simdlen` by the user. 10896 // 1. A `simdlen(1)` doesn't produce vector signatures, 10897 if (UserVLEN == 1) { 10898 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10899 DiagnosticsEngine::Warning, 10900 "The clause simdlen(1) has no effect when targeting aarch64."); 10901 CGM.getDiags().Report(SLoc, DiagID); 10902 return; 10903 } 10904 10905 // 2. Section 3.3.1, item 1: user input must be a power of 2 for 10906 // Advanced SIMD output. 10907 if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) { 10908 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10909 DiagnosticsEngine::Warning, "The value specified in simdlen must be a " 10910 "power of 2 when targeting Advanced SIMD."); 10911 CGM.getDiags().Report(SLoc, DiagID); 10912 return; 10913 } 10914 10915 // 3. Section 3.4.1. SVE fixed lengh must obey the architectural 10916 // limits. 10917 if (ISA == 's' && UserVLEN != 0) { 10918 if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) { 10919 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10920 DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit " 10921 "lanes in the architectural constraints " 10922 "for SVE (min is 128-bit, max is " 10923 "2048-bit, by steps of 128-bit)"); 10924 CGM.getDiags().Report(SLoc, DiagID) << WDS; 10925 return; 10926 } 10927 } 10928 10929 // Sort out parameter sequence. 10930 const std::string ParSeq = mangleVectorParameters(ParamAttrs); 10931 StringRef Prefix = "_ZGV"; 10932 // Generate simdlen from user input (if any). 10933 if (UserVLEN) { 10934 if (ISA == 's') { 10935 // SVE generates only a masked function. 10936 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10937 OutputBecomesInput, Fn); 10938 } else { 10939 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 10940 // Advanced SIMD generates one or two functions, depending on 10941 // the `[not]inbranch` clause. 10942 switch (State) { 10943 case OMPDeclareSimdDeclAttr::BS_Undefined: 10944 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 10945 OutputBecomesInput, Fn); 10946 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10947 OutputBecomesInput, Fn); 10948 break; 10949 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10950 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 10951 OutputBecomesInput, Fn); 10952 break; 10953 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10954 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10955 OutputBecomesInput, Fn); 10956 break; 10957 } 10958 } 10959 } else { 10960 // If no user simdlen is provided, follow the AAVFABI rules for 10961 // generating the vector length. 10962 if (ISA == 's') { 10963 // SVE, section 3.4.1, item 1. 10964 addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName, 10965 OutputBecomesInput, Fn); 10966 } else { 10967 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 10968 // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or 10969 // two vector names depending on the use of the clause 10970 // `[not]inbranch`. 10971 switch (State) { 10972 case OMPDeclareSimdDeclAttr::BS_Undefined: 10973 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 10974 OutputBecomesInput, Fn); 10975 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 10976 OutputBecomesInput, Fn); 10977 break; 10978 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10979 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 10980 OutputBecomesInput, Fn); 10981 break; 10982 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10983 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 10984 OutputBecomesInput, Fn); 10985 break; 10986 } 10987 } 10988 } 10989 } 10990 10991 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD, 10992 llvm::Function *Fn) { 10993 ASTContext &C = CGM.getContext(); 10994 FD = FD->getMostRecentDecl(); 10995 // Map params to their positions in function decl. 10996 llvm::DenseMap<const Decl *, unsigned> ParamPositions; 10997 if (isa<CXXMethodDecl>(FD)) 10998 ParamPositions.try_emplace(FD, 0); 10999 unsigned ParamPos = ParamPositions.size(); 11000 for (const ParmVarDecl *P : FD->parameters()) { 11001 ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos); 11002 ++ParamPos; 11003 } 11004 while (FD) { 11005 for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) { 11006 llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size()); 11007 // Mark uniform parameters. 11008 for (const Expr *E : Attr->uniforms()) { 11009 E = E->IgnoreParenImpCasts(); 11010 unsigned Pos; 11011 if (isa<CXXThisExpr>(E)) { 11012 Pos = ParamPositions[FD]; 11013 } else { 11014 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 11015 ->getCanonicalDecl(); 11016 Pos = ParamPositions[PVD]; 11017 } 11018 ParamAttrs[Pos].Kind = Uniform; 11019 } 11020 // Get alignment info. 11021 auto NI = Attr->alignments_begin(); 11022 for (const Expr *E : Attr->aligneds()) { 11023 E = E->IgnoreParenImpCasts(); 11024 unsigned Pos; 11025 QualType ParmTy; 11026 if (isa<CXXThisExpr>(E)) { 11027 Pos = ParamPositions[FD]; 11028 ParmTy = E->getType(); 11029 } else { 11030 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 11031 ->getCanonicalDecl(); 11032 Pos = ParamPositions[PVD]; 11033 ParmTy = PVD->getType(); 11034 } 11035 ParamAttrs[Pos].Alignment = 11036 (*NI) 11037 ? (*NI)->EvaluateKnownConstInt(C) 11038 : llvm::APSInt::getUnsigned( 11039 C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy)) 11040 .getQuantity()); 11041 ++NI; 11042 } 11043 // Mark linear parameters. 11044 auto SI = Attr->steps_begin(); 11045 auto MI = Attr->modifiers_begin(); 11046 for (const Expr *E : Attr->linears()) { 11047 E = E->IgnoreParenImpCasts(); 11048 unsigned Pos; 11049 if (isa<CXXThisExpr>(E)) { 11050 Pos = ParamPositions[FD]; 11051 } else { 11052 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 11053 ->getCanonicalDecl(); 11054 Pos = ParamPositions[PVD]; 11055 } 11056 ParamAttrTy &ParamAttr = ParamAttrs[Pos]; 11057 ParamAttr.Kind = Linear; 11058 if (*SI) { 11059 Expr::EvalResult Result; 11060 if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) { 11061 if (const auto *DRE = 11062 cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) { 11063 if (const auto *StridePVD = cast<ParmVarDecl>(DRE->getDecl())) { 11064 ParamAttr.Kind = LinearWithVarStride; 11065 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned( 11066 ParamPositions[StridePVD->getCanonicalDecl()]); 11067 } 11068 } 11069 } else { 11070 ParamAttr.StrideOrArg = Result.Val.getInt(); 11071 } 11072 } 11073 ++SI; 11074 ++MI; 11075 } 11076 llvm::APSInt VLENVal; 11077 SourceLocation ExprLoc; 11078 const Expr *VLENExpr = Attr->getSimdlen(); 11079 if (VLENExpr) { 11080 VLENVal = VLENExpr->EvaluateKnownConstInt(C); 11081 ExprLoc = VLENExpr->getExprLoc(); 11082 } 11083 OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState(); 11084 if (CGM.getTriple().isX86()) { 11085 emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State); 11086 } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) { 11087 unsigned VLEN = VLENVal.getExtValue(); 11088 StringRef MangledName = Fn->getName(); 11089 if (CGM.getTarget().hasFeature("sve")) 11090 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 11091 MangledName, 's', 128, Fn, ExprLoc); 11092 if (CGM.getTarget().hasFeature("neon")) 11093 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 11094 MangledName, 'n', 128, Fn, ExprLoc); 11095 } 11096 } 11097 FD = FD->getPreviousDecl(); 11098 } 11099 } 11100 11101 namespace { 11102 /// Cleanup action for doacross support. 11103 class DoacrossCleanupTy final : public EHScopeStack::Cleanup { 11104 public: 11105 static const int DoacrossFinArgs = 2; 11106 11107 private: 11108 llvm::FunctionCallee RTLFn; 11109 llvm::Value *Args[DoacrossFinArgs]; 11110 11111 public: 11112 DoacrossCleanupTy(llvm::FunctionCallee RTLFn, 11113 ArrayRef<llvm::Value *> CallArgs) 11114 : RTLFn(RTLFn) { 11115 assert(CallArgs.size() == DoacrossFinArgs); 11116 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 11117 } 11118 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 11119 if (!CGF.HaveInsertPoint()) 11120 return; 11121 CGF.EmitRuntimeCall(RTLFn, Args); 11122 } 11123 }; 11124 } // namespace 11125 11126 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF, 11127 const OMPLoopDirective &D, 11128 ArrayRef<Expr *> NumIterations) { 11129 if (!CGF.HaveInsertPoint()) 11130 return; 11131 11132 ASTContext &C = CGM.getContext(); 11133 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 11134 RecordDecl *RD; 11135 if (KmpDimTy.isNull()) { 11136 // Build struct kmp_dim { // loop bounds info casted to kmp_int64 11137 // kmp_int64 lo; // lower 11138 // kmp_int64 up; // upper 11139 // kmp_int64 st; // stride 11140 // }; 11141 RD = C.buildImplicitRecord("kmp_dim"); 11142 RD->startDefinition(); 11143 addFieldToRecordDecl(C, RD, Int64Ty); 11144 addFieldToRecordDecl(C, RD, Int64Ty); 11145 addFieldToRecordDecl(C, RD, Int64Ty); 11146 RD->completeDefinition(); 11147 KmpDimTy = C.getRecordType(RD); 11148 } else { 11149 RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl()); 11150 } 11151 llvm::APInt Size(/*numBits=*/32, NumIterations.size()); 11152 QualType ArrayTy = 11153 C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0); 11154 11155 Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims"); 11156 CGF.EmitNullInitialization(DimsAddr, ArrayTy); 11157 enum { LowerFD = 0, UpperFD, StrideFD }; 11158 // Fill dims with data. 11159 for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) { 11160 LValue DimsLVal = CGF.MakeAddrLValue( 11161 CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy); 11162 // dims.upper = num_iterations; 11163 LValue UpperLVal = CGF.EmitLValueForField( 11164 DimsLVal, *std::next(RD->field_begin(), UpperFD)); 11165 llvm::Value *NumIterVal = 11166 CGF.EmitScalarConversion(CGF.EmitScalarExpr(NumIterations[I]), 11167 D.getNumIterations()->getType(), Int64Ty, 11168 D.getNumIterations()->getExprLoc()); 11169 CGF.EmitStoreOfScalar(NumIterVal, UpperLVal); 11170 // dims.stride = 1; 11171 LValue StrideLVal = CGF.EmitLValueForField( 11172 DimsLVal, *std::next(RD->field_begin(), StrideFD)); 11173 CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1), 11174 StrideLVal); 11175 } 11176 11177 // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, 11178 // kmp_int32 num_dims, struct kmp_dim * dims); 11179 llvm::Value *Args[] = { 11180 emitUpdateLocation(CGF, D.getBeginLoc()), 11181 getThreadID(CGF, D.getBeginLoc()), 11182 llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()), 11183 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11184 CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(), 11185 CGM.VoidPtrTy)}; 11186 11187 llvm::FunctionCallee RTLFn = 11188 createRuntimeFunction(OMPRTL__kmpc_doacross_init); 11189 CGF.EmitRuntimeCall(RTLFn, Args); 11190 llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = { 11191 emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())}; 11192 llvm::FunctionCallee FiniRTLFn = 11193 createRuntimeFunction(OMPRTL__kmpc_doacross_fini); 11194 CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 11195 llvm::makeArrayRef(FiniArgs)); 11196 } 11197 11198 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 11199 const OMPDependClause *C) { 11200 QualType Int64Ty = 11201 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 11202 llvm::APInt Size(/*numBits=*/32, C->getNumLoops()); 11203 QualType ArrayTy = CGM.getContext().getConstantArrayType( 11204 Int64Ty, Size, nullptr, ArrayType::Normal, 0); 11205 Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr"); 11206 for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) { 11207 const Expr *CounterVal = C->getLoopData(I); 11208 assert(CounterVal); 11209 llvm::Value *CntVal = CGF.EmitScalarConversion( 11210 CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty, 11211 CounterVal->getExprLoc()); 11212 CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I), 11213 /*Volatile=*/false, Int64Ty); 11214 } 11215 llvm::Value *Args[] = { 11216 emitUpdateLocation(CGF, C->getBeginLoc()), 11217 getThreadID(CGF, C->getBeginLoc()), 11218 CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()}; 11219 llvm::FunctionCallee RTLFn; 11220 if (C->getDependencyKind() == OMPC_DEPEND_source) { 11221 RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_post); 11222 } else { 11223 assert(C->getDependencyKind() == OMPC_DEPEND_sink); 11224 RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_wait); 11225 } 11226 CGF.EmitRuntimeCall(RTLFn, Args); 11227 } 11228 11229 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc, 11230 llvm::FunctionCallee Callee, 11231 ArrayRef<llvm::Value *> Args) const { 11232 assert(Loc.isValid() && "Outlined function call location must be valid."); 11233 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 11234 11235 if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) { 11236 if (Fn->doesNotThrow()) { 11237 CGF.EmitNounwindRuntimeCall(Fn, Args); 11238 return; 11239 } 11240 } 11241 CGF.EmitRuntimeCall(Callee, Args); 11242 } 11243 11244 void CGOpenMPRuntime::emitOutlinedFunctionCall( 11245 CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn, 11246 ArrayRef<llvm::Value *> Args) const { 11247 emitCall(CGF, Loc, OutlinedFn, Args); 11248 } 11249 11250 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) { 11251 if (const auto *FD = dyn_cast<FunctionDecl>(D)) 11252 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD)) 11253 HasEmittedDeclareTargetRegion = true; 11254 } 11255 11256 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF, 11257 const VarDecl *NativeParam, 11258 const VarDecl *TargetParam) const { 11259 return CGF.GetAddrOfLocalVar(NativeParam); 11260 } 11261 11262 namespace { 11263 /// Cleanup action for allocate support. 11264 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup { 11265 public: 11266 static const int CleanupArgs = 3; 11267 11268 private: 11269 llvm::FunctionCallee RTLFn; 11270 llvm::Value *Args[CleanupArgs]; 11271 11272 public: 11273 OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn, 11274 ArrayRef<llvm::Value *> CallArgs) 11275 : RTLFn(RTLFn) { 11276 assert(CallArgs.size() == CleanupArgs && 11277 "Size of arguments does not match."); 11278 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 11279 } 11280 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 11281 if (!CGF.HaveInsertPoint()) 11282 return; 11283 CGF.EmitRuntimeCall(RTLFn, Args); 11284 } 11285 }; 11286 } // namespace 11287 11288 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF, 11289 const VarDecl *VD) { 11290 if (!VD) 11291 return Address::invalid(); 11292 const VarDecl *CVD = VD->getCanonicalDecl(); 11293 if (!CVD->hasAttr<OMPAllocateDeclAttr>()) 11294 return Address::invalid(); 11295 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>(); 11296 // Use the default allocation. 11297 if (AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc && 11298 !AA->getAllocator()) 11299 return Address::invalid(); 11300 llvm::Value *Size; 11301 CharUnits Align = CGM.getContext().getDeclAlign(CVD); 11302 if (CVD->getType()->isVariablyModifiedType()) { 11303 Size = CGF.getTypeSize(CVD->getType()); 11304 // Align the size: ((size + align - 1) / align) * align 11305 Size = CGF.Builder.CreateNUWAdd( 11306 Size, CGM.getSize(Align - CharUnits::fromQuantity(1))); 11307 Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align)); 11308 Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align)); 11309 } else { 11310 CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType()); 11311 Size = CGM.getSize(Sz.alignTo(Align)); 11312 } 11313 llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc()); 11314 assert(AA->getAllocator() && 11315 "Expected allocator expression for non-default allocator."); 11316 llvm::Value *Allocator = CGF.EmitScalarExpr(AA->getAllocator()); 11317 // According to the standard, the original allocator type is a enum (integer). 11318 // Convert to pointer type, if required. 11319 if (Allocator->getType()->isIntegerTy()) 11320 Allocator = CGF.Builder.CreateIntToPtr(Allocator, CGM.VoidPtrTy); 11321 else if (Allocator->getType()->isPointerTy()) 11322 Allocator = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Allocator, 11323 CGM.VoidPtrTy); 11324 llvm::Value *Args[] = {ThreadID, Size, Allocator}; 11325 11326 llvm::Value *Addr = 11327 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_alloc), Args, 11328 getName({CVD->getName(), ".void.addr"})); 11329 llvm::Value *FiniArgs[OMPAllocateCleanupTy::CleanupArgs] = {ThreadID, Addr, 11330 Allocator}; 11331 llvm::FunctionCallee FiniRTLFn = createRuntimeFunction(OMPRTL__kmpc_free); 11332 11333 CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 11334 llvm::makeArrayRef(FiniArgs)); 11335 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11336 Addr, 11337 CGF.ConvertTypeForMem(CGM.getContext().getPointerType(CVD->getType())), 11338 getName({CVD->getName(), ".addr"})); 11339 return Address(Addr, Align); 11340 } 11341 11342 CGOpenMPRuntime::NontemporalDeclsRAII::NontemporalDeclsRAII( 11343 CodeGenModule &CGM, const OMPLoopDirective &S) 11344 : CGM(CGM), NeedToPush(S.hasClausesOfKind<OMPNontemporalClause>()) { 11345 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11346 if (!NeedToPush) 11347 return; 11348 NontemporalDeclsSet &DS = 11349 CGM.getOpenMPRuntime().NontemporalDeclsStack.emplace_back(); 11350 for (const auto *C : S.getClausesOfKind<OMPNontemporalClause>()) { 11351 for (const Stmt *Ref : C->private_refs()) { 11352 const auto *SimpleRefExpr = cast<Expr>(Ref)->IgnoreParenImpCasts(); 11353 const ValueDecl *VD; 11354 if (const auto *DRE = dyn_cast<DeclRefExpr>(SimpleRefExpr)) { 11355 VD = DRE->getDecl(); 11356 } else { 11357 const auto *ME = cast<MemberExpr>(SimpleRefExpr); 11358 assert((ME->isImplicitCXXThis() || 11359 isa<CXXThisExpr>(ME->getBase()->IgnoreParenImpCasts())) && 11360 "Expected member of current class."); 11361 VD = ME->getMemberDecl(); 11362 } 11363 DS.insert(VD); 11364 } 11365 } 11366 } 11367 11368 CGOpenMPRuntime::NontemporalDeclsRAII::~NontemporalDeclsRAII() { 11369 if (!NeedToPush) 11370 return; 11371 CGM.getOpenMPRuntime().NontemporalDeclsStack.pop_back(); 11372 } 11373 11374 bool CGOpenMPRuntime::isNontemporalDecl(const ValueDecl *VD) const { 11375 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11376 11377 return llvm::any_of( 11378 CGM.getOpenMPRuntime().NontemporalDeclsStack, 11379 [VD](const NontemporalDeclsSet &Set) { return Set.count(VD) > 0; }); 11380 } 11381 11382 void CGOpenMPRuntime::LastprivateConditionalRAII::tryToDisableInnerAnalysis( 11383 const OMPExecutableDirective &S, 11384 llvm::DenseSet<CanonicalDeclPtr<const Decl>> &NeedToAddForLPCsAsDisabled) 11385 const { 11386 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToCheckForLPCs; 11387 // Vars in target/task regions must be excluded completely. 11388 if (isOpenMPTargetExecutionDirective(S.getDirectiveKind()) || 11389 isOpenMPTaskingDirective(S.getDirectiveKind())) { 11390 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions; 11391 getOpenMPCaptureRegions(CaptureRegions, S.getDirectiveKind()); 11392 const CapturedStmt *CS = S.getCapturedStmt(CaptureRegions.front()); 11393 for (const CapturedStmt::Capture &Cap : CS->captures()) { 11394 if (Cap.capturesVariable() || Cap.capturesVariableByCopy()) 11395 NeedToCheckForLPCs.insert(Cap.getCapturedVar()); 11396 } 11397 } 11398 // Exclude vars in private clauses. 11399 for (const auto *C : S.getClausesOfKind<OMPPrivateClause>()) { 11400 for (const Expr *Ref : C->varlists()) { 11401 if (!Ref->getType()->isScalarType()) 11402 continue; 11403 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11404 if (!DRE) 11405 continue; 11406 NeedToCheckForLPCs.insert(DRE->getDecl()); 11407 } 11408 } 11409 for (const auto *C : S.getClausesOfKind<OMPFirstprivateClause>()) { 11410 for (const Expr *Ref : C->varlists()) { 11411 if (!Ref->getType()->isScalarType()) 11412 continue; 11413 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11414 if (!DRE) 11415 continue; 11416 NeedToCheckForLPCs.insert(DRE->getDecl()); 11417 } 11418 } 11419 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) { 11420 for (const Expr *Ref : C->varlists()) { 11421 if (!Ref->getType()->isScalarType()) 11422 continue; 11423 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11424 if (!DRE) 11425 continue; 11426 NeedToCheckForLPCs.insert(DRE->getDecl()); 11427 } 11428 } 11429 for (const auto *C : S.getClausesOfKind<OMPReductionClause>()) { 11430 for (const Expr *Ref : C->varlists()) { 11431 if (!Ref->getType()->isScalarType()) 11432 continue; 11433 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11434 if (!DRE) 11435 continue; 11436 NeedToCheckForLPCs.insert(DRE->getDecl()); 11437 } 11438 } 11439 for (const auto *C : S.getClausesOfKind<OMPLinearClause>()) { 11440 for (const Expr *Ref : C->varlists()) { 11441 if (!Ref->getType()->isScalarType()) 11442 continue; 11443 const auto *DRE = dyn_cast<DeclRefExpr>(Ref->IgnoreParenImpCasts()); 11444 if (!DRE) 11445 continue; 11446 NeedToCheckForLPCs.insert(DRE->getDecl()); 11447 } 11448 } 11449 for (const Decl *VD : NeedToCheckForLPCs) { 11450 for (const LastprivateConditionalData &Data : 11451 llvm::reverse(CGM.getOpenMPRuntime().LastprivateConditionalStack)) { 11452 if (Data.DeclToUniqueName.count(VD) > 0) { 11453 if (!Data.Disabled) 11454 NeedToAddForLPCsAsDisabled.insert(VD); 11455 break; 11456 } 11457 } 11458 } 11459 } 11460 11461 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII( 11462 CodeGenFunction &CGF, const OMPExecutableDirective &S, LValue IVLVal) 11463 : CGM(CGF.CGM), 11464 Action((CGM.getLangOpts().OpenMP >= 50 && 11465 llvm::any_of(S.getClausesOfKind<OMPLastprivateClause>(), 11466 [](const OMPLastprivateClause *C) { 11467 return C->getKind() == 11468 OMPC_LASTPRIVATE_conditional; 11469 })) 11470 ? ActionToDo::PushAsLastprivateConditional 11471 : ActionToDo::DoNotPush) { 11472 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11473 if (CGM.getLangOpts().OpenMP < 50 || Action == ActionToDo::DoNotPush) 11474 return; 11475 assert(Action == ActionToDo::PushAsLastprivateConditional && 11476 "Expected a push action."); 11477 LastprivateConditionalData &Data = 11478 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back(); 11479 for (const auto *C : S.getClausesOfKind<OMPLastprivateClause>()) { 11480 if (C->getKind() != OMPC_LASTPRIVATE_conditional) 11481 continue; 11482 11483 for (const Expr *Ref : C->varlists()) { 11484 Data.DeclToUniqueName.insert(std::make_pair( 11485 cast<DeclRefExpr>(Ref->IgnoreParenImpCasts())->getDecl(), 11486 SmallString<16>(generateUniqueName(CGM, "pl_cond", Ref)))); 11487 } 11488 } 11489 Data.IVLVal = IVLVal; 11490 Data.Fn = CGF.CurFn; 11491 } 11492 11493 CGOpenMPRuntime::LastprivateConditionalRAII::LastprivateConditionalRAII( 11494 CodeGenFunction &CGF, const OMPExecutableDirective &S) 11495 : CGM(CGF.CGM), Action(ActionToDo::DoNotPush) { 11496 assert(CGM.getLangOpts().OpenMP && "Not in OpenMP mode."); 11497 if (CGM.getLangOpts().OpenMP < 50) 11498 return; 11499 llvm::DenseSet<CanonicalDeclPtr<const Decl>> NeedToAddForLPCsAsDisabled; 11500 tryToDisableInnerAnalysis(S, NeedToAddForLPCsAsDisabled); 11501 if (!NeedToAddForLPCsAsDisabled.empty()) { 11502 Action = ActionToDo::DisableLastprivateConditional; 11503 LastprivateConditionalData &Data = 11504 CGM.getOpenMPRuntime().LastprivateConditionalStack.emplace_back(); 11505 for (const Decl *VD : NeedToAddForLPCsAsDisabled) 11506 Data.DeclToUniqueName.insert(std::make_pair(VD, SmallString<16>())); 11507 Data.Fn = CGF.CurFn; 11508 Data.Disabled = true; 11509 } 11510 } 11511 11512 CGOpenMPRuntime::LastprivateConditionalRAII 11513 CGOpenMPRuntime::LastprivateConditionalRAII::disable( 11514 CodeGenFunction &CGF, const OMPExecutableDirective &S) { 11515 return LastprivateConditionalRAII(CGF, S); 11516 } 11517 11518 CGOpenMPRuntime::LastprivateConditionalRAII::~LastprivateConditionalRAII() { 11519 if (CGM.getLangOpts().OpenMP < 50) 11520 return; 11521 if (Action == ActionToDo::DisableLastprivateConditional) { 11522 assert(CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled && 11523 "Expected list of disabled private vars."); 11524 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back(); 11525 } 11526 if (Action == ActionToDo::PushAsLastprivateConditional) { 11527 assert( 11528 !CGM.getOpenMPRuntime().LastprivateConditionalStack.back().Disabled && 11529 "Expected list of lastprivate conditional vars."); 11530 CGM.getOpenMPRuntime().LastprivateConditionalStack.pop_back(); 11531 } 11532 } 11533 11534 Address CGOpenMPRuntime::emitLastprivateConditionalInit(CodeGenFunction &CGF, 11535 const VarDecl *VD) { 11536 ASTContext &C = CGM.getContext(); 11537 auto I = LastprivateConditionalToTypes.find(CGF.CurFn); 11538 if (I == LastprivateConditionalToTypes.end()) 11539 I = LastprivateConditionalToTypes.try_emplace(CGF.CurFn).first; 11540 QualType NewType; 11541 const FieldDecl *VDField; 11542 const FieldDecl *FiredField; 11543 LValue BaseLVal; 11544 auto VI = I->getSecond().find(VD); 11545 if (VI == I->getSecond().end()) { 11546 RecordDecl *RD = C.buildImplicitRecord("lasprivate.conditional"); 11547 RD->startDefinition(); 11548 VDField = addFieldToRecordDecl(C, RD, VD->getType().getNonReferenceType()); 11549 FiredField = addFieldToRecordDecl(C, RD, C.CharTy); 11550 RD->completeDefinition(); 11551 NewType = C.getRecordType(RD); 11552 Address Addr = CGF.CreateMemTemp(NewType, C.getDeclAlign(VD), VD->getName()); 11553 BaseLVal = CGF.MakeAddrLValue(Addr, NewType, AlignmentSource::Decl); 11554 I->getSecond().try_emplace(VD, NewType, VDField, FiredField, BaseLVal); 11555 } else { 11556 NewType = std::get<0>(VI->getSecond()); 11557 VDField = std::get<1>(VI->getSecond()); 11558 FiredField = std::get<2>(VI->getSecond()); 11559 BaseLVal = std::get<3>(VI->getSecond()); 11560 } 11561 LValue FiredLVal = 11562 CGF.EmitLValueForField(BaseLVal, FiredField); 11563 CGF.EmitStoreOfScalar( 11564 llvm::ConstantInt::getNullValue(CGF.ConvertTypeForMem(C.CharTy)), 11565 FiredLVal); 11566 return CGF.EmitLValueForField(BaseLVal, VDField).getAddress(CGF); 11567 } 11568 11569 namespace { 11570 /// Checks if the lastprivate conditional variable is referenced in LHS. 11571 class LastprivateConditionalRefChecker final 11572 : public ConstStmtVisitor<LastprivateConditionalRefChecker, bool> { 11573 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM; 11574 const Expr *FoundE = nullptr; 11575 const Decl *FoundD = nullptr; 11576 StringRef UniqueDeclName; 11577 LValue IVLVal; 11578 llvm::Function *FoundFn = nullptr; 11579 SourceLocation Loc; 11580 11581 public: 11582 bool VisitDeclRefExpr(const DeclRefExpr *E) { 11583 for (const CGOpenMPRuntime::LastprivateConditionalData &D : 11584 llvm::reverse(LPM)) { 11585 auto It = D.DeclToUniqueName.find(E->getDecl()); 11586 if (It == D.DeclToUniqueName.end()) 11587 continue; 11588 if (D.Disabled) 11589 return false; 11590 FoundE = E; 11591 FoundD = E->getDecl()->getCanonicalDecl(); 11592 UniqueDeclName = It->second; 11593 IVLVal = D.IVLVal; 11594 FoundFn = D.Fn; 11595 break; 11596 } 11597 return FoundE == E; 11598 } 11599 bool VisitMemberExpr(const MemberExpr *E) { 11600 if (!CodeGenFunction::IsWrappedCXXThis(E->getBase())) 11601 return false; 11602 for (const CGOpenMPRuntime::LastprivateConditionalData &D : 11603 llvm::reverse(LPM)) { 11604 auto It = D.DeclToUniqueName.find(E->getMemberDecl()); 11605 if (It == D.DeclToUniqueName.end()) 11606 continue; 11607 if (D.Disabled) 11608 return false; 11609 FoundE = E; 11610 FoundD = E->getMemberDecl()->getCanonicalDecl(); 11611 UniqueDeclName = It->second; 11612 IVLVal = D.IVLVal; 11613 FoundFn = D.Fn; 11614 break; 11615 } 11616 return FoundE == E; 11617 } 11618 bool VisitStmt(const Stmt *S) { 11619 for (const Stmt *Child : S->children()) { 11620 if (!Child) 11621 continue; 11622 if (const auto *E = dyn_cast<Expr>(Child)) 11623 if (!E->isGLValue()) 11624 continue; 11625 if (Visit(Child)) 11626 return true; 11627 } 11628 return false; 11629 } 11630 explicit LastprivateConditionalRefChecker( 11631 ArrayRef<CGOpenMPRuntime::LastprivateConditionalData> LPM) 11632 : LPM(LPM) {} 11633 std::tuple<const Expr *, const Decl *, StringRef, LValue, llvm::Function *> 11634 getFoundData() const { 11635 return std::make_tuple(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn); 11636 } 11637 }; 11638 } // namespace 11639 11640 void CGOpenMPRuntime::emitLastprivateConditionalUpdate(CodeGenFunction &CGF, 11641 LValue IVLVal, 11642 StringRef UniqueDeclName, 11643 LValue LVal, 11644 SourceLocation Loc) { 11645 // Last updated loop counter for the lastprivate conditional var. 11646 // int<xx> last_iv = 0; 11647 llvm::Type *LLIVTy = CGF.ConvertTypeForMem(IVLVal.getType()); 11648 llvm::Constant *LastIV = 11649 getOrCreateInternalVariable(LLIVTy, getName({UniqueDeclName, "iv"})); 11650 cast<llvm::GlobalVariable>(LastIV)->setAlignment( 11651 IVLVal.getAlignment().getAsAlign()); 11652 LValue LastIVLVal = CGF.MakeNaturalAlignAddrLValue(LastIV, IVLVal.getType()); 11653 11654 // Last value of the lastprivate conditional. 11655 // decltype(priv_a) last_a; 11656 llvm::Constant *Last = getOrCreateInternalVariable( 11657 CGF.ConvertTypeForMem(LVal.getType()), UniqueDeclName); 11658 cast<llvm::GlobalVariable>(Last)->setAlignment( 11659 LVal.getAlignment().getAsAlign()); 11660 LValue LastLVal = 11661 CGF.MakeAddrLValue(Last, LVal.getType(), LVal.getAlignment()); 11662 11663 // Global loop counter. Required to handle inner parallel-for regions. 11664 // iv 11665 llvm::Value *IVVal = CGF.EmitLoadOfScalar(IVLVal, Loc); 11666 11667 // #pragma omp critical(a) 11668 // if (last_iv <= iv) { 11669 // last_iv = iv; 11670 // last_a = priv_a; 11671 // } 11672 auto &&CodeGen = [&LastIVLVal, &IVLVal, IVVal, &LVal, &LastLVal, 11673 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 11674 Action.Enter(CGF); 11675 llvm::Value *LastIVVal = CGF.EmitLoadOfScalar(LastIVLVal, Loc); 11676 // (last_iv <= iv) ? Check if the variable is updated and store new 11677 // value in global var. 11678 llvm::Value *CmpRes; 11679 if (IVLVal.getType()->isSignedIntegerType()) { 11680 CmpRes = CGF.Builder.CreateICmpSLE(LastIVVal, IVVal); 11681 } else { 11682 assert(IVLVal.getType()->isUnsignedIntegerType() && 11683 "Loop iteration variable must be integer."); 11684 CmpRes = CGF.Builder.CreateICmpULE(LastIVVal, IVVal); 11685 } 11686 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lp_cond_then"); 11687 llvm::BasicBlock *ExitBB = CGF.createBasicBlock("lp_cond_exit"); 11688 CGF.Builder.CreateCondBr(CmpRes, ThenBB, ExitBB); 11689 // { 11690 CGF.EmitBlock(ThenBB); 11691 11692 // last_iv = iv; 11693 CGF.EmitStoreOfScalar(IVVal, LastIVLVal); 11694 11695 // last_a = priv_a; 11696 switch (CGF.getEvaluationKind(LVal.getType())) { 11697 case TEK_Scalar: { 11698 llvm::Value *PrivVal = CGF.EmitLoadOfScalar(LVal, Loc); 11699 CGF.EmitStoreOfScalar(PrivVal, LastLVal); 11700 break; 11701 } 11702 case TEK_Complex: { 11703 CodeGenFunction::ComplexPairTy PrivVal = CGF.EmitLoadOfComplex(LVal, Loc); 11704 CGF.EmitStoreOfComplex(PrivVal, LastLVal, /*isInit=*/false); 11705 break; 11706 } 11707 case TEK_Aggregate: 11708 llvm_unreachable( 11709 "Aggregates are not supported in lastprivate conditional."); 11710 } 11711 // } 11712 CGF.EmitBranch(ExitBB); 11713 // There is no need to emit line number for unconditional branch. 11714 (void)ApplyDebugLocation::CreateEmpty(CGF); 11715 CGF.EmitBlock(ExitBB, /*IsFinished=*/true); 11716 }; 11717 11718 if (CGM.getLangOpts().OpenMPSimd) { 11719 // Do not emit as a critical region as no parallel region could be emitted. 11720 RegionCodeGenTy ThenRCG(CodeGen); 11721 ThenRCG(CGF); 11722 } else { 11723 emitCriticalRegion(CGF, UniqueDeclName, CodeGen, Loc); 11724 } 11725 } 11726 11727 void CGOpenMPRuntime::checkAndEmitLastprivateConditional(CodeGenFunction &CGF, 11728 const Expr *LHS) { 11729 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty()) 11730 return; 11731 LastprivateConditionalRefChecker Checker(LastprivateConditionalStack); 11732 if (!Checker.Visit(LHS)) 11733 return; 11734 const Expr *FoundE; 11735 const Decl *FoundD; 11736 StringRef UniqueDeclName; 11737 LValue IVLVal; 11738 llvm::Function *FoundFn; 11739 std::tie(FoundE, FoundD, UniqueDeclName, IVLVal, FoundFn) = 11740 Checker.getFoundData(); 11741 if (FoundFn != CGF.CurFn) { 11742 // Special codegen for inner parallel regions. 11743 // ((struct.lastprivate.conditional*)&priv_a)->Fired = 1; 11744 auto It = LastprivateConditionalToTypes[FoundFn].find(FoundD); 11745 assert(It != LastprivateConditionalToTypes[FoundFn].end() && 11746 "Lastprivate conditional is not found in outer region."); 11747 QualType StructTy = std::get<0>(It->getSecond()); 11748 const FieldDecl* FiredDecl = std::get<2>(It->getSecond()); 11749 LValue PrivLVal = CGF.EmitLValue(FoundE); 11750 Address StructAddr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11751 PrivLVal.getAddress(CGF), 11752 CGF.ConvertTypeForMem(CGF.getContext().getPointerType(StructTy))); 11753 LValue BaseLVal = 11754 CGF.MakeAddrLValue(StructAddr, StructTy, AlignmentSource::Decl); 11755 LValue FiredLVal = CGF.EmitLValueForField(BaseLVal, FiredDecl); 11756 CGF.EmitAtomicStore(RValue::get(llvm::ConstantInt::get( 11757 CGF.ConvertTypeForMem(FiredDecl->getType()), 1)), 11758 FiredLVal, llvm::AtomicOrdering::Unordered, 11759 /*IsVolatile=*/true, /*isInit=*/false); 11760 return; 11761 } 11762 11763 // Private address of the lastprivate conditional in the current context. 11764 // priv_a 11765 LValue LVal = CGF.EmitLValue(FoundE); 11766 emitLastprivateConditionalUpdate(CGF, IVLVal, UniqueDeclName, LVal, 11767 FoundE->getExprLoc()); 11768 } 11769 11770 void CGOpenMPRuntime::checkAndEmitSharedLastprivateConditional( 11771 CodeGenFunction &CGF, const OMPExecutableDirective &D, 11772 const llvm::DenseSet<CanonicalDeclPtr<const VarDecl>> &IgnoredDecls) { 11773 if (CGF.getLangOpts().OpenMP < 50 || LastprivateConditionalStack.empty()) 11774 return; 11775 auto Range = llvm::reverse(LastprivateConditionalStack); 11776 auto It = llvm::find_if( 11777 Range, [](const LastprivateConditionalData &D) { return !D.Disabled; }); 11778 if (It == Range.end() || It->Fn != CGF.CurFn) 11779 return; 11780 auto LPCI = LastprivateConditionalToTypes.find(It->Fn); 11781 assert(LPCI != LastprivateConditionalToTypes.end() && 11782 "Lastprivates must be registered already."); 11783 SmallVector<OpenMPDirectiveKind, 4> CaptureRegions; 11784 getOpenMPCaptureRegions(CaptureRegions, D.getDirectiveKind()); 11785 const CapturedStmt *CS = D.getCapturedStmt(CaptureRegions.back()); 11786 for (const auto &Pair : It->DeclToUniqueName) { 11787 const auto *VD = cast<VarDecl>(Pair.first->getCanonicalDecl()); 11788 if (!CS->capturesVariable(VD) || IgnoredDecls.count(VD) > 0) 11789 continue; 11790 auto I = LPCI->getSecond().find(Pair.first); 11791 assert(I != LPCI->getSecond().end() && 11792 "Lastprivate must be rehistered already."); 11793 // bool Cmp = priv_a.Fired != 0; 11794 LValue BaseLVal = std::get<3>(I->getSecond()); 11795 LValue FiredLVal = 11796 CGF.EmitLValueForField(BaseLVal, std::get<2>(I->getSecond())); 11797 llvm::Value *Res = CGF.EmitLoadOfScalar(FiredLVal, D.getBeginLoc()); 11798 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Res); 11799 llvm::BasicBlock *ThenBB = CGF.createBasicBlock("lpc.then"); 11800 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("lpc.done"); 11801 // if (Cmp) { 11802 CGF.Builder.CreateCondBr(Cmp, ThenBB, DoneBB); 11803 CGF.EmitBlock(ThenBB); 11804 Address Addr = CGF.GetAddrOfLocalVar(VD); 11805 LValue LVal; 11806 if (VD->getType()->isReferenceType()) 11807 LVal = CGF.EmitLoadOfReferenceLValue(Addr, VD->getType(), 11808 AlignmentSource::Decl); 11809 else 11810 LVal = CGF.MakeAddrLValue(Addr, VD->getType().getNonReferenceType(), 11811 AlignmentSource::Decl); 11812 emitLastprivateConditionalUpdate(CGF, It->IVLVal, Pair.second, LVal, 11813 D.getBeginLoc()); 11814 auto AL = ApplyDebugLocation::CreateArtificial(CGF); 11815 CGF.EmitBlock(DoneBB, /*IsFinal=*/true); 11816 // } 11817 } 11818 } 11819 11820 void CGOpenMPRuntime::emitLastprivateConditionalFinalUpdate( 11821 CodeGenFunction &CGF, LValue PrivLVal, const VarDecl *VD, 11822 SourceLocation Loc) { 11823 if (CGF.getLangOpts().OpenMP < 50) 11824 return; 11825 auto It = LastprivateConditionalStack.back().DeclToUniqueName.find(VD); 11826 assert(It != LastprivateConditionalStack.back().DeclToUniqueName.end() && 11827 "Unknown lastprivate conditional variable."); 11828 StringRef UniqueName = It->second; 11829 llvm::GlobalVariable *GV = CGM.getModule().getNamedGlobal(UniqueName); 11830 // The variable was not updated in the region - exit. 11831 if (!GV) 11832 return; 11833 LValue LPLVal = CGF.MakeAddrLValue( 11834 GV, PrivLVal.getType().getNonReferenceType(), PrivLVal.getAlignment()); 11835 llvm::Value *Res = CGF.EmitLoadOfScalar(LPLVal, Loc); 11836 CGF.EmitStoreOfScalar(Res, PrivLVal); 11837 } 11838 11839 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction( 11840 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11841 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 11842 llvm_unreachable("Not supported in SIMD-only mode"); 11843 } 11844 11845 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction( 11846 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11847 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 11848 llvm_unreachable("Not supported in SIMD-only mode"); 11849 } 11850 11851 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction( 11852 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11853 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 11854 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 11855 bool Tied, unsigned &NumberOfParts) { 11856 llvm_unreachable("Not supported in SIMD-only mode"); 11857 } 11858 11859 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF, 11860 SourceLocation Loc, 11861 llvm::Function *OutlinedFn, 11862 ArrayRef<llvm::Value *> CapturedVars, 11863 const Expr *IfCond) { 11864 llvm_unreachable("Not supported in SIMD-only mode"); 11865 } 11866 11867 void CGOpenMPSIMDRuntime::emitCriticalRegion( 11868 CodeGenFunction &CGF, StringRef CriticalName, 11869 const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc, 11870 const Expr *Hint) { 11871 llvm_unreachable("Not supported in SIMD-only mode"); 11872 } 11873 11874 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF, 11875 const RegionCodeGenTy &MasterOpGen, 11876 SourceLocation Loc) { 11877 llvm_unreachable("Not supported in SIMD-only mode"); 11878 } 11879 11880 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 11881 SourceLocation Loc) { 11882 llvm_unreachable("Not supported in SIMD-only mode"); 11883 } 11884 11885 void CGOpenMPSIMDRuntime::emitTaskgroupRegion( 11886 CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen, 11887 SourceLocation Loc) { 11888 llvm_unreachable("Not supported in SIMD-only mode"); 11889 } 11890 11891 void CGOpenMPSIMDRuntime::emitSingleRegion( 11892 CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen, 11893 SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars, 11894 ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs, 11895 ArrayRef<const Expr *> AssignmentOps) { 11896 llvm_unreachable("Not supported in SIMD-only mode"); 11897 } 11898 11899 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF, 11900 const RegionCodeGenTy &OrderedOpGen, 11901 SourceLocation Loc, 11902 bool IsThreads) { 11903 llvm_unreachable("Not supported in SIMD-only mode"); 11904 } 11905 11906 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF, 11907 SourceLocation Loc, 11908 OpenMPDirectiveKind Kind, 11909 bool EmitChecks, 11910 bool ForceSimpleCall) { 11911 llvm_unreachable("Not supported in SIMD-only mode"); 11912 } 11913 11914 void CGOpenMPSIMDRuntime::emitForDispatchInit( 11915 CodeGenFunction &CGF, SourceLocation Loc, 11916 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 11917 bool Ordered, const DispatchRTInput &DispatchValues) { 11918 llvm_unreachable("Not supported in SIMD-only mode"); 11919 } 11920 11921 void CGOpenMPSIMDRuntime::emitForStaticInit( 11922 CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind, 11923 const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) { 11924 llvm_unreachable("Not supported in SIMD-only mode"); 11925 } 11926 11927 void CGOpenMPSIMDRuntime::emitDistributeStaticInit( 11928 CodeGenFunction &CGF, SourceLocation Loc, 11929 OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) { 11930 llvm_unreachable("Not supported in SIMD-only mode"); 11931 } 11932 11933 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 11934 SourceLocation Loc, 11935 unsigned IVSize, 11936 bool IVSigned) { 11937 llvm_unreachable("Not supported in SIMD-only mode"); 11938 } 11939 11940 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF, 11941 SourceLocation Loc, 11942 OpenMPDirectiveKind DKind) { 11943 llvm_unreachable("Not supported in SIMD-only mode"); 11944 } 11945 11946 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF, 11947 SourceLocation Loc, 11948 unsigned IVSize, bool IVSigned, 11949 Address IL, Address LB, 11950 Address UB, Address ST) { 11951 llvm_unreachable("Not supported in SIMD-only mode"); 11952 } 11953 11954 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 11955 llvm::Value *NumThreads, 11956 SourceLocation Loc) { 11957 llvm_unreachable("Not supported in SIMD-only mode"); 11958 } 11959 11960 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF, 11961 ProcBindKind ProcBind, 11962 SourceLocation Loc) { 11963 llvm_unreachable("Not supported in SIMD-only mode"); 11964 } 11965 11966 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 11967 const VarDecl *VD, 11968 Address VDAddr, 11969 SourceLocation Loc) { 11970 llvm_unreachable("Not supported in SIMD-only mode"); 11971 } 11972 11973 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition( 11974 const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit, 11975 CodeGenFunction *CGF) { 11976 llvm_unreachable("Not supported in SIMD-only mode"); 11977 } 11978 11979 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate( 11980 CodeGenFunction &CGF, QualType VarType, StringRef Name) { 11981 llvm_unreachable("Not supported in SIMD-only mode"); 11982 } 11983 11984 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF, 11985 ArrayRef<const Expr *> Vars, 11986 SourceLocation Loc, 11987 llvm::AtomicOrdering AO) { 11988 llvm_unreachable("Not supported in SIMD-only mode"); 11989 } 11990 11991 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 11992 const OMPExecutableDirective &D, 11993 llvm::Function *TaskFunction, 11994 QualType SharedsTy, Address Shareds, 11995 const Expr *IfCond, 11996 const OMPTaskDataTy &Data) { 11997 llvm_unreachable("Not supported in SIMD-only mode"); 11998 } 11999 12000 void CGOpenMPSIMDRuntime::emitTaskLoopCall( 12001 CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D, 12002 llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, 12003 const Expr *IfCond, const OMPTaskDataTy &Data) { 12004 llvm_unreachable("Not supported in SIMD-only mode"); 12005 } 12006 12007 void CGOpenMPSIMDRuntime::emitReduction( 12008 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates, 12009 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 12010 ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) { 12011 assert(Options.SimpleReduction && "Only simple reduction is expected."); 12012 CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs, 12013 ReductionOps, Options); 12014 } 12015 12016 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit( 12017 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 12018 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 12019 llvm_unreachable("Not supported in SIMD-only mode"); 12020 } 12021 12022 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 12023 SourceLocation Loc, 12024 ReductionCodeGen &RCG, 12025 unsigned N) { 12026 llvm_unreachable("Not supported in SIMD-only mode"); 12027 } 12028 12029 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF, 12030 SourceLocation Loc, 12031 llvm::Value *ReductionsPtr, 12032 LValue SharedLVal) { 12033 llvm_unreachable("Not supported in SIMD-only mode"); 12034 } 12035 12036 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 12037 SourceLocation Loc) { 12038 llvm_unreachable("Not supported in SIMD-only mode"); 12039 } 12040 12041 void CGOpenMPSIMDRuntime::emitCancellationPointCall( 12042 CodeGenFunction &CGF, SourceLocation Loc, 12043 OpenMPDirectiveKind CancelRegion) { 12044 llvm_unreachable("Not supported in SIMD-only mode"); 12045 } 12046 12047 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF, 12048 SourceLocation Loc, const Expr *IfCond, 12049 OpenMPDirectiveKind CancelRegion) { 12050 llvm_unreachable("Not supported in SIMD-only mode"); 12051 } 12052 12053 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction( 12054 const OMPExecutableDirective &D, StringRef ParentName, 12055 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 12056 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 12057 llvm_unreachable("Not supported in SIMD-only mode"); 12058 } 12059 12060 void CGOpenMPSIMDRuntime::emitTargetCall( 12061 CodeGenFunction &CGF, const OMPExecutableDirective &D, 12062 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 12063 llvm::PointerIntPair<const Expr *, 2, OpenMPDeviceClauseModifier> Device, 12064 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 12065 const OMPLoopDirective &D)> 12066 SizeEmitter) { 12067 llvm_unreachable("Not supported in SIMD-only mode"); 12068 } 12069 12070 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) { 12071 llvm_unreachable("Not supported in SIMD-only mode"); 12072 } 12073 12074 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 12075 llvm_unreachable("Not supported in SIMD-only mode"); 12076 } 12077 12078 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) { 12079 return false; 12080 } 12081 12082 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF, 12083 const OMPExecutableDirective &D, 12084 SourceLocation Loc, 12085 llvm::Function *OutlinedFn, 12086 ArrayRef<llvm::Value *> CapturedVars) { 12087 llvm_unreachable("Not supported in SIMD-only mode"); 12088 } 12089 12090 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 12091 const Expr *NumTeams, 12092 const Expr *ThreadLimit, 12093 SourceLocation Loc) { 12094 llvm_unreachable("Not supported in SIMD-only mode"); 12095 } 12096 12097 void CGOpenMPSIMDRuntime::emitTargetDataCalls( 12098 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 12099 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 12100 llvm_unreachable("Not supported in SIMD-only mode"); 12101 } 12102 12103 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall( 12104 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 12105 const Expr *Device) { 12106 llvm_unreachable("Not supported in SIMD-only mode"); 12107 } 12108 12109 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF, 12110 const OMPLoopDirective &D, 12111 ArrayRef<Expr *> NumIterations) { 12112 llvm_unreachable("Not supported in SIMD-only mode"); 12113 } 12114 12115 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 12116 const OMPDependClause *C) { 12117 llvm_unreachable("Not supported in SIMD-only mode"); 12118 } 12119 12120 const VarDecl * 12121 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD, 12122 const VarDecl *NativeParam) const { 12123 llvm_unreachable("Not supported in SIMD-only mode"); 12124 } 12125 12126 Address 12127 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF, 12128 const VarDecl *NativeParam, 12129 const VarDecl *TargetParam) const { 12130 llvm_unreachable("Not supported in SIMD-only mode"); 12131 } 12132