1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This provides a class for OpenMP runtime code generation. 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "CGCXXABI.h" 14 #include "CGCleanup.h" 15 #include "CGOpenMPRuntime.h" 16 #include "CGRecordLayout.h" 17 #include "CodeGenFunction.h" 18 #include "clang/CodeGen/ConstantInitBuilder.h" 19 #include "clang/AST/Decl.h" 20 #include "clang/AST/StmtOpenMP.h" 21 #include "clang/Basic/BitmaskEnum.h" 22 #include "llvm/ADT/ArrayRef.h" 23 #include "llvm/Bitcode/BitcodeReader.h" 24 #include "llvm/IR/DerivedTypes.h" 25 #include "llvm/IR/GlobalValue.h" 26 #include "llvm/IR/Value.h" 27 #include "llvm/Support/Format.h" 28 #include "llvm/Support/raw_ostream.h" 29 #include <cassert> 30 31 using namespace clang; 32 using namespace CodeGen; 33 34 namespace { 35 /// Base class for handling code generation inside OpenMP regions. 36 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo { 37 public: 38 /// Kinds of OpenMP regions used in codegen. 39 enum CGOpenMPRegionKind { 40 /// Region with outlined function for standalone 'parallel' 41 /// directive. 42 ParallelOutlinedRegion, 43 /// Region with outlined function for standalone 'task' directive. 44 TaskOutlinedRegion, 45 /// Region for constructs that do not require function outlining, 46 /// like 'for', 'sections', 'atomic' etc. directives. 47 InlinedRegion, 48 /// Region with outlined function for standalone 'target' directive. 49 TargetRegion, 50 }; 51 52 CGOpenMPRegionInfo(const CapturedStmt &CS, 53 const CGOpenMPRegionKind RegionKind, 54 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 55 bool HasCancel) 56 : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind), 57 CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {} 58 59 CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind, 60 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 61 bool HasCancel) 62 : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen), 63 Kind(Kind), HasCancel(HasCancel) {} 64 65 /// Get a variable or parameter for storing global thread id 66 /// inside OpenMP construct. 67 virtual const VarDecl *getThreadIDVariable() const = 0; 68 69 /// Emit the captured statement body. 70 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override; 71 72 /// Get an LValue for the current ThreadID variable. 73 /// \return LValue for thread id variable. This LValue always has type int32*. 74 virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF); 75 76 virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {} 77 78 CGOpenMPRegionKind getRegionKind() const { return RegionKind; } 79 80 OpenMPDirectiveKind getDirectiveKind() const { return Kind; } 81 82 bool hasCancel() const { return HasCancel; } 83 84 static bool classof(const CGCapturedStmtInfo *Info) { 85 return Info->getKind() == CR_OpenMP; 86 } 87 88 ~CGOpenMPRegionInfo() override = default; 89 90 protected: 91 CGOpenMPRegionKind RegionKind; 92 RegionCodeGenTy CodeGen; 93 OpenMPDirectiveKind Kind; 94 bool HasCancel; 95 }; 96 97 /// API for captured statement code generation in OpenMP constructs. 98 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo { 99 public: 100 CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar, 101 const RegionCodeGenTy &CodeGen, 102 OpenMPDirectiveKind Kind, bool HasCancel, 103 StringRef HelperName) 104 : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind, 105 HasCancel), 106 ThreadIDVar(ThreadIDVar), HelperName(HelperName) { 107 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 108 } 109 110 /// Get a variable or parameter for storing global thread id 111 /// inside OpenMP construct. 112 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 113 114 /// Get the name of the capture helper. 115 StringRef getHelperName() const override { return HelperName; } 116 117 static bool classof(const CGCapturedStmtInfo *Info) { 118 return CGOpenMPRegionInfo::classof(Info) && 119 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 120 ParallelOutlinedRegion; 121 } 122 123 private: 124 /// A variable or parameter storing global thread id for OpenMP 125 /// constructs. 126 const VarDecl *ThreadIDVar; 127 StringRef HelperName; 128 }; 129 130 /// API for captured statement code generation in OpenMP constructs. 131 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo { 132 public: 133 class UntiedTaskActionTy final : public PrePostActionTy { 134 bool Untied; 135 const VarDecl *PartIDVar; 136 const RegionCodeGenTy UntiedCodeGen; 137 llvm::SwitchInst *UntiedSwitch = nullptr; 138 139 public: 140 UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar, 141 const RegionCodeGenTy &UntiedCodeGen) 142 : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {} 143 void Enter(CodeGenFunction &CGF) override { 144 if (Untied) { 145 // Emit task switching point. 146 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 147 CGF.GetAddrOfLocalVar(PartIDVar), 148 PartIDVar->getType()->castAs<PointerType>()); 149 llvm::Value *Res = 150 CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation()); 151 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done."); 152 UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB); 153 CGF.EmitBlock(DoneBB); 154 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 155 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 156 UntiedSwitch->addCase(CGF.Builder.getInt32(0), 157 CGF.Builder.GetInsertBlock()); 158 emitUntiedSwitch(CGF); 159 } 160 } 161 void emitUntiedSwitch(CodeGenFunction &CGF) const { 162 if (Untied) { 163 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 164 CGF.GetAddrOfLocalVar(PartIDVar), 165 PartIDVar->getType()->castAs<PointerType>()); 166 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 167 PartIdLVal); 168 UntiedCodeGen(CGF); 169 CodeGenFunction::JumpDest CurPoint = 170 CGF.getJumpDestInCurrentScope(".untied.next."); 171 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 172 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 173 UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 174 CGF.Builder.GetInsertBlock()); 175 CGF.EmitBranchThroughCleanup(CurPoint); 176 CGF.EmitBlock(CurPoint.getBlock()); 177 } 178 } 179 unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); } 180 }; 181 CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS, 182 const VarDecl *ThreadIDVar, 183 const RegionCodeGenTy &CodeGen, 184 OpenMPDirectiveKind Kind, bool HasCancel, 185 const UntiedTaskActionTy &Action) 186 : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel), 187 ThreadIDVar(ThreadIDVar), Action(Action) { 188 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 189 } 190 191 /// Get a variable or parameter for storing global thread id 192 /// inside OpenMP construct. 193 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 194 195 /// Get an LValue for the current ThreadID variable. 196 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override; 197 198 /// Get the name of the capture helper. 199 StringRef getHelperName() const override { return ".omp_outlined."; } 200 201 void emitUntiedSwitch(CodeGenFunction &CGF) override { 202 Action.emitUntiedSwitch(CGF); 203 } 204 205 static bool classof(const CGCapturedStmtInfo *Info) { 206 return CGOpenMPRegionInfo::classof(Info) && 207 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 208 TaskOutlinedRegion; 209 } 210 211 private: 212 /// A variable or parameter storing global thread id for OpenMP 213 /// constructs. 214 const VarDecl *ThreadIDVar; 215 /// Action for emitting code for untied tasks. 216 const UntiedTaskActionTy &Action; 217 }; 218 219 /// API for inlined captured statement code generation in OpenMP 220 /// constructs. 221 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo { 222 public: 223 CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI, 224 const RegionCodeGenTy &CodeGen, 225 OpenMPDirectiveKind Kind, bool HasCancel) 226 : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel), 227 OldCSI(OldCSI), 228 OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {} 229 230 // Retrieve the value of the context parameter. 231 llvm::Value *getContextValue() const override { 232 if (OuterRegionInfo) 233 return OuterRegionInfo->getContextValue(); 234 llvm_unreachable("No context value for inlined OpenMP region"); 235 } 236 237 void setContextValue(llvm::Value *V) override { 238 if (OuterRegionInfo) { 239 OuterRegionInfo->setContextValue(V); 240 return; 241 } 242 llvm_unreachable("No context value for inlined OpenMP region"); 243 } 244 245 /// Lookup the captured field decl for a variable. 246 const FieldDecl *lookup(const VarDecl *VD) const override { 247 if (OuterRegionInfo) 248 return OuterRegionInfo->lookup(VD); 249 // If there is no outer outlined region,no need to lookup in a list of 250 // captured variables, we can use the original one. 251 return nullptr; 252 } 253 254 FieldDecl *getThisFieldDecl() const override { 255 if (OuterRegionInfo) 256 return OuterRegionInfo->getThisFieldDecl(); 257 return nullptr; 258 } 259 260 /// Get a variable or parameter for storing global thread id 261 /// inside OpenMP construct. 262 const VarDecl *getThreadIDVariable() const override { 263 if (OuterRegionInfo) 264 return OuterRegionInfo->getThreadIDVariable(); 265 return nullptr; 266 } 267 268 /// Get an LValue for the current ThreadID variable. 269 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override { 270 if (OuterRegionInfo) 271 return OuterRegionInfo->getThreadIDVariableLValue(CGF); 272 llvm_unreachable("No LValue for inlined OpenMP construct"); 273 } 274 275 /// Get the name of the capture helper. 276 StringRef getHelperName() const override { 277 if (auto *OuterRegionInfo = getOldCSI()) 278 return OuterRegionInfo->getHelperName(); 279 llvm_unreachable("No helper name for inlined OpenMP construct"); 280 } 281 282 void emitUntiedSwitch(CodeGenFunction &CGF) override { 283 if (OuterRegionInfo) 284 OuterRegionInfo->emitUntiedSwitch(CGF); 285 } 286 287 CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; } 288 289 static bool classof(const CGCapturedStmtInfo *Info) { 290 return CGOpenMPRegionInfo::classof(Info) && 291 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion; 292 } 293 294 ~CGOpenMPInlinedRegionInfo() override = default; 295 296 private: 297 /// CodeGen info about outer OpenMP region. 298 CodeGenFunction::CGCapturedStmtInfo *OldCSI; 299 CGOpenMPRegionInfo *OuterRegionInfo; 300 }; 301 302 /// API for captured statement code generation in OpenMP target 303 /// constructs. For this captures, implicit parameters are used instead of the 304 /// captured fields. The name of the target region has to be unique in a given 305 /// application so it is provided by the client, because only the client has 306 /// the information to generate that. 307 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo { 308 public: 309 CGOpenMPTargetRegionInfo(const CapturedStmt &CS, 310 const RegionCodeGenTy &CodeGen, StringRef HelperName) 311 : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target, 312 /*HasCancel=*/false), 313 HelperName(HelperName) {} 314 315 /// This is unused for target regions because each starts executing 316 /// with a single thread. 317 const VarDecl *getThreadIDVariable() const override { return nullptr; } 318 319 /// Get the name of the capture helper. 320 StringRef getHelperName() const override { return HelperName; } 321 322 static bool classof(const CGCapturedStmtInfo *Info) { 323 return CGOpenMPRegionInfo::classof(Info) && 324 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion; 325 } 326 327 private: 328 StringRef HelperName; 329 }; 330 331 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) { 332 llvm_unreachable("No codegen for expressions"); 333 } 334 /// API for generation of expressions captured in a innermost OpenMP 335 /// region. 336 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo { 337 public: 338 CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS) 339 : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen, 340 OMPD_unknown, 341 /*HasCancel=*/false), 342 PrivScope(CGF) { 343 // Make sure the globals captured in the provided statement are local by 344 // using the privatization logic. We assume the same variable is not 345 // captured more than once. 346 for (const auto &C : CS.captures()) { 347 if (!C.capturesVariable() && !C.capturesVariableByCopy()) 348 continue; 349 350 const VarDecl *VD = C.getCapturedVar(); 351 if (VD->isLocalVarDeclOrParm()) 352 continue; 353 354 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD), 355 /*RefersToEnclosingVariableOrCapture=*/false, 356 VD->getType().getNonReferenceType(), VK_LValue, 357 C.getLocation()); 358 PrivScope.addPrivate( 359 VD, [&CGF, &DRE]() { return CGF.EmitLValue(&DRE).getAddress(); }); 360 } 361 (void)PrivScope.Privatize(); 362 } 363 364 /// Lookup the captured field decl for a variable. 365 const FieldDecl *lookup(const VarDecl *VD) const override { 366 if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD)) 367 return FD; 368 return nullptr; 369 } 370 371 /// Emit the captured statement body. 372 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override { 373 llvm_unreachable("No body for expressions"); 374 } 375 376 /// Get a variable or parameter for storing global thread id 377 /// inside OpenMP construct. 378 const VarDecl *getThreadIDVariable() const override { 379 llvm_unreachable("No thread id for expressions"); 380 } 381 382 /// Get the name of the capture helper. 383 StringRef getHelperName() const override { 384 llvm_unreachable("No helper name for expressions"); 385 } 386 387 static bool classof(const CGCapturedStmtInfo *Info) { return false; } 388 389 private: 390 /// Private scope to capture global variables. 391 CodeGenFunction::OMPPrivateScope PrivScope; 392 }; 393 394 /// RAII for emitting code of OpenMP constructs. 395 class InlinedOpenMPRegionRAII { 396 CodeGenFunction &CGF; 397 llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields; 398 FieldDecl *LambdaThisCaptureField = nullptr; 399 const CodeGen::CGBlockInfo *BlockInfo = nullptr; 400 401 public: 402 /// Constructs region for combined constructs. 403 /// \param CodeGen Code generation sequence for combined directives. Includes 404 /// a list of functions used for code generation of implicitly inlined 405 /// regions. 406 InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen, 407 OpenMPDirectiveKind Kind, bool HasCancel) 408 : CGF(CGF) { 409 // Start emission for the construct. 410 CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo( 411 CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel); 412 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 413 LambdaThisCaptureField = CGF.LambdaThisCaptureField; 414 CGF.LambdaThisCaptureField = nullptr; 415 BlockInfo = CGF.BlockInfo; 416 CGF.BlockInfo = nullptr; 417 } 418 419 ~InlinedOpenMPRegionRAII() { 420 // Restore original CapturedStmtInfo only if we're done with code emission. 421 auto *OldCSI = 422 cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI(); 423 delete CGF.CapturedStmtInfo; 424 CGF.CapturedStmtInfo = OldCSI; 425 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 426 CGF.LambdaThisCaptureField = LambdaThisCaptureField; 427 CGF.BlockInfo = BlockInfo; 428 } 429 }; 430 431 /// Values for bit flags used in the ident_t to describe the fields. 432 /// All enumeric elements are named and described in accordance with the code 433 /// from https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 434 enum OpenMPLocationFlags : unsigned { 435 /// Use trampoline for internal microtask. 436 OMP_IDENT_IMD = 0x01, 437 /// Use c-style ident structure. 438 OMP_IDENT_KMPC = 0x02, 439 /// Atomic reduction option for kmpc_reduce. 440 OMP_ATOMIC_REDUCE = 0x10, 441 /// Explicit 'barrier' directive. 442 OMP_IDENT_BARRIER_EXPL = 0x20, 443 /// Implicit barrier in code. 444 OMP_IDENT_BARRIER_IMPL = 0x40, 445 /// Implicit barrier in 'for' directive. 446 OMP_IDENT_BARRIER_IMPL_FOR = 0x40, 447 /// Implicit barrier in 'sections' directive. 448 OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0, 449 /// Implicit barrier in 'single' directive. 450 OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140, 451 /// Call of __kmp_for_static_init for static loop. 452 OMP_IDENT_WORK_LOOP = 0x200, 453 /// Call of __kmp_for_static_init for sections. 454 OMP_IDENT_WORK_SECTIONS = 0x400, 455 /// Call of __kmp_for_static_init for distribute. 456 OMP_IDENT_WORK_DISTRIBUTE = 0x800, 457 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE) 458 }; 459 460 namespace { 461 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 462 /// Values for bit flags for marking which requires clauses have been used. 463 enum OpenMPOffloadingRequiresDirFlags : int64_t { 464 /// flag undefined. 465 OMP_REQ_UNDEFINED = 0x000, 466 /// no requires clause present. 467 OMP_REQ_NONE = 0x001, 468 /// reverse_offload clause. 469 OMP_REQ_REVERSE_OFFLOAD = 0x002, 470 /// unified_address clause. 471 OMP_REQ_UNIFIED_ADDRESS = 0x004, 472 /// unified_shared_memory clause. 473 OMP_REQ_UNIFIED_SHARED_MEMORY = 0x008, 474 /// dynamic_allocators clause. 475 OMP_REQ_DYNAMIC_ALLOCATORS = 0x010, 476 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_REQ_DYNAMIC_ALLOCATORS) 477 }; 478 479 enum OpenMPOffloadingReservedDeviceIDs { 480 /// Device ID if the device was not defined, runtime should get it 481 /// from environment variables in the spec. 482 OMP_DEVICEID_UNDEF = -1, 483 }; 484 } // anonymous namespace 485 486 /// Describes ident structure that describes a source location. 487 /// All descriptions are taken from 488 /// https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 489 /// Original structure: 490 /// typedef struct ident { 491 /// kmp_int32 reserved_1; /**< might be used in Fortran; 492 /// see above */ 493 /// kmp_int32 flags; /**< also f.flags; KMP_IDENT_xxx flags; 494 /// KMP_IDENT_KMPC identifies this union 495 /// member */ 496 /// kmp_int32 reserved_2; /**< not really used in Fortran any more; 497 /// see above */ 498 ///#if USE_ITT_BUILD 499 /// /* but currently used for storing 500 /// region-specific ITT */ 501 /// /* contextual information. */ 502 ///#endif /* USE_ITT_BUILD */ 503 /// kmp_int32 reserved_3; /**< source[4] in Fortran, do not use for 504 /// C++ */ 505 /// char const *psource; /**< String describing the source location. 506 /// The string is composed of semi-colon separated 507 // fields which describe the source file, 508 /// the function and a pair of line numbers that 509 /// delimit the construct. 510 /// */ 511 /// } ident_t; 512 enum IdentFieldIndex { 513 /// might be used in Fortran 514 IdentField_Reserved_1, 515 /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member. 516 IdentField_Flags, 517 /// Not really used in Fortran any more 518 IdentField_Reserved_2, 519 /// Source[4] in Fortran, do not use for C++ 520 IdentField_Reserved_3, 521 /// String describing the source location. The string is composed of 522 /// semi-colon separated fields which describe the source file, the function 523 /// and a pair of line numbers that delimit the construct. 524 IdentField_PSource 525 }; 526 527 /// Schedule types for 'omp for' loops (these enumerators are taken from 528 /// the enum sched_type in kmp.h). 529 enum OpenMPSchedType { 530 /// Lower bound for default (unordered) versions. 531 OMP_sch_lower = 32, 532 OMP_sch_static_chunked = 33, 533 OMP_sch_static = 34, 534 OMP_sch_dynamic_chunked = 35, 535 OMP_sch_guided_chunked = 36, 536 OMP_sch_runtime = 37, 537 OMP_sch_auto = 38, 538 /// static with chunk adjustment (e.g., simd) 539 OMP_sch_static_balanced_chunked = 45, 540 /// Lower bound for 'ordered' versions. 541 OMP_ord_lower = 64, 542 OMP_ord_static_chunked = 65, 543 OMP_ord_static = 66, 544 OMP_ord_dynamic_chunked = 67, 545 OMP_ord_guided_chunked = 68, 546 OMP_ord_runtime = 69, 547 OMP_ord_auto = 70, 548 OMP_sch_default = OMP_sch_static, 549 /// dist_schedule types 550 OMP_dist_sch_static_chunked = 91, 551 OMP_dist_sch_static = 92, 552 /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers. 553 /// Set if the monotonic schedule modifier was present. 554 OMP_sch_modifier_monotonic = (1 << 29), 555 /// Set if the nonmonotonic schedule modifier was present. 556 OMP_sch_modifier_nonmonotonic = (1 << 30), 557 }; 558 559 enum OpenMPRTLFunction { 560 /// Call to void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, 561 /// kmpc_micro microtask, ...); 562 OMPRTL__kmpc_fork_call, 563 /// Call to void *__kmpc_threadprivate_cached(ident_t *loc, 564 /// kmp_int32 global_tid, void *data, size_t size, void ***cache); 565 OMPRTL__kmpc_threadprivate_cached, 566 /// Call to void __kmpc_threadprivate_register( ident_t *, 567 /// void *data, kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor); 568 OMPRTL__kmpc_threadprivate_register, 569 // Call to __kmpc_int32 kmpc_global_thread_num(ident_t *loc); 570 OMPRTL__kmpc_global_thread_num, 571 // Call to void __kmpc_critical(ident_t *loc, kmp_int32 global_tid, 572 // kmp_critical_name *crit); 573 OMPRTL__kmpc_critical, 574 // Call to void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 575 // global_tid, kmp_critical_name *crit, uintptr_t hint); 576 OMPRTL__kmpc_critical_with_hint, 577 // Call to void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid, 578 // kmp_critical_name *crit); 579 OMPRTL__kmpc_end_critical, 580 // Call to kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32 581 // global_tid); 582 OMPRTL__kmpc_cancel_barrier, 583 // Call to void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid); 584 OMPRTL__kmpc_barrier, 585 // Call to void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid); 586 OMPRTL__kmpc_for_static_fini, 587 // Call to void __kmpc_serialized_parallel(ident_t *loc, kmp_int32 588 // global_tid); 589 OMPRTL__kmpc_serialized_parallel, 590 // Call to void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32 591 // global_tid); 592 OMPRTL__kmpc_end_serialized_parallel, 593 // Call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid, 594 // kmp_int32 num_threads); 595 OMPRTL__kmpc_push_num_threads, 596 // Call to void __kmpc_flush(ident_t *loc); 597 OMPRTL__kmpc_flush, 598 // Call to kmp_int32 __kmpc_master(ident_t *, kmp_int32 global_tid); 599 OMPRTL__kmpc_master, 600 // Call to void __kmpc_end_master(ident_t *, kmp_int32 global_tid); 601 OMPRTL__kmpc_end_master, 602 // Call to kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid, 603 // int end_part); 604 OMPRTL__kmpc_omp_taskyield, 605 // Call to kmp_int32 __kmpc_single(ident_t *, kmp_int32 global_tid); 606 OMPRTL__kmpc_single, 607 // Call to void __kmpc_end_single(ident_t *, kmp_int32 global_tid); 608 OMPRTL__kmpc_end_single, 609 // Call to kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 610 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 611 // kmp_routine_entry_t *task_entry); 612 OMPRTL__kmpc_omp_task_alloc, 613 // Call to kmp_task_t * __kmpc_omp_target_task_alloc(ident_t *, 614 // kmp_int32 gtid, kmp_int32 flags, size_t sizeof_kmp_task_t, 615 // size_t sizeof_shareds, kmp_routine_entry_t *task_entry, 616 // kmp_int64 device_id); 617 OMPRTL__kmpc_omp_target_task_alloc, 618 // Call to kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t * 619 // new_task); 620 OMPRTL__kmpc_omp_task, 621 // Call to void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid, 622 // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *), 623 // kmp_int32 didit); 624 OMPRTL__kmpc_copyprivate, 625 // Call to kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid, 626 // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void 627 // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck); 628 OMPRTL__kmpc_reduce, 629 // Call to kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32 630 // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, 631 // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name 632 // *lck); 633 OMPRTL__kmpc_reduce_nowait, 634 // Call to void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid, 635 // kmp_critical_name *lck); 636 OMPRTL__kmpc_end_reduce, 637 // Call to void __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid, 638 // kmp_critical_name *lck); 639 OMPRTL__kmpc_end_reduce_nowait, 640 // Call to void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 641 // kmp_task_t * new_task); 642 OMPRTL__kmpc_omp_task_begin_if0, 643 // Call to void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 644 // kmp_task_t * new_task); 645 OMPRTL__kmpc_omp_task_complete_if0, 646 // Call to void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid); 647 OMPRTL__kmpc_ordered, 648 // Call to void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid); 649 OMPRTL__kmpc_end_ordered, 650 // Call to kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 651 // global_tid); 652 OMPRTL__kmpc_omp_taskwait, 653 // Call to void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid); 654 OMPRTL__kmpc_taskgroup, 655 // Call to void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid); 656 OMPRTL__kmpc_end_taskgroup, 657 // Call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid, 658 // int proc_bind); 659 OMPRTL__kmpc_push_proc_bind, 660 // Call to kmp_int32 __kmpc_omp_task_with_deps(ident_t *loc_ref, kmp_int32 661 // gtid, kmp_task_t * new_task, kmp_int32 ndeps, kmp_depend_info_t 662 // *dep_list, kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 663 OMPRTL__kmpc_omp_task_with_deps, 664 // Call to void __kmpc_omp_wait_deps(ident_t *loc_ref, kmp_int32 665 // gtid, kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 666 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 667 OMPRTL__kmpc_omp_wait_deps, 668 // Call to kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 669 // global_tid, kmp_int32 cncl_kind); 670 OMPRTL__kmpc_cancellationpoint, 671 // Call to kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 672 // kmp_int32 cncl_kind); 673 OMPRTL__kmpc_cancel, 674 // Call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32 global_tid, 675 // kmp_int32 num_teams, kmp_int32 thread_limit); 676 OMPRTL__kmpc_push_num_teams, 677 // Call to void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro 678 // microtask, ...); 679 OMPRTL__kmpc_fork_teams, 680 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 681 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 682 // sched, kmp_uint64 grainsize, void *task_dup); 683 OMPRTL__kmpc_taskloop, 684 // Call to void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32 685 // num_dims, struct kmp_dim *dims); 686 OMPRTL__kmpc_doacross_init, 687 // Call to void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid); 688 OMPRTL__kmpc_doacross_fini, 689 // Call to void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64 690 // *vec); 691 OMPRTL__kmpc_doacross_post, 692 // Call to void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64 693 // *vec); 694 OMPRTL__kmpc_doacross_wait, 695 // Call to void *__kmpc_task_reduction_init(int gtid, int num_data, void 696 // *data); 697 OMPRTL__kmpc_task_reduction_init, 698 // Call to void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 699 // *d); 700 OMPRTL__kmpc_task_reduction_get_th_data, 701 // Call to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t al); 702 OMPRTL__kmpc_alloc, 703 // Call to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t al); 704 OMPRTL__kmpc_free, 705 706 // 707 // Offloading related calls 708 // 709 // Call to void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64 710 // size); 711 OMPRTL__kmpc_push_target_tripcount, 712 // Call to int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t 713 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 714 // *arg_types); 715 OMPRTL__tgt_target, 716 // Call to int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr, 717 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 718 // *arg_types); 719 OMPRTL__tgt_target_nowait, 720 // Call to int32_t __tgt_target_teams(int64_t device_id, void *host_ptr, 721 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 722 // *arg_types, int32_t num_teams, int32_t thread_limit); 723 OMPRTL__tgt_target_teams, 724 // Call to int32_t __tgt_target_teams_nowait(int64_t device_id, void 725 // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t 726 // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 727 OMPRTL__tgt_target_teams_nowait, 728 // Call to void __tgt_register_requires(int64_t flags); 729 OMPRTL__tgt_register_requires, 730 // Call to void __tgt_register_lib(__tgt_bin_desc *desc); 731 OMPRTL__tgt_register_lib, 732 // Call to void __tgt_unregister_lib(__tgt_bin_desc *desc); 733 OMPRTL__tgt_unregister_lib, 734 // Call to void __tgt_target_data_begin(int64_t device_id, int32_t arg_num, 735 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 736 OMPRTL__tgt_target_data_begin, 737 // Call to void __tgt_target_data_begin_nowait(int64_t device_id, int32_t 738 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 739 // *arg_types); 740 OMPRTL__tgt_target_data_begin_nowait, 741 // Call to void __tgt_target_data_end(int64_t device_id, int32_t arg_num, 742 // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types); 743 OMPRTL__tgt_target_data_end, 744 // Call to void __tgt_target_data_end_nowait(int64_t device_id, int32_t 745 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 746 // *arg_types); 747 OMPRTL__tgt_target_data_end_nowait, 748 // Call to void __tgt_target_data_update(int64_t device_id, int32_t arg_num, 749 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 750 OMPRTL__tgt_target_data_update, 751 // Call to void __tgt_target_data_update_nowait(int64_t device_id, int32_t 752 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 753 // *arg_types); 754 OMPRTL__tgt_target_data_update_nowait, 755 // Call to int64_t __tgt_mapper_num_components(void *rt_mapper_handle); 756 OMPRTL__tgt_mapper_num_components, 757 // Call to void __tgt_push_mapper_component(void *rt_mapper_handle, void 758 // *base, void *begin, int64_t size, int64_t type); 759 OMPRTL__tgt_push_mapper_component, 760 }; 761 762 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP 763 /// region. 764 class CleanupTy final : public EHScopeStack::Cleanup { 765 PrePostActionTy *Action; 766 767 public: 768 explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {} 769 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 770 if (!CGF.HaveInsertPoint()) 771 return; 772 Action->Exit(CGF); 773 } 774 }; 775 776 } // anonymous namespace 777 778 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const { 779 CodeGenFunction::RunCleanupsScope Scope(CGF); 780 if (PrePostAction) { 781 CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction); 782 Callback(CodeGen, CGF, *PrePostAction); 783 } else { 784 PrePostActionTy Action; 785 Callback(CodeGen, CGF, Action); 786 } 787 } 788 789 /// Check if the combiner is a call to UDR combiner and if it is so return the 790 /// UDR decl used for reduction. 791 static const OMPDeclareReductionDecl * 792 getReductionInit(const Expr *ReductionOp) { 793 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 794 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 795 if (const auto *DRE = 796 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 797 if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) 798 return DRD; 799 return nullptr; 800 } 801 802 static void emitInitWithReductionInitializer(CodeGenFunction &CGF, 803 const OMPDeclareReductionDecl *DRD, 804 const Expr *InitOp, 805 Address Private, Address Original, 806 QualType Ty) { 807 if (DRD->getInitializer()) { 808 std::pair<llvm::Function *, llvm::Function *> Reduction = 809 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 810 const auto *CE = cast<CallExpr>(InitOp); 811 const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee()); 812 const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts(); 813 const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts(); 814 const auto *LHSDRE = 815 cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr()); 816 const auto *RHSDRE = 817 cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr()); 818 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 819 PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()), 820 [=]() { return Private; }); 821 PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()), 822 [=]() { return Original; }); 823 (void)PrivateScope.Privatize(); 824 RValue Func = RValue::get(Reduction.second); 825 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 826 CGF.EmitIgnoredExpr(InitOp); 827 } else { 828 llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty); 829 std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"}); 830 auto *GV = new llvm::GlobalVariable( 831 CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true, 832 llvm::GlobalValue::PrivateLinkage, Init, Name); 833 LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty); 834 RValue InitRVal; 835 switch (CGF.getEvaluationKind(Ty)) { 836 case TEK_Scalar: 837 InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation()); 838 break; 839 case TEK_Complex: 840 InitRVal = 841 RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation())); 842 break; 843 case TEK_Aggregate: 844 InitRVal = RValue::getAggregate(LV.getAddress()); 845 break; 846 } 847 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_RValue); 848 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal); 849 CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(), 850 /*IsInitializer=*/false); 851 } 852 } 853 854 /// Emit initialization of arrays of complex types. 855 /// \param DestAddr Address of the array. 856 /// \param Type Type of array. 857 /// \param Init Initial expression of array. 858 /// \param SrcAddr Address of the original array. 859 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr, 860 QualType Type, bool EmitDeclareReductionInit, 861 const Expr *Init, 862 const OMPDeclareReductionDecl *DRD, 863 Address SrcAddr = Address::invalid()) { 864 // Perform element-by-element initialization. 865 QualType ElementTy; 866 867 // Drill down to the base element type on both arrays. 868 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 869 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr); 870 DestAddr = 871 CGF.Builder.CreateElementBitCast(DestAddr, DestAddr.getElementType()); 872 if (DRD) 873 SrcAddr = 874 CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType()); 875 876 llvm::Value *SrcBegin = nullptr; 877 if (DRD) 878 SrcBegin = SrcAddr.getPointer(); 879 llvm::Value *DestBegin = DestAddr.getPointer(); 880 // Cast from pointer to array type to pointer to single element. 881 llvm::Value *DestEnd = CGF.Builder.CreateGEP(DestBegin, NumElements); 882 // The basic structure here is a while-do loop. 883 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body"); 884 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done"); 885 llvm::Value *IsEmpty = 886 CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty"); 887 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 888 889 // Enter the loop body, making that address the current address. 890 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 891 CGF.EmitBlock(BodyBB); 892 893 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 894 895 llvm::PHINode *SrcElementPHI = nullptr; 896 Address SrcElementCurrent = Address::invalid(); 897 if (DRD) { 898 SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2, 899 "omp.arraycpy.srcElementPast"); 900 SrcElementPHI->addIncoming(SrcBegin, EntryBB); 901 SrcElementCurrent = 902 Address(SrcElementPHI, 903 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 904 } 905 llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI( 906 DestBegin->getType(), 2, "omp.arraycpy.destElementPast"); 907 DestElementPHI->addIncoming(DestBegin, EntryBB); 908 Address DestElementCurrent = 909 Address(DestElementPHI, 910 DestAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 911 912 // Emit copy. 913 { 914 CodeGenFunction::RunCleanupsScope InitScope(CGF); 915 if (EmitDeclareReductionInit) { 916 emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent, 917 SrcElementCurrent, ElementTy); 918 } else 919 CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(), 920 /*IsInitializer=*/false); 921 } 922 923 if (DRD) { 924 // Shift the address forward by one element. 925 llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32( 926 SrcElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 927 SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock()); 928 } 929 930 // Shift the address forward by one element. 931 llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32( 932 DestElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 933 // Check whether we've reached the end. 934 llvm::Value *Done = 935 CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done"); 936 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 937 DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock()); 938 939 // Done. 940 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 941 } 942 943 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) { 944 return CGF.EmitOMPSharedLValue(E); 945 } 946 947 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF, 948 const Expr *E) { 949 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E)) 950 return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false); 951 return LValue(); 952 } 953 954 void ReductionCodeGen::emitAggregateInitialization( 955 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 956 const OMPDeclareReductionDecl *DRD) { 957 // Emit VarDecl with copy init for arrays. 958 // Get the address of the original variable captured in current 959 // captured region. 960 const auto *PrivateVD = 961 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 962 bool EmitDeclareReductionInit = 963 DRD && (DRD->getInitializer() || !PrivateVD->hasInit()); 964 EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(), 965 EmitDeclareReductionInit, 966 EmitDeclareReductionInit ? ClausesData[N].ReductionOp 967 : PrivateVD->getInit(), 968 DRD, SharedLVal.getAddress()); 969 } 970 971 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds, 972 ArrayRef<const Expr *> Privates, 973 ArrayRef<const Expr *> ReductionOps) { 974 ClausesData.reserve(Shareds.size()); 975 SharedAddresses.reserve(Shareds.size()); 976 Sizes.reserve(Shareds.size()); 977 BaseDecls.reserve(Shareds.size()); 978 auto IPriv = Privates.begin(); 979 auto IRed = ReductionOps.begin(); 980 for (const Expr *Ref : Shareds) { 981 ClausesData.emplace_back(Ref, *IPriv, *IRed); 982 std::advance(IPriv, 1); 983 std::advance(IRed, 1); 984 } 985 } 986 987 void ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, unsigned N) { 988 assert(SharedAddresses.size() == N && 989 "Number of generated lvalues must be exactly N."); 990 LValue First = emitSharedLValue(CGF, ClausesData[N].Ref); 991 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref); 992 SharedAddresses.emplace_back(First, Second); 993 } 994 995 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) { 996 const auto *PrivateVD = 997 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 998 QualType PrivateType = PrivateVD->getType(); 999 bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref); 1000 if (!PrivateType->isVariablyModifiedType()) { 1001 Sizes.emplace_back( 1002 CGF.getTypeSize( 1003 SharedAddresses[N].first.getType().getNonReferenceType()), 1004 nullptr); 1005 return; 1006 } 1007 llvm::Value *Size; 1008 llvm::Value *SizeInChars; 1009 auto *ElemType = 1010 cast<llvm::PointerType>(SharedAddresses[N].first.getPointer()->getType()) 1011 ->getElementType(); 1012 auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType); 1013 if (AsArraySection) { 1014 Size = CGF.Builder.CreatePtrDiff(SharedAddresses[N].second.getPointer(), 1015 SharedAddresses[N].first.getPointer()); 1016 Size = CGF.Builder.CreateNUWAdd( 1017 Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1)); 1018 SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf); 1019 } else { 1020 SizeInChars = CGF.getTypeSize( 1021 SharedAddresses[N].first.getType().getNonReferenceType()); 1022 Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf); 1023 } 1024 Sizes.emplace_back(SizeInChars, Size); 1025 CodeGenFunction::OpaqueValueMapping OpaqueMap( 1026 CGF, 1027 cast<OpaqueValueExpr>( 1028 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 1029 RValue::get(Size)); 1030 CGF.EmitVariablyModifiedType(PrivateType); 1031 } 1032 1033 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N, 1034 llvm::Value *Size) { 1035 const auto *PrivateVD = 1036 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1037 QualType PrivateType = PrivateVD->getType(); 1038 if (!PrivateType->isVariablyModifiedType()) { 1039 assert(!Size && !Sizes[N].second && 1040 "Size should be nullptr for non-variably modified reduction " 1041 "items."); 1042 return; 1043 } 1044 CodeGenFunction::OpaqueValueMapping OpaqueMap( 1045 CGF, 1046 cast<OpaqueValueExpr>( 1047 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 1048 RValue::get(Size)); 1049 CGF.EmitVariablyModifiedType(PrivateType); 1050 } 1051 1052 void ReductionCodeGen::emitInitialization( 1053 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 1054 llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) { 1055 assert(SharedAddresses.size() > N && "No variable was generated"); 1056 const auto *PrivateVD = 1057 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1058 const OMPDeclareReductionDecl *DRD = 1059 getReductionInit(ClausesData[N].ReductionOp); 1060 QualType PrivateType = PrivateVD->getType(); 1061 PrivateAddr = CGF.Builder.CreateElementBitCast( 1062 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 1063 QualType SharedType = SharedAddresses[N].first.getType(); 1064 SharedLVal = CGF.MakeAddrLValue( 1065 CGF.Builder.CreateElementBitCast(SharedLVal.getAddress(), 1066 CGF.ConvertTypeForMem(SharedType)), 1067 SharedType, SharedAddresses[N].first.getBaseInfo(), 1068 CGF.CGM.getTBAAInfoForSubobject(SharedAddresses[N].first, SharedType)); 1069 if (CGF.getContext().getAsArrayType(PrivateVD->getType())) { 1070 emitAggregateInitialization(CGF, N, PrivateAddr, SharedLVal, DRD); 1071 } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) { 1072 emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp, 1073 PrivateAddr, SharedLVal.getAddress(), 1074 SharedLVal.getType()); 1075 } else if (!DefaultInit(CGF) && PrivateVD->hasInit() && 1076 !CGF.isTrivialInitializer(PrivateVD->getInit())) { 1077 CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr, 1078 PrivateVD->getType().getQualifiers(), 1079 /*IsInitializer=*/false); 1080 } 1081 } 1082 1083 bool ReductionCodeGen::needCleanups(unsigned N) { 1084 const auto *PrivateVD = 1085 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1086 QualType PrivateType = PrivateVD->getType(); 1087 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 1088 return DTorKind != QualType::DK_none; 1089 } 1090 1091 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N, 1092 Address PrivateAddr) { 1093 const auto *PrivateVD = 1094 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1095 QualType PrivateType = PrivateVD->getType(); 1096 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 1097 if (needCleanups(N)) { 1098 PrivateAddr = CGF.Builder.CreateElementBitCast( 1099 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 1100 CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType); 1101 } 1102 } 1103 1104 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 1105 LValue BaseLV) { 1106 BaseTy = BaseTy.getNonReferenceType(); 1107 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 1108 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 1109 if (const auto *PtrTy = BaseTy->getAs<PointerType>()) { 1110 BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(), PtrTy); 1111 } else { 1112 LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(), BaseTy); 1113 BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal); 1114 } 1115 BaseTy = BaseTy->getPointeeType(); 1116 } 1117 return CGF.MakeAddrLValue( 1118 CGF.Builder.CreateElementBitCast(BaseLV.getAddress(), 1119 CGF.ConvertTypeForMem(ElTy)), 1120 BaseLV.getType(), BaseLV.getBaseInfo(), 1121 CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType())); 1122 } 1123 1124 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 1125 llvm::Type *BaseLVType, CharUnits BaseLVAlignment, 1126 llvm::Value *Addr) { 1127 Address Tmp = Address::invalid(); 1128 Address TopTmp = Address::invalid(); 1129 Address MostTopTmp = Address::invalid(); 1130 BaseTy = BaseTy.getNonReferenceType(); 1131 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 1132 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 1133 Tmp = CGF.CreateMemTemp(BaseTy); 1134 if (TopTmp.isValid()) 1135 CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp); 1136 else 1137 MostTopTmp = Tmp; 1138 TopTmp = Tmp; 1139 BaseTy = BaseTy->getPointeeType(); 1140 } 1141 llvm::Type *Ty = BaseLVType; 1142 if (Tmp.isValid()) 1143 Ty = Tmp.getElementType(); 1144 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty); 1145 if (Tmp.isValid()) { 1146 CGF.Builder.CreateStore(Addr, Tmp); 1147 return MostTopTmp; 1148 } 1149 return Address(Addr, BaseLVAlignment); 1150 } 1151 1152 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) { 1153 const VarDecl *OrigVD = nullptr; 1154 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) { 1155 const Expr *Base = OASE->getBase()->IgnoreParenImpCasts(); 1156 while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base)) 1157 Base = TempOASE->getBase()->IgnoreParenImpCasts(); 1158 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 1159 Base = TempASE->getBase()->IgnoreParenImpCasts(); 1160 DE = cast<DeclRefExpr>(Base); 1161 OrigVD = cast<VarDecl>(DE->getDecl()); 1162 } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) { 1163 const Expr *Base = ASE->getBase()->IgnoreParenImpCasts(); 1164 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 1165 Base = TempASE->getBase()->IgnoreParenImpCasts(); 1166 DE = cast<DeclRefExpr>(Base); 1167 OrigVD = cast<VarDecl>(DE->getDecl()); 1168 } 1169 return OrigVD; 1170 } 1171 1172 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N, 1173 Address PrivateAddr) { 1174 const DeclRefExpr *DE; 1175 if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) { 1176 BaseDecls.emplace_back(OrigVD); 1177 LValue OriginalBaseLValue = CGF.EmitLValue(DE); 1178 LValue BaseLValue = 1179 loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(), 1180 OriginalBaseLValue); 1181 llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff( 1182 BaseLValue.getPointer(), SharedAddresses[N].first.getPointer()); 1183 llvm::Value *PrivatePointer = 1184 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 1185 PrivateAddr.getPointer(), 1186 SharedAddresses[N].first.getAddress().getType()); 1187 llvm::Value *Ptr = CGF.Builder.CreateGEP(PrivatePointer, Adjustment); 1188 return castToBase(CGF, OrigVD->getType(), 1189 SharedAddresses[N].first.getType(), 1190 OriginalBaseLValue.getAddress().getType(), 1191 OriginalBaseLValue.getAlignment(), Ptr); 1192 } 1193 BaseDecls.emplace_back( 1194 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl())); 1195 return PrivateAddr; 1196 } 1197 1198 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const { 1199 const OMPDeclareReductionDecl *DRD = 1200 getReductionInit(ClausesData[N].ReductionOp); 1201 return DRD && DRD->getInitializer(); 1202 } 1203 1204 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) { 1205 return CGF.EmitLoadOfPointerLValue( 1206 CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1207 getThreadIDVariable()->getType()->castAs<PointerType>()); 1208 } 1209 1210 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) { 1211 if (!CGF.HaveInsertPoint()) 1212 return; 1213 // 1.2.2 OpenMP Language Terminology 1214 // Structured block - An executable statement with a single entry at the 1215 // top and a single exit at the bottom. 1216 // The point of exit cannot be a branch out of the structured block. 1217 // longjmp() and throw() must not violate the entry/exit criteria. 1218 CGF.EHStack.pushTerminate(); 1219 CodeGen(CGF); 1220 CGF.EHStack.popTerminate(); 1221 } 1222 1223 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue( 1224 CodeGenFunction &CGF) { 1225 return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1226 getThreadIDVariable()->getType(), 1227 AlignmentSource::Decl); 1228 } 1229 1230 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC, 1231 QualType FieldTy) { 1232 auto *Field = FieldDecl::Create( 1233 C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy, 1234 C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()), 1235 /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit); 1236 Field->setAccess(AS_public); 1237 DC->addDecl(Field); 1238 return Field; 1239 } 1240 1241 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator, 1242 StringRef Separator) 1243 : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator), 1244 OffloadEntriesInfoManager(CGM) { 1245 ASTContext &C = CGM.getContext(); 1246 RecordDecl *RD = C.buildImplicitRecord("ident_t"); 1247 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 1248 RD->startDefinition(); 1249 // reserved_1 1250 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1251 // flags 1252 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1253 // reserved_2 1254 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1255 // reserved_3 1256 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1257 // psource 1258 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 1259 RD->completeDefinition(); 1260 IdentQTy = C.getRecordType(RD); 1261 IdentTy = CGM.getTypes().ConvertRecordDeclType(RD); 1262 KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8); 1263 1264 loadOffloadInfoMetadata(); 1265 } 1266 1267 bool CGOpenMPRuntime::tryEmitDeclareVariant(const GlobalDecl &NewGD, 1268 const GlobalDecl &OldGD, 1269 llvm::GlobalValue *OrigAddr, 1270 bool IsForDefinition) { 1271 // Emit at least a definition for the aliasee if the the address of the 1272 // original function is requested. 1273 if (IsForDefinition || OrigAddr) 1274 (void)CGM.GetAddrOfGlobal(NewGD); 1275 StringRef NewMangledName = CGM.getMangledName(NewGD); 1276 llvm::GlobalValue *Addr = CGM.GetGlobalValue(NewMangledName); 1277 if (Addr && !Addr->isDeclaration()) { 1278 const auto *D = cast<FunctionDecl>(OldGD.getDecl()); 1279 const CGFunctionInfo &FI = CGM.getTypes().arrangeGlobalDeclaration(OldGD); 1280 llvm::Type *DeclTy = CGM.getTypes().GetFunctionType(FI); 1281 1282 // Create a reference to the named value. This ensures that it is emitted 1283 // if a deferred decl. 1284 llvm::GlobalValue::LinkageTypes LT = CGM.getFunctionLinkage(OldGD); 1285 1286 // Create the new alias itself, but don't set a name yet. 1287 auto *GA = 1288 llvm::GlobalAlias::create(DeclTy, 0, LT, "", Addr, &CGM.getModule()); 1289 1290 if (OrigAddr) { 1291 assert(OrigAddr->isDeclaration() && "Expected declaration"); 1292 1293 GA->takeName(OrigAddr); 1294 OrigAddr->replaceAllUsesWith( 1295 llvm::ConstantExpr::getBitCast(GA, OrigAddr->getType())); 1296 OrigAddr->eraseFromParent(); 1297 } else { 1298 GA->setName(CGM.getMangledName(OldGD)); 1299 } 1300 1301 // Set attributes which are particular to an alias; this is a 1302 // specialization of the attributes which may be set on a global function. 1303 if (D->hasAttr<WeakAttr>() || D->hasAttr<WeakRefAttr>() || 1304 D->isWeakImported()) 1305 GA->setLinkage(llvm::Function::WeakAnyLinkage); 1306 1307 CGM.SetCommonAttributes(OldGD, GA); 1308 return true; 1309 } 1310 return false; 1311 } 1312 1313 void CGOpenMPRuntime::clear() { 1314 InternalVars.clear(); 1315 // Clean non-target variable declarations possibly used only in debug info. 1316 for (const auto &Data : EmittedNonTargetVariables) { 1317 if (!Data.getValue().pointsToAliveValue()) 1318 continue; 1319 auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue()); 1320 if (!GV) 1321 continue; 1322 if (!GV->isDeclaration() || GV->getNumUses() > 0) 1323 continue; 1324 GV->eraseFromParent(); 1325 } 1326 // Emit aliases for the deferred aliasees. 1327 for (const auto &Pair : DeferredVariantFunction) { 1328 StringRef MangledName = CGM.getMangledName(Pair.second.second); 1329 llvm::GlobalValue *Addr = CGM.GetGlobalValue(MangledName); 1330 // If not able to emit alias, just emit original declaration. 1331 (void)tryEmitDeclareVariant(Pair.second.first, Pair.second.second, Addr, 1332 /*IsForDefinition=*/false); 1333 } 1334 } 1335 1336 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const { 1337 SmallString<128> Buffer; 1338 llvm::raw_svector_ostream OS(Buffer); 1339 StringRef Sep = FirstSeparator; 1340 for (StringRef Part : Parts) { 1341 OS << Sep << Part; 1342 Sep = Separator; 1343 } 1344 return OS.str(); 1345 } 1346 1347 static llvm::Function * 1348 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty, 1349 const Expr *CombinerInitializer, const VarDecl *In, 1350 const VarDecl *Out, bool IsCombiner) { 1351 // void .omp_combiner.(Ty *in, Ty *out); 1352 ASTContext &C = CGM.getContext(); 1353 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 1354 FunctionArgList Args; 1355 ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(), 1356 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1357 ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(), 1358 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1359 Args.push_back(&OmpOutParm); 1360 Args.push_back(&OmpInParm); 1361 const CGFunctionInfo &FnInfo = 1362 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 1363 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 1364 std::string Name = CGM.getOpenMPRuntime().getName( 1365 {IsCombiner ? "omp_combiner" : "omp_initializer", ""}); 1366 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 1367 Name, &CGM.getModule()); 1368 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 1369 if (CGM.getLangOpts().Optimize) { 1370 Fn->removeFnAttr(llvm::Attribute::NoInline); 1371 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 1372 Fn->addFnAttr(llvm::Attribute::AlwaysInline); 1373 } 1374 CodeGenFunction CGF(CGM); 1375 // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions. 1376 // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions. 1377 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(), 1378 Out->getLocation()); 1379 CodeGenFunction::OMPPrivateScope Scope(CGF); 1380 Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm); 1381 Scope.addPrivate(In, [&CGF, AddrIn, PtrTy]() { 1382 return CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>()) 1383 .getAddress(); 1384 }); 1385 Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm); 1386 Scope.addPrivate(Out, [&CGF, AddrOut, PtrTy]() { 1387 return CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>()) 1388 .getAddress(); 1389 }); 1390 (void)Scope.Privatize(); 1391 if (!IsCombiner && Out->hasInit() && 1392 !CGF.isTrivialInitializer(Out->getInit())) { 1393 CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out), 1394 Out->getType().getQualifiers(), 1395 /*IsInitializer=*/true); 1396 } 1397 if (CombinerInitializer) 1398 CGF.EmitIgnoredExpr(CombinerInitializer); 1399 Scope.ForceCleanup(); 1400 CGF.FinishFunction(); 1401 return Fn; 1402 } 1403 1404 void CGOpenMPRuntime::emitUserDefinedReduction( 1405 CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) { 1406 if (UDRMap.count(D) > 0) 1407 return; 1408 llvm::Function *Combiner = emitCombinerOrInitializer( 1409 CGM, D->getType(), D->getCombiner(), 1410 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()), 1411 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()), 1412 /*IsCombiner=*/true); 1413 llvm::Function *Initializer = nullptr; 1414 if (const Expr *Init = D->getInitializer()) { 1415 Initializer = emitCombinerOrInitializer( 1416 CGM, D->getType(), 1417 D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init 1418 : nullptr, 1419 cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()), 1420 cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()), 1421 /*IsCombiner=*/false); 1422 } 1423 UDRMap.try_emplace(D, Combiner, Initializer); 1424 if (CGF) { 1425 auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn); 1426 Decls.second.push_back(D); 1427 } 1428 } 1429 1430 std::pair<llvm::Function *, llvm::Function *> 1431 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) { 1432 auto I = UDRMap.find(D); 1433 if (I != UDRMap.end()) 1434 return I->second; 1435 emitUserDefinedReduction(/*CGF=*/nullptr, D); 1436 return UDRMap.lookup(D); 1437 } 1438 1439 static llvm::Function *emitParallelOrTeamsOutlinedFunction( 1440 CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS, 1441 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, 1442 const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) { 1443 assert(ThreadIDVar->getType()->isPointerType() && 1444 "thread id variable must be of type kmp_int32 *"); 1445 CodeGenFunction CGF(CGM, true); 1446 bool HasCancel = false; 1447 if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D)) 1448 HasCancel = OPD->hasCancel(); 1449 else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D)) 1450 HasCancel = OPSD->hasCancel(); 1451 else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D)) 1452 HasCancel = OPFD->hasCancel(); 1453 else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D)) 1454 HasCancel = OPFD->hasCancel(); 1455 else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D)) 1456 HasCancel = OPFD->hasCancel(); 1457 else if (const auto *OPFD = 1458 dyn_cast<OMPTeamsDistributeParallelForDirective>(&D)) 1459 HasCancel = OPFD->hasCancel(); 1460 else if (const auto *OPFD = 1461 dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D)) 1462 HasCancel = OPFD->hasCancel(); 1463 CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind, 1464 HasCancel, OutlinedHelperName); 1465 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1466 return CGF.GenerateOpenMPCapturedStmtFunction(*CS); 1467 } 1468 1469 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction( 1470 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1471 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1472 const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel); 1473 return emitParallelOrTeamsOutlinedFunction( 1474 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1475 } 1476 1477 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction( 1478 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1479 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1480 const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams); 1481 return emitParallelOrTeamsOutlinedFunction( 1482 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1483 } 1484 1485 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction( 1486 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1487 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 1488 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 1489 bool Tied, unsigned &NumberOfParts) { 1490 auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF, 1491 PrePostActionTy &) { 1492 llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc()); 1493 llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 1494 llvm::Value *TaskArgs[] = { 1495 UpLoc, ThreadID, 1496 CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar), 1497 TaskTVar->getType()->castAs<PointerType>()) 1498 .getPointer()}; 1499 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs); 1500 }; 1501 CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar, 1502 UntiedCodeGen); 1503 CodeGen.setAction(Action); 1504 assert(!ThreadIDVar->getType()->isPointerType() && 1505 "thread id variable must be of type kmp_int32 for tasks"); 1506 const OpenMPDirectiveKind Region = 1507 isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop 1508 : OMPD_task; 1509 const CapturedStmt *CS = D.getCapturedStmt(Region); 1510 const auto *TD = dyn_cast<OMPTaskDirective>(&D); 1511 CodeGenFunction CGF(CGM, true); 1512 CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, 1513 InnermostKind, 1514 TD ? TD->hasCancel() : false, Action); 1515 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1516 llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS); 1517 if (!Tied) 1518 NumberOfParts = Action.getNumberOfParts(); 1519 return Res; 1520 } 1521 1522 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM, 1523 const RecordDecl *RD, const CGRecordLayout &RL, 1524 ArrayRef<llvm::Constant *> Data) { 1525 llvm::StructType *StructTy = RL.getLLVMType(); 1526 unsigned PrevIdx = 0; 1527 ConstantInitBuilder CIBuilder(CGM); 1528 auto DI = Data.begin(); 1529 for (const FieldDecl *FD : RD->fields()) { 1530 unsigned Idx = RL.getLLVMFieldNo(FD); 1531 // Fill the alignment. 1532 for (unsigned I = PrevIdx; I < Idx; ++I) 1533 Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I))); 1534 PrevIdx = Idx + 1; 1535 Fields.add(*DI); 1536 ++DI; 1537 } 1538 } 1539 1540 template <class... As> 1541 static llvm::GlobalVariable * 1542 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant, 1543 ArrayRef<llvm::Constant *> Data, const Twine &Name, 1544 As &&... Args) { 1545 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1546 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1547 ConstantInitBuilder CIBuilder(CGM); 1548 ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType()); 1549 buildStructValue(Fields, CGM, RD, RL, Data); 1550 return Fields.finishAndCreateGlobal( 1551 Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant, 1552 std::forward<As>(Args)...); 1553 } 1554 1555 template <typename T> 1556 static void 1557 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty, 1558 ArrayRef<llvm::Constant *> Data, 1559 T &Parent) { 1560 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1561 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1562 ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType()); 1563 buildStructValue(Fields, CGM, RD, RL, Data); 1564 Fields.finishAndAddTo(Parent); 1565 } 1566 1567 Address CGOpenMPRuntime::getOrCreateDefaultLocation(unsigned Flags) { 1568 CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy); 1569 unsigned Reserved2Flags = getDefaultLocationReserved2Flags(); 1570 FlagsTy FlagsKey(Flags, Reserved2Flags); 1571 llvm::Value *Entry = OpenMPDefaultLocMap.lookup(FlagsKey); 1572 if (!Entry) { 1573 if (!DefaultOpenMPPSource) { 1574 // Initialize default location for psource field of ident_t structure of 1575 // all ident_t objects. Format is ";file;function;line;column;;". 1576 // Taken from 1577 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp_str.cpp 1578 DefaultOpenMPPSource = 1579 CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;").getPointer(); 1580 DefaultOpenMPPSource = 1581 llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy); 1582 } 1583 1584 llvm::Constant *Data[] = { 1585 llvm::ConstantInt::getNullValue(CGM.Int32Ty), 1586 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 1587 llvm::ConstantInt::get(CGM.Int32Ty, Reserved2Flags), 1588 llvm::ConstantInt::getNullValue(CGM.Int32Ty), DefaultOpenMPPSource}; 1589 llvm::GlobalValue *DefaultOpenMPLocation = 1590 createGlobalStruct(CGM, IdentQTy, isDefaultLocationConstant(), Data, "", 1591 llvm::GlobalValue::PrivateLinkage); 1592 DefaultOpenMPLocation->setUnnamedAddr( 1593 llvm::GlobalValue::UnnamedAddr::Global); 1594 1595 OpenMPDefaultLocMap[FlagsKey] = Entry = DefaultOpenMPLocation; 1596 } 1597 return Address(Entry, Align); 1598 } 1599 1600 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF, 1601 bool AtCurrentPoint) { 1602 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1603 assert(!Elem.second.ServiceInsertPt && "Insert point is set already."); 1604 1605 llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty); 1606 if (AtCurrentPoint) { 1607 Elem.second.ServiceInsertPt = new llvm::BitCastInst( 1608 Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock()); 1609 } else { 1610 Elem.second.ServiceInsertPt = 1611 new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt"); 1612 Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt); 1613 } 1614 } 1615 1616 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) { 1617 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1618 if (Elem.second.ServiceInsertPt) { 1619 llvm::Instruction *Ptr = Elem.second.ServiceInsertPt; 1620 Elem.second.ServiceInsertPt = nullptr; 1621 Ptr->eraseFromParent(); 1622 } 1623 } 1624 1625 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF, 1626 SourceLocation Loc, 1627 unsigned Flags) { 1628 Flags |= OMP_IDENT_KMPC; 1629 // If no debug info is generated - return global default location. 1630 if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo || 1631 Loc.isInvalid()) 1632 return getOrCreateDefaultLocation(Flags).getPointer(); 1633 1634 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1635 1636 CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy); 1637 Address LocValue = Address::invalid(); 1638 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1639 if (I != OpenMPLocThreadIDMap.end()) 1640 LocValue = Address(I->second.DebugLoc, Align); 1641 1642 // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if 1643 // GetOpenMPThreadID was called before this routine. 1644 if (!LocValue.isValid()) { 1645 // Generate "ident_t .kmpc_loc.addr;" 1646 Address AI = CGF.CreateMemTemp(IdentQTy, ".kmpc_loc.addr"); 1647 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1648 Elem.second.DebugLoc = AI.getPointer(); 1649 LocValue = AI; 1650 1651 if (!Elem.second.ServiceInsertPt) 1652 setLocThreadIdInsertPt(CGF); 1653 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1654 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1655 CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags), 1656 CGF.getTypeSize(IdentQTy)); 1657 } 1658 1659 // char **psource = &.kmpc_loc_<flags>.addr.psource; 1660 LValue Base = CGF.MakeAddrLValue(LocValue, IdentQTy); 1661 auto Fields = cast<RecordDecl>(IdentQTy->getAsTagDecl())->field_begin(); 1662 LValue PSource = 1663 CGF.EmitLValueForField(Base, *std::next(Fields, IdentField_PSource)); 1664 1665 llvm::Value *OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding()); 1666 if (OMPDebugLoc == nullptr) { 1667 SmallString<128> Buffer2; 1668 llvm::raw_svector_ostream OS2(Buffer2); 1669 // Build debug location 1670 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc); 1671 OS2 << ";" << PLoc.getFilename() << ";"; 1672 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) 1673 OS2 << FD->getQualifiedNameAsString(); 1674 OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;"; 1675 OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str()); 1676 OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc; 1677 } 1678 // *psource = ";<File>;<Function>;<Line>;<Column>;;"; 1679 CGF.EmitStoreOfScalar(OMPDebugLoc, PSource); 1680 1681 // Our callers always pass this to a runtime function, so for 1682 // convenience, go ahead and return a naked pointer. 1683 return LocValue.getPointer(); 1684 } 1685 1686 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF, 1687 SourceLocation Loc) { 1688 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1689 1690 llvm::Value *ThreadID = nullptr; 1691 // Check whether we've already cached a load of the thread id in this 1692 // function. 1693 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1694 if (I != OpenMPLocThreadIDMap.end()) { 1695 ThreadID = I->second.ThreadID; 1696 if (ThreadID != nullptr) 1697 return ThreadID; 1698 } 1699 // If exceptions are enabled, do not use parameter to avoid possible crash. 1700 if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions || 1701 !CGF.getLangOpts().CXXExceptions || 1702 CGF.Builder.GetInsertBlock() == CGF.AllocaInsertPt->getParent()) { 1703 if (auto *OMPRegionInfo = 1704 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 1705 if (OMPRegionInfo->getThreadIDVariable()) { 1706 // Check if this an outlined function with thread id passed as argument. 1707 LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF); 1708 ThreadID = CGF.EmitLoadOfScalar(LVal, Loc); 1709 // If value loaded in entry block, cache it and use it everywhere in 1710 // function. 1711 if (CGF.Builder.GetInsertBlock() == CGF.AllocaInsertPt->getParent()) { 1712 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1713 Elem.second.ThreadID = ThreadID; 1714 } 1715 return ThreadID; 1716 } 1717 } 1718 } 1719 1720 // This is not an outlined function region - need to call __kmpc_int32 1721 // kmpc_global_thread_num(ident_t *loc). 1722 // Generate thread id value and cache this value for use across the 1723 // function. 1724 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1725 if (!Elem.second.ServiceInsertPt) 1726 setLocThreadIdInsertPt(CGF); 1727 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1728 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1729 llvm::CallInst *Call = CGF.Builder.CreateCall( 1730 createRuntimeFunction(OMPRTL__kmpc_global_thread_num), 1731 emitUpdateLocation(CGF, Loc)); 1732 Call->setCallingConv(CGF.getRuntimeCC()); 1733 Elem.second.ThreadID = Call; 1734 return Call; 1735 } 1736 1737 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) { 1738 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1739 if (OpenMPLocThreadIDMap.count(CGF.CurFn)) { 1740 clearLocThreadIdInsertPt(CGF); 1741 OpenMPLocThreadIDMap.erase(CGF.CurFn); 1742 } 1743 if (FunctionUDRMap.count(CGF.CurFn) > 0) { 1744 for(auto *D : FunctionUDRMap[CGF.CurFn]) 1745 UDRMap.erase(D); 1746 FunctionUDRMap.erase(CGF.CurFn); 1747 } 1748 auto I = FunctionUDMMap.find(CGF.CurFn); 1749 if (I != FunctionUDMMap.end()) { 1750 for(auto *D : I->second) 1751 UDMMap.erase(D); 1752 FunctionUDMMap.erase(I); 1753 } 1754 } 1755 1756 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() { 1757 return IdentTy->getPointerTo(); 1758 } 1759 1760 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() { 1761 if (!Kmpc_MicroTy) { 1762 // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...) 1763 llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty), 1764 llvm::PointerType::getUnqual(CGM.Int32Ty)}; 1765 Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true); 1766 } 1767 return llvm::PointerType::getUnqual(Kmpc_MicroTy); 1768 } 1769 1770 llvm::FunctionCallee CGOpenMPRuntime::createRuntimeFunction(unsigned Function) { 1771 llvm::FunctionCallee RTLFn = nullptr; 1772 switch (static_cast<OpenMPRTLFunction>(Function)) { 1773 case OMPRTL__kmpc_fork_call: { 1774 // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro 1775 // microtask, ...); 1776 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1777 getKmpc_MicroPointerTy()}; 1778 auto *FnTy = 1779 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true); 1780 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call"); 1781 if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) { 1782 if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) { 1783 llvm::LLVMContext &Ctx = F->getContext(); 1784 llvm::MDBuilder MDB(Ctx); 1785 // Annotate the callback behavior of the __kmpc_fork_call: 1786 // - The callback callee is argument number 2 (microtask). 1787 // - The first two arguments of the callback callee are unknown (-1). 1788 // - All variadic arguments to the __kmpc_fork_call are passed to the 1789 // callback callee. 1790 F->addMetadata( 1791 llvm::LLVMContext::MD_callback, 1792 *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding( 1793 2, {-1, -1}, 1794 /* VarArgsArePassed */ true)})); 1795 } 1796 } 1797 break; 1798 } 1799 case OMPRTL__kmpc_global_thread_num: { 1800 // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc); 1801 llvm::Type *TypeParams[] = {getIdentTyPointerTy()}; 1802 auto *FnTy = 1803 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1804 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num"); 1805 break; 1806 } 1807 case OMPRTL__kmpc_threadprivate_cached: { 1808 // Build void *__kmpc_threadprivate_cached(ident_t *loc, 1809 // kmp_int32 global_tid, void *data, size_t size, void ***cache); 1810 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1811 CGM.VoidPtrTy, CGM.SizeTy, 1812 CGM.VoidPtrTy->getPointerTo()->getPointerTo()}; 1813 auto *FnTy = 1814 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false); 1815 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached"); 1816 break; 1817 } 1818 case OMPRTL__kmpc_critical: { 1819 // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid, 1820 // kmp_critical_name *crit); 1821 llvm::Type *TypeParams[] = { 1822 getIdentTyPointerTy(), CGM.Int32Ty, 1823 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1824 auto *FnTy = 1825 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1826 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical"); 1827 break; 1828 } 1829 case OMPRTL__kmpc_critical_with_hint: { 1830 // Build void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 global_tid, 1831 // kmp_critical_name *crit, uintptr_t hint); 1832 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1833 llvm::PointerType::getUnqual(KmpCriticalNameTy), 1834 CGM.IntPtrTy}; 1835 auto *FnTy = 1836 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1837 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical_with_hint"); 1838 break; 1839 } 1840 case OMPRTL__kmpc_threadprivate_register: { 1841 // Build void __kmpc_threadprivate_register(ident_t *, void *data, 1842 // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor); 1843 // typedef void *(*kmpc_ctor)(void *); 1844 auto *KmpcCtorTy = 1845 llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 1846 /*isVarArg*/ false)->getPointerTo(); 1847 // typedef void *(*kmpc_cctor)(void *, void *); 1848 llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 1849 auto *KmpcCopyCtorTy = 1850 llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs, 1851 /*isVarArg*/ false) 1852 ->getPointerTo(); 1853 // typedef void (*kmpc_dtor)(void *); 1854 auto *KmpcDtorTy = 1855 llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false) 1856 ->getPointerTo(); 1857 llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy, 1858 KmpcCopyCtorTy, KmpcDtorTy}; 1859 auto *FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs, 1860 /*isVarArg*/ false); 1861 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register"); 1862 break; 1863 } 1864 case OMPRTL__kmpc_end_critical: { 1865 // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid, 1866 // kmp_critical_name *crit); 1867 llvm::Type *TypeParams[] = { 1868 getIdentTyPointerTy(), CGM.Int32Ty, 1869 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1870 auto *FnTy = 1871 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1872 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical"); 1873 break; 1874 } 1875 case OMPRTL__kmpc_cancel_barrier: { 1876 // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32 1877 // global_tid); 1878 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1879 auto *FnTy = 1880 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1881 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier"); 1882 break; 1883 } 1884 case OMPRTL__kmpc_barrier: { 1885 // Build void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid); 1886 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1887 auto *FnTy = 1888 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1889 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_barrier"); 1890 break; 1891 } 1892 case OMPRTL__kmpc_for_static_fini: { 1893 // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid); 1894 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1895 auto *FnTy = 1896 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1897 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini"); 1898 break; 1899 } 1900 case OMPRTL__kmpc_push_num_threads: { 1901 // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid, 1902 // kmp_int32 num_threads) 1903 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1904 CGM.Int32Ty}; 1905 auto *FnTy = 1906 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1907 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads"); 1908 break; 1909 } 1910 case OMPRTL__kmpc_serialized_parallel: { 1911 // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32 1912 // global_tid); 1913 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1914 auto *FnTy = 1915 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1916 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel"); 1917 break; 1918 } 1919 case OMPRTL__kmpc_end_serialized_parallel: { 1920 // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32 1921 // global_tid); 1922 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1923 auto *FnTy = 1924 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1925 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel"); 1926 break; 1927 } 1928 case OMPRTL__kmpc_flush: { 1929 // Build void __kmpc_flush(ident_t *loc); 1930 llvm::Type *TypeParams[] = {getIdentTyPointerTy()}; 1931 auto *FnTy = 1932 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1933 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush"); 1934 break; 1935 } 1936 case OMPRTL__kmpc_master: { 1937 // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid); 1938 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1939 auto *FnTy = 1940 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1941 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master"); 1942 break; 1943 } 1944 case OMPRTL__kmpc_end_master: { 1945 // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid); 1946 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1947 auto *FnTy = 1948 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1949 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master"); 1950 break; 1951 } 1952 case OMPRTL__kmpc_omp_taskyield: { 1953 // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid, 1954 // int end_part); 1955 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 1956 auto *FnTy = 1957 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1958 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield"); 1959 break; 1960 } 1961 case OMPRTL__kmpc_single: { 1962 // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid); 1963 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1964 auto *FnTy = 1965 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1966 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single"); 1967 break; 1968 } 1969 case OMPRTL__kmpc_end_single: { 1970 // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid); 1971 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1972 auto *FnTy = 1973 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1974 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single"); 1975 break; 1976 } 1977 case OMPRTL__kmpc_omp_task_alloc: { 1978 // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 1979 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 1980 // kmp_routine_entry_t *task_entry); 1981 assert(KmpRoutineEntryPtrTy != nullptr && 1982 "Type kmp_routine_entry_t must be created."); 1983 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 1984 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy}; 1985 // Return void * and then cast to particular kmp_task_t type. 1986 auto *FnTy = 1987 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 1988 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc"); 1989 break; 1990 } 1991 case OMPRTL__kmpc_omp_target_task_alloc: { 1992 // Build kmp_task_t *__kmpc_omp_target_task_alloc(ident_t *, kmp_int32 gtid, 1993 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 1994 // kmp_routine_entry_t *task_entry, kmp_int64 device_id); 1995 assert(KmpRoutineEntryPtrTy != nullptr && 1996 "Type kmp_routine_entry_t must be created."); 1997 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 1998 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy, 1999 CGM.Int64Ty}; 2000 // Return void * and then cast to particular kmp_task_t type. 2001 auto *FnTy = 2002 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2003 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_target_task_alloc"); 2004 break; 2005 } 2006 case OMPRTL__kmpc_omp_task: { 2007 // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2008 // *new_task); 2009 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2010 CGM.VoidPtrTy}; 2011 auto *FnTy = 2012 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2013 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task"); 2014 break; 2015 } 2016 case OMPRTL__kmpc_copyprivate: { 2017 // Build void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid, 2018 // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *), 2019 // kmp_int32 didit); 2020 llvm::Type *CpyTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2021 auto *CpyFnTy = 2022 llvm::FunctionType::get(CGM.VoidTy, CpyTypeParams, /*isVarArg=*/false); 2023 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.SizeTy, 2024 CGM.VoidPtrTy, CpyFnTy->getPointerTo(), 2025 CGM.Int32Ty}; 2026 auto *FnTy = 2027 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2028 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_copyprivate"); 2029 break; 2030 } 2031 case OMPRTL__kmpc_reduce: { 2032 // Build kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid, 2033 // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void 2034 // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck); 2035 llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2036 auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams, 2037 /*isVarArg=*/false); 2038 llvm::Type *TypeParams[] = { 2039 getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy, 2040 CGM.VoidPtrTy, ReduceFnTy->getPointerTo(), 2041 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2042 auto *FnTy = 2043 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2044 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce"); 2045 break; 2046 } 2047 case OMPRTL__kmpc_reduce_nowait: { 2048 // Build kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32 2049 // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, 2050 // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name 2051 // *lck); 2052 llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2053 auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams, 2054 /*isVarArg=*/false); 2055 llvm::Type *TypeParams[] = { 2056 getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy, 2057 CGM.VoidPtrTy, ReduceFnTy->getPointerTo(), 2058 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2059 auto *FnTy = 2060 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2061 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce_nowait"); 2062 break; 2063 } 2064 case OMPRTL__kmpc_end_reduce: { 2065 // Build void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid, 2066 // kmp_critical_name *lck); 2067 llvm::Type *TypeParams[] = { 2068 getIdentTyPointerTy(), CGM.Int32Ty, 2069 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2070 auto *FnTy = 2071 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2072 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce"); 2073 break; 2074 } 2075 case OMPRTL__kmpc_end_reduce_nowait: { 2076 // Build __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid, 2077 // kmp_critical_name *lck); 2078 llvm::Type *TypeParams[] = { 2079 getIdentTyPointerTy(), CGM.Int32Ty, 2080 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 2081 auto *FnTy = 2082 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2083 RTLFn = 2084 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce_nowait"); 2085 break; 2086 } 2087 case OMPRTL__kmpc_omp_task_begin_if0: { 2088 // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2089 // *new_task); 2090 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2091 CGM.VoidPtrTy}; 2092 auto *FnTy = 2093 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2094 RTLFn = 2095 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_begin_if0"); 2096 break; 2097 } 2098 case OMPRTL__kmpc_omp_task_complete_if0: { 2099 // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 2100 // *new_task); 2101 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2102 CGM.VoidPtrTy}; 2103 auto *FnTy = 2104 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2105 RTLFn = CGM.CreateRuntimeFunction(FnTy, 2106 /*Name=*/"__kmpc_omp_task_complete_if0"); 2107 break; 2108 } 2109 case OMPRTL__kmpc_ordered: { 2110 // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid); 2111 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2112 auto *FnTy = 2113 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2114 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_ordered"); 2115 break; 2116 } 2117 case OMPRTL__kmpc_end_ordered: { 2118 // Build void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid); 2119 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2120 auto *FnTy = 2121 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2122 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_ordered"); 2123 break; 2124 } 2125 case OMPRTL__kmpc_omp_taskwait: { 2126 // Build kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 global_tid); 2127 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2128 auto *FnTy = 2129 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2130 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_omp_taskwait"); 2131 break; 2132 } 2133 case OMPRTL__kmpc_taskgroup: { 2134 // Build void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid); 2135 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2136 auto *FnTy = 2137 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2138 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_taskgroup"); 2139 break; 2140 } 2141 case OMPRTL__kmpc_end_taskgroup: { 2142 // Build void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid); 2143 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2144 auto *FnTy = 2145 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2146 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_taskgroup"); 2147 break; 2148 } 2149 case OMPRTL__kmpc_push_proc_bind: { 2150 // Build void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid, 2151 // int proc_bind) 2152 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2153 auto *FnTy = 2154 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2155 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_proc_bind"); 2156 break; 2157 } 2158 case OMPRTL__kmpc_omp_task_with_deps: { 2159 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 2160 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 2161 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 2162 llvm::Type *TypeParams[] = { 2163 getIdentTyPointerTy(), CGM.Int32Ty, CGM.VoidPtrTy, CGM.Int32Ty, 2164 CGM.VoidPtrTy, CGM.Int32Ty, CGM.VoidPtrTy}; 2165 auto *FnTy = 2166 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2167 RTLFn = 2168 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_with_deps"); 2169 break; 2170 } 2171 case OMPRTL__kmpc_omp_wait_deps: { 2172 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 2173 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 ndeps_noalias, 2174 // kmp_depend_info_t *noalias_dep_list); 2175 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2176 CGM.Int32Ty, CGM.VoidPtrTy, 2177 CGM.Int32Ty, CGM.VoidPtrTy}; 2178 auto *FnTy = 2179 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2180 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_wait_deps"); 2181 break; 2182 } 2183 case OMPRTL__kmpc_cancellationpoint: { 2184 // Build kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 2185 // global_tid, kmp_int32 cncl_kind) 2186 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2187 auto *FnTy = 2188 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2189 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancellationpoint"); 2190 break; 2191 } 2192 case OMPRTL__kmpc_cancel: { 2193 // Build kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 2194 // kmp_int32 cncl_kind) 2195 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2196 auto *FnTy = 2197 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2198 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancel"); 2199 break; 2200 } 2201 case OMPRTL__kmpc_push_num_teams: { 2202 // Build void kmpc_push_num_teams (ident_t loc, kmp_int32 global_tid, 2203 // kmp_int32 num_teams, kmp_int32 num_threads) 2204 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2205 CGM.Int32Ty}; 2206 auto *FnTy = 2207 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2208 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_teams"); 2209 break; 2210 } 2211 case OMPRTL__kmpc_fork_teams: { 2212 // Build void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro 2213 // microtask, ...); 2214 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2215 getKmpc_MicroPointerTy()}; 2216 auto *FnTy = 2217 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true); 2218 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_teams"); 2219 if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) { 2220 if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) { 2221 llvm::LLVMContext &Ctx = F->getContext(); 2222 llvm::MDBuilder MDB(Ctx); 2223 // Annotate the callback behavior of the __kmpc_fork_teams: 2224 // - The callback callee is argument number 2 (microtask). 2225 // - The first two arguments of the callback callee are unknown (-1). 2226 // - All variadic arguments to the __kmpc_fork_teams are passed to the 2227 // callback callee. 2228 F->addMetadata( 2229 llvm::LLVMContext::MD_callback, 2230 *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding( 2231 2, {-1, -1}, 2232 /* VarArgsArePassed */ true)})); 2233 } 2234 } 2235 break; 2236 } 2237 case OMPRTL__kmpc_taskloop: { 2238 // Build void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 2239 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 2240 // sched, kmp_uint64 grainsize, void *task_dup); 2241 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), 2242 CGM.IntTy, 2243 CGM.VoidPtrTy, 2244 CGM.IntTy, 2245 CGM.Int64Ty->getPointerTo(), 2246 CGM.Int64Ty->getPointerTo(), 2247 CGM.Int64Ty, 2248 CGM.IntTy, 2249 CGM.IntTy, 2250 CGM.Int64Ty, 2251 CGM.VoidPtrTy}; 2252 auto *FnTy = 2253 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2254 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_taskloop"); 2255 break; 2256 } 2257 case OMPRTL__kmpc_doacross_init: { 2258 // Build void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32 2259 // num_dims, struct kmp_dim *dims); 2260 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), 2261 CGM.Int32Ty, 2262 CGM.Int32Ty, 2263 CGM.VoidPtrTy}; 2264 auto *FnTy = 2265 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2266 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_init"); 2267 break; 2268 } 2269 case OMPRTL__kmpc_doacross_fini: { 2270 // Build void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid); 2271 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2272 auto *FnTy = 2273 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2274 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_fini"); 2275 break; 2276 } 2277 case OMPRTL__kmpc_doacross_post: { 2278 // Build void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64 2279 // *vec); 2280 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2281 CGM.Int64Ty->getPointerTo()}; 2282 auto *FnTy = 2283 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2284 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_post"); 2285 break; 2286 } 2287 case OMPRTL__kmpc_doacross_wait: { 2288 // Build void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64 2289 // *vec); 2290 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2291 CGM.Int64Ty->getPointerTo()}; 2292 auto *FnTy = 2293 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2294 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_wait"); 2295 break; 2296 } 2297 case OMPRTL__kmpc_task_reduction_init: { 2298 // Build void *__kmpc_task_reduction_init(int gtid, int num_data, void 2299 // *data); 2300 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.IntTy, CGM.VoidPtrTy}; 2301 auto *FnTy = 2302 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2303 RTLFn = 2304 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_task_reduction_init"); 2305 break; 2306 } 2307 case OMPRTL__kmpc_task_reduction_get_th_data: { 2308 // Build void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 2309 // *d); 2310 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy}; 2311 auto *FnTy = 2312 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2313 RTLFn = CGM.CreateRuntimeFunction( 2314 FnTy, /*Name=*/"__kmpc_task_reduction_get_th_data"); 2315 break; 2316 } 2317 case OMPRTL__kmpc_alloc: { 2318 // Build to void *__kmpc_alloc(int gtid, size_t sz, omp_allocator_handle_t 2319 // al); omp_allocator_handle_t type is void *. 2320 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.SizeTy, CGM.VoidPtrTy}; 2321 auto *FnTy = 2322 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2323 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_alloc"); 2324 break; 2325 } 2326 case OMPRTL__kmpc_free: { 2327 // Build to void __kmpc_free(int gtid, void *ptr, omp_allocator_handle_t 2328 // al); omp_allocator_handle_t type is void *. 2329 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy}; 2330 auto *FnTy = 2331 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2332 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_free"); 2333 break; 2334 } 2335 case OMPRTL__kmpc_push_target_tripcount: { 2336 // Build void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64 2337 // size); 2338 llvm::Type *TypeParams[] = {CGM.Int64Ty, CGM.Int64Ty}; 2339 llvm::FunctionType *FnTy = 2340 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2341 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_target_tripcount"); 2342 break; 2343 } 2344 case OMPRTL__tgt_target: { 2345 // Build int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t 2346 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2347 // *arg_types); 2348 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2349 CGM.VoidPtrTy, 2350 CGM.Int32Ty, 2351 CGM.VoidPtrPtrTy, 2352 CGM.VoidPtrPtrTy, 2353 CGM.Int64Ty->getPointerTo(), 2354 CGM.Int64Ty->getPointerTo()}; 2355 auto *FnTy = 2356 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2357 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target"); 2358 break; 2359 } 2360 case OMPRTL__tgt_target_nowait: { 2361 // Build int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr, 2362 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, 2363 // int64_t *arg_types); 2364 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2365 CGM.VoidPtrTy, 2366 CGM.Int32Ty, 2367 CGM.VoidPtrPtrTy, 2368 CGM.VoidPtrPtrTy, 2369 CGM.Int64Ty->getPointerTo(), 2370 CGM.Int64Ty->getPointerTo()}; 2371 auto *FnTy = 2372 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2373 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_nowait"); 2374 break; 2375 } 2376 case OMPRTL__tgt_target_teams: { 2377 // Build int32_t __tgt_target_teams(int64_t device_id, void *host_ptr, 2378 // int32_t arg_num, void** args_base, void **args, int64_t *arg_sizes, 2379 // int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 2380 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2381 CGM.VoidPtrTy, 2382 CGM.Int32Ty, 2383 CGM.VoidPtrPtrTy, 2384 CGM.VoidPtrPtrTy, 2385 CGM.Int64Ty->getPointerTo(), 2386 CGM.Int64Ty->getPointerTo(), 2387 CGM.Int32Ty, 2388 CGM.Int32Ty}; 2389 auto *FnTy = 2390 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2391 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams"); 2392 break; 2393 } 2394 case OMPRTL__tgt_target_teams_nowait: { 2395 // Build int32_t __tgt_target_teams_nowait(int64_t device_id, void 2396 // *host_ptr, int32_t arg_num, void** args_base, void **args, int64_t 2397 // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 2398 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2399 CGM.VoidPtrTy, 2400 CGM.Int32Ty, 2401 CGM.VoidPtrPtrTy, 2402 CGM.VoidPtrPtrTy, 2403 CGM.Int64Ty->getPointerTo(), 2404 CGM.Int64Ty->getPointerTo(), 2405 CGM.Int32Ty, 2406 CGM.Int32Ty}; 2407 auto *FnTy = 2408 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2409 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams_nowait"); 2410 break; 2411 } 2412 case OMPRTL__tgt_register_requires: { 2413 // Build void __tgt_register_requires(int64_t flags); 2414 llvm::Type *TypeParams[] = {CGM.Int64Ty}; 2415 auto *FnTy = 2416 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2417 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_requires"); 2418 break; 2419 } 2420 case OMPRTL__tgt_register_lib: { 2421 // Build void __tgt_register_lib(__tgt_bin_desc *desc); 2422 QualType ParamTy = 2423 CGM.getContext().getPointerType(getTgtBinaryDescriptorQTy()); 2424 llvm::Type *TypeParams[] = {CGM.getTypes().ConvertTypeForMem(ParamTy)}; 2425 auto *FnTy = 2426 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2427 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_lib"); 2428 break; 2429 } 2430 case OMPRTL__tgt_unregister_lib: { 2431 // Build void __tgt_unregister_lib(__tgt_bin_desc *desc); 2432 QualType ParamTy = 2433 CGM.getContext().getPointerType(getTgtBinaryDescriptorQTy()); 2434 llvm::Type *TypeParams[] = {CGM.getTypes().ConvertTypeForMem(ParamTy)}; 2435 auto *FnTy = 2436 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2437 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_unregister_lib"); 2438 break; 2439 } 2440 case OMPRTL__tgt_target_data_begin: { 2441 // Build void __tgt_target_data_begin(int64_t device_id, int32_t arg_num, 2442 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2443 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2444 CGM.Int32Ty, 2445 CGM.VoidPtrPtrTy, 2446 CGM.VoidPtrPtrTy, 2447 CGM.Int64Ty->getPointerTo(), 2448 CGM.Int64Ty->getPointerTo()}; 2449 auto *FnTy = 2450 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2451 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin"); 2452 break; 2453 } 2454 case OMPRTL__tgt_target_data_begin_nowait: { 2455 // Build void __tgt_target_data_begin_nowait(int64_t device_id, int32_t 2456 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2457 // *arg_types); 2458 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2459 CGM.Int32Ty, 2460 CGM.VoidPtrPtrTy, 2461 CGM.VoidPtrPtrTy, 2462 CGM.Int64Ty->getPointerTo(), 2463 CGM.Int64Ty->getPointerTo()}; 2464 auto *FnTy = 2465 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2466 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin_nowait"); 2467 break; 2468 } 2469 case OMPRTL__tgt_target_data_end: { 2470 // Build void __tgt_target_data_end(int64_t device_id, int32_t arg_num, 2471 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2472 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2473 CGM.Int32Ty, 2474 CGM.VoidPtrPtrTy, 2475 CGM.VoidPtrPtrTy, 2476 CGM.Int64Ty->getPointerTo(), 2477 CGM.Int64Ty->getPointerTo()}; 2478 auto *FnTy = 2479 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2480 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end"); 2481 break; 2482 } 2483 case OMPRTL__tgt_target_data_end_nowait: { 2484 // Build void __tgt_target_data_end_nowait(int64_t device_id, int32_t 2485 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2486 // *arg_types); 2487 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2488 CGM.Int32Ty, 2489 CGM.VoidPtrPtrTy, 2490 CGM.VoidPtrPtrTy, 2491 CGM.Int64Ty->getPointerTo(), 2492 CGM.Int64Ty->getPointerTo()}; 2493 auto *FnTy = 2494 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2495 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end_nowait"); 2496 break; 2497 } 2498 case OMPRTL__tgt_target_data_update: { 2499 // Build void __tgt_target_data_update(int64_t device_id, int32_t arg_num, 2500 // void** args_base, void **args, int64_t *arg_sizes, int64_t *arg_types); 2501 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2502 CGM.Int32Ty, 2503 CGM.VoidPtrPtrTy, 2504 CGM.VoidPtrPtrTy, 2505 CGM.Int64Ty->getPointerTo(), 2506 CGM.Int64Ty->getPointerTo()}; 2507 auto *FnTy = 2508 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2509 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update"); 2510 break; 2511 } 2512 case OMPRTL__tgt_target_data_update_nowait: { 2513 // Build void __tgt_target_data_update_nowait(int64_t device_id, int32_t 2514 // arg_num, void** args_base, void **args, int64_t *arg_sizes, int64_t 2515 // *arg_types); 2516 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2517 CGM.Int32Ty, 2518 CGM.VoidPtrPtrTy, 2519 CGM.VoidPtrPtrTy, 2520 CGM.Int64Ty->getPointerTo(), 2521 CGM.Int64Ty->getPointerTo()}; 2522 auto *FnTy = 2523 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2524 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update_nowait"); 2525 break; 2526 } 2527 case OMPRTL__tgt_mapper_num_components: { 2528 // Build int64_t __tgt_mapper_num_components(void *rt_mapper_handle); 2529 llvm::Type *TypeParams[] = {CGM.VoidPtrTy}; 2530 auto *FnTy = 2531 llvm::FunctionType::get(CGM.Int64Ty, TypeParams, /*isVarArg*/ false); 2532 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_mapper_num_components"); 2533 break; 2534 } 2535 case OMPRTL__tgt_push_mapper_component: { 2536 // Build void __tgt_push_mapper_component(void *rt_mapper_handle, void 2537 // *base, void *begin, int64_t size, int64_t type); 2538 llvm::Type *TypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy, CGM.VoidPtrTy, 2539 CGM.Int64Ty, CGM.Int64Ty}; 2540 auto *FnTy = 2541 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2542 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_push_mapper_component"); 2543 break; 2544 } 2545 } 2546 assert(RTLFn && "Unable to find OpenMP runtime function"); 2547 return RTLFn; 2548 } 2549 2550 llvm::FunctionCallee 2551 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned) { 2552 assert((IVSize == 32 || IVSize == 64) && 2553 "IV size is not compatible with the omp runtime"); 2554 StringRef Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4" 2555 : "__kmpc_for_static_init_4u") 2556 : (IVSigned ? "__kmpc_for_static_init_8" 2557 : "__kmpc_for_static_init_8u"); 2558 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2559 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 2560 llvm::Type *TypeParams[] = { 2561 getIdentTyPointerTy(), // loc 2562 CGM.Int32Ty, // tid 2563 CGM.Int32Ty, // schedtype 2564 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 2565 PtrTy, // p_lower 2566 PtrTy, // p_upper 2567 PtrTy, // p_stride 2568 ITy, // incr 2569 ITy // chunk 2570 }; 2571 auto *FnTy = 2572 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2573 return CGM.CreateRuntimeFunction(FnTy, Name); 2574 } 2575 2576 llvm::FunctionCallee 2577 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) { 2578 assert((IVSize == 32 || IVSize == 64) && 2579 "IV size is not compatible with the omp runtime"); 2580 StringRef Name = 2581 IVSize == 32 2582 ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u") 2583 : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u"); 2584 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2585 llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc 2586 CGM.Int32Ty, // tid 2587 CGM.Int32Ty, // schedtype 2588 ITy, // lower 2589 ITy, // upper 2590 ITy, // stride 2591 ITy // chunk 2592 }; 2593 auto *FnTy = 2594 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2595 return CGM.CreateRuntimeFunction(FnTy, Name); 2596 } 2597 2598 llvm::FunctionCallee 2599 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) { 2600 assert((IVSize == 32 || IVSize == 64) && 2601 "IV size is not compatible with the omp runtime"); 2602 StringRef Name = 2603 IVSize == 32 2604 ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u") 2605 : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u"); 2606 llvm::Type *TypeParams[] = { 2607 getIdentTyPointerTy(), // loc 2608 CGM.Int32Ty, // tid 2609 }; 2610 auto *FnTy = 2611 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2612 return CGM.CreateRuntimeFunction(FnTy, Name); 2613 } 2614 2615 llvm::FunctionCallee 2616 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) { 2617 assert((IVSize == 32 || IVSize == 64) && 2618 "IV size is not compatible with the omp runtime"); 2619 StringRef Name = 2620 IVSize == 32 2621 ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u") 2622 : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u"); 2623 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2624 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 2625 llvm::Type *TypeParams[] = { 2626 getIdentTyPointerTy(), // loc 2627 CGM.Int32Ty, // tid 2628 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 2629 PtrTy, // p_lower 2630 PtrTy, // p_upper 2631 PtrTy // p_stride 2632 }; 2633 auto *FnTy = 2634 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2635 return CGM.CreateRuntimeFunction(FnTy, Name); 2636 } 2637 2638 /// Obtain information that uniquely identifies a target entry. This 2639 /// consists of the file and device IDs as well as line number associated with 2640 /// the relevant entry source location. 2641 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc, 2642 unsigned &DeviceID, unsigned &FileID, 2643 unsigned &LineNum) { 2644 SourceManager &SM = C.getSourceManager(); 2645 2646 // The loc should be always valid and have a file ID (the user cannot use 2647 // #pragma directives in macros) 2648 2649 assert(Loc.isValid() && "Source location is expected to be always valid."); 2650 2651 PresumedLoc PLoc = SM.getPresumedLoc(Loc); 2652 assert(PLoc.isValid() && "Source location is expected to be always valid."); 2653 2654 llvm::sys::fs::UniqueID ID; 2655 if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID)) 2656 SM.getDiagnostics().Report(diag::err_cannot_open_file) 2657 << PLoc.getFilename() << EC.message(); 2658 2659 DeviceID = ID.getDevice(); 2660 FileID = ID.getFile(); 2661 LineNum = PLoc.getLine(); 2662 } 2663 2664 Address CGOpenMPRuntime::getAddrOfDeclareTargetVar(const VarDecl *VD) { 2665 if (CGM.getLangOpts().OpenMPSimd) 2666 return Address::invalid(); 2667 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 2668 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 2669 if (Res && (*Res == OMPDeclareTargetDeclAttr::MT_Link || 2670 (*Res == OMPDeclareTargetDeclAttr::MT_To && 2671 HasRequiresUnifiedSharedMemory))) { 2672 SmallString<64> PtrName; 2673 { 2674 llvm::raw_svector_ostream OS(PtrName); 2675 OS << CGM.getMangledName(GlobalDecl(VD)); 2676 if (!VD->isExternallyVisible()) { 2677 unsigned DeviceID, FileID, Line; 2678 getTargetEntryUniqueInfo(CGM.getContext(), 2679 VD->getCanonicalDecl()->getBeginLoc(), 2680 DeviceID, FileID, Line); 2681 OS << llvm::format("_%x", FileID); 2682 } 2683 OS << "_decl_tgt_ref_ptr"; 2684 } 2685 llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName); 2686 if (!Ptr) { 2687 QualType PtrTy = CGM.getContext().getPointerType(VD->getType()); 2688 Ptr = getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(PtrTy), 2689 PtrName); 2690 2691 auto *GV = cast<llvm::GlobalVariable>(Ptr); 2692 GV->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 2693 2694 if (!CGM.getLangOpts().OpenMPIsDevice) 2695 GV->setInitializer(CGM.GetAddrOfGlobal(VD)); 2696 registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr)); 2697 } 2698 return Address(Ptr, CGM.getContext().getDeclAlign(VD)); 2699 } 2700 return Address::invalid(); 2701 } 2702 2703 llvm::Constant * 2704 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) { 2705 assert(!CGM.getLangOpts().OpenMPUseTLS || 2706 !CGM.getContext().getTargetInfo().isTLSSupported()); 2707 // Lookup the entry, lazily creating it if necessary. 2708 std::string Suffix = getName({"cache", ""}); 2709 return getOrCreateInternalVariable( 2710 CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix)); 2711 } 2712 2713 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 2714 const VarDecl *VD, 2715 Address VDAddr, 2716 SourceLocation Loc) { 2717 if (CGM.getLangOpts().OpenMPUseTLS && 2718 CGM.getContext().getTargetInfo().isTLSSupported()) 2719 return VDAddr; 2720 2721 llvm::Type *VarTy = VDAddr.getElementType(); 2722 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2723 CGF.Builder.CreatePointerCast(VDAddr.getPointer(), 2724 CGM.Int8PtrTy), 2725 CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)), 2726 getOrCreateThreadPrivateCache(VD)}; 2727 return Address(CGF.EmitRuntimeCall( 2728 createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args), 2729 VDAddr.getAlignment()); 2730 } 2731 2732 void CGOpenMPRuntime::emitThreadPrivateVarInit( 2733 CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor, 2734 llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) { 2735 // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime 2736 // library. 2737 llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc); 2738 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num), 2739 OMPLoc); 2740 // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor) 2741 // to register constructor/destructor for variable. 2742 llvm::Value *Args[] = { 2743 OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy), 2744 Ctor, CopyCtor, Dtor}; 2745 CGF.EmitRuntimeCall( 2746 createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args); 2747 } 2748 2749 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition( 2750 const VarDecl *VD, Address VDAddr, SourceLocation Loc, 2751 bool PerformInit, CodeGenFunction *CGF) { 2752 if (CGM.getLangOpts().OpenMPUseTLS && 2753 CGM.getContext().getTargetInfo().isTLSSupported()) 2754 return nullptr; 2755 2756 VD = VD->getDefinition(CGM.getContext()); 2757 if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) { 2758 QualType ASTTy = VD->getType(); 2759 2760 llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr; 2761 const Expr *Init = VD->getAnyInitializer(); 2762 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 2763 // Generate function that re-emits the declaration's initializer into the 2764 // threadprivate copy of the variable VD 2765 CodeGenFunction CtorCGF(CGM); 2766 FunctionArgList Args; 2767 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 2768 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 2769 ImplicitParamDecl::Other); 2770 Args.push_back(&Dst); 2771 2772 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 2773 CGM.getContext().VoidPtrTy, Args); 2774 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2775 std::string Name = getName({"__kmpc_global_ctor_", ""}); 2776 llvm::Function *Fn = 2777 CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc); 2778 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI, 2779 Args, Loc, Loc); 2780 llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar( 2781 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 2782 CGM.getContext().VoidPtrTy, Dst.getLocation()); 2783 Address Arg = Address(ArgVal, VDAddr.getAlignment()); 2784 Arg = CtorCGF.Builder.CreateElementBitCast( 2785 Arg, CtorCGF.ConvertTypeForMem(ASTTy)); 2786 CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(), 2787 /*IsInitializer=*/true); 2788 ArgVal = CtorCGF.EmitLoadOfScalar( 2789 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 2790 CGM.getContext().VoidPtrTy, Dst.getLocation()); 2791 CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue); 2792 CtorCGF.FinishFunction(); 2793 Ctor = Fn; 2794 } 2795 if (VD->getType().isDestructedType() != QualType::DK_none) { 2796 // Generate function that emits destructor call for the threadprivate copy 2797 // of the variable VD 2798 CodeGenFunction DtorCGF(CGM); 2799 FunctionArgList Args; 2800 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 2801 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 2802 ImplicitParamDecl::Other); 2803 Args.push_back(&Dst); 2804 2805 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 2806 CGM.getContext().VoidTy, Args); 2807 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2808 std::string Name = getName({"__kmpc_global_dtor_", ""}); 2809 llvm::Function *Fn = 2810 CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc); 2811 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 2812 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args, 2813 Loc, Loc); 2814 // Create a scope with an artificial location for the body of this function. 2815 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 2816 llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar( 2817 DtorCGF.GetAddrOfLocalVar(&Dst), 2818 /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation()); 2819 DtorCGF.emitDestroy(Address(ArgVal, VDAddr.getAlignment()), ASTTy, 2820 DtorCGF.getDestroyer(ASTTy.isDestructedType()), 2821 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 2822 DtorCGF.FinishFunction(); 2823 Dtor = Fn; 2824 } 2825 // Do not emit init function if it is not required. 2826 if (!Ctor && !Dtor) 2827 return nullptr; 2828 2829 llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2830 auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs, 2831 /*isVarArg=*/false) 2832 ->getPointerTo(); 2833 // Copying constructor for the threadprivate variable. 2834 // Must be NULL - reserved by runtime, but currently it requires that this 2835 // parameter is always NULL. Otherwise it fires assertion. 2836 CopyCtor = llvm::Constant::getNullValue(CopyCtorTy); 2837 if (Ctor == nullptr) { 2838 auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 2839 /*isVarArg=*/false) 2840 ->getPointerTo(); 2841 Ctor = llvm::Constant::getNullValue(CtorTy); 2842 } 2843 if (Dtor == nullptr) { 2844 auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, 2845 /*isVarArg=*/false) 2846 ->getPointerTo(); 2847 Dtor = llvm::Constant::getNullValue(DtorTy); 2848 } 2849 if (!CGF) { 2850 auto *InitFunctionTy = 2851 llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false); 2852 std::string Name = getName({"__omp_threadprivate_init_", ""}); 2853 llvm::Function *InitFunction = CGM.CreateGlobalInitOrDestructFunction( 2854 InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction()); 2855 CodeGenFunction InitCGF(CGM); 2856 FunctionArgList ArgList; 2857 InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction, 2858 CGM.getTypes().arrangeNullaryFunction(), ArgList, 2859 Loc, Loc); 2860 emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 2861 InitCGF.FinishFunction(); 2862 return InitFunction; 2863 } 2864 emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 2865 } 2866 return nullptr; 2867 } 2868 2869 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD, 2870 llvm::GlobalVariable *Addr, 2871 bool PerformInit) { 2872 if (CGM.getLangOpts().OMPTargetTriples.empty() && 2873 !CGM.getLangOpts().OpenMPIsDevice) 2874 return false; 2875 Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 2876 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 2877 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 2878 (*Res == OMPDeclareTargetDeclAttr::MT_To && 2879 HasRequiresUnifiedSharedMemory)) 2880 return CGM.getLangOpts().OpenMPIsDevice; 2881 VD = VD->getDefinition(CGM.getContext()); 2882 if (VD && !DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second) 2883 return CGM.getLangOpts().OpenMPIsDevice; 2884 2885 QualType ASTTy = VD->getType(); 2886 2887 SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc(); 2888 // Produce the unique prefix to identify the new target regions. We use 2889 // the source location of the variable declaration which we know to not 2890 // conflict with any target region. 2891 unsigned DeviceID; 2892 unsigned FileID; 2893 unsigned Line; 2894 getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line); 2895 SmallString<128> Buffer, Out; 2896 { 2897 llvm::raw_svector_ostream OS(Buffer); 2898 OS << "__omp_offloading_" << llvm::format("_%x", DeviceID) 2899 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 2900 } 2901 2902 const Expr *Init = VD->getAnyInitializer(); 2903 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 2904 llvm::Constant *Ctor; 2905 llvm::Constant *ID; 2906 if (CGM.getLangOpts().OpenMPIsDevice) { 2907 // Generate function that re-emits the declaration's initializer into 2908 // the threadprivate copy of the variable VD 2909 CodeGenFunction CtorCGF(CGM); 2910 2911 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 2912 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2913 llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction( 2914 FTy, Twine(Buffer, "_ctor"), FI, Loc); 2915 auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF); 2916 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 2917 FunctionArgList(), Loc, Loc); 2918 auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF); 2919 CtorCGF.EmitAnyExprToMem(Init, 2920 Address(Addr, CGM.getContext().getDeclAlign(VD)), 2921 Init->getType().getQualifiers(), 2922 /*IsInitializer=*/true); 2923 CtorCGF.FinishFunction(); 2924 Ctor = Fn; 2925 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 2926 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor)); 2927 } else { 2928 Ctor = new llvm::GlobalVariable( 2929 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 2930 llvm::GlobalValue::PrivateLinkage, 2931 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor")); 2932 ID = Ctor; 2933 } 2934 2935 // Register the information for the entry associated with the constructor. 2936 Out.clear(); 2937 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 2938 DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor, 2939 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor); 2940 } 2941 if (VD->getType().isDestructedType() != QualType::DK_none) { 2942 llvm::Constant *Dtor; 2943 llvm::Constant *ID; 2944 if (CGM.getLangOpts().OpenMPIsDevice) { 2945 // Generate function that emits destructor call for the threadprivate 2946 // copy of the variable VD 2947 CodeGenFunction DtorCGF(CGM); 2948 2949 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 2950 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2951 llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction( 2952 FTy, Twine(Buffer, "_dtor"), FI, Loc); 2953 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 2954 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 2955 FunctionArgList(), Loc, Loc); 2956 // Create a scope with an artificial location for the body of this 2957 // function. 2958 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 2959 DtorCGF.emitDestroy(Address(Addr, CGM.getContext().getDeclAlign(VD)), 2960 ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()), 2961 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 2962 DtorCGF.FinishFunction(); 2963 Dtor = Fn; 2964 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 2965 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor)); 2966 } else { 2967 Dtor = new llvm::GlobalVariable( 2968 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 2969 llvm::GlobalValue::PrivateLinkage, 2970 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor")); 2971 ID = Dtor; 2972 } 2973 // Register the information for the entry associated with the destructor. 2974 Out.clear(); 2975 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 2976 DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor, 2977 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor); 2978 } 2979 return CGM.getLangOpts().OpenMPIsDevice; 2980 } 2981 2982 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF, 2983 QualType VarType, 2984 StringRef Name) { 2985 std::string Suffix = getName({"artificial", ""}); 2986 std::string CacheSuffix = getName({"cache", ""}); 2987 llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType); 2988 llvm::Value *GAddr = 2989 getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix)); 2990 llvm::Value *Args[] = { 2991 emitUpdateLocation(CGF, SourceLocation()), 2992 getThreadID(CGF, SourceLocation()), 2993 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy), 2994 CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy, 2995 /*isSigned=*/false), 2996 getOrCreateInternalVariable( 2997 CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))}; 2998 return Address( 2999 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3000 CGF.EmitRuntimeCall( 3001 createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args), 3002 VarLVType->getPointerTo(/*AddrSpace=*/0)), 3003 CGM.getPointerAlign()); 3004 } 3005 3006 void CGOpenMPRuntime::emitOMPIfClause(CodeGenFunction &CGF, const Expr *Cond, 3007 const RegionCodeGenTy &ThenGen, 3008 const RegionCodeGenTy &ElseGen) { 3009 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange()); 3010 3011 // If the condition constant folds and can be elided, try to avoid emitting 3012 // the condition and the dead arm of the if/else. 3013 bool CondConstant; 3014 if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) { 3015 if (CondConstant) 3016 ThenGen(CGF); 3017 else 3018 ElseGen(CGF); 3019 return; 3020 } 3021 3022 // Otherwise, the condition did not fold, or we couldn't elide it. Just 3023 // emit the conditional branch. 3024 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then"); 3025 llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else"); 3026 llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end"); 3027 CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0); 3028 3029 // Emit the 'then' code. 3030 CGF.EmitBlock(ThenBlock); 3031 ThenGen(CGF); 3032 CGF.EmitBranch(ContBlock); 3033 // Emit the 'else' code if present. 3034 // There is no need to emit line number for unconditional branch. 3035 (void)ApplyDebugLocation::CreateEmpty(CGF); 3036 CGF.EmitBlock(ElseBlock); 3037 ElseGen(CGF); 3038 // There is no need to emit line number for unconditional branch. 3039 (void)ApplyDebugLocation::CreateEmpty(CGF); 3040 CGF.EmitBranch(ContBlock); 3041 // Emit the continuation block for code after the if. 3042 CGF.EmitBlock(ContBlock, /*IsFinished=*/true); 3043 } 3044 3045 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc, 3046 llvm::Function *OutlinedFn, 3047 ArrayRef<llvm::Value *> CapturedVars, 3048 const Expr *IfCond) { 3049 if (!CGF.HaveInsertPoint()) 3050 return; 3051 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 3052 auto &&ThenGen = [OutlinedFn, CapturedVars, RTLoc](CodeGenFunction &CGF, 3053 PrePostActionTy &) { 3054 // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn); 3055 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 3056 llvm::Value *Args[] = { 3057 RTLoc, 3058 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 3059 CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())}; 3060 llvm::SmallVector<llvm::Value *, 16> RealArgs; 3061 RealArgs.append(std::begin(Args), std::end(Args)); 3062 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 3063 3064 llvm::FunctionCallee RTLFn = 3065 RT.createRuntimeFunction(OMPRTL__kmpc_fork_call); 3066 CGF.EmitRuntimeCall(RTLFn, RealArgs); 3067 }; 3068 auto &&ElseGen = [OutlinedFn, CapturedVars, RTLoc, Loc](CodeGenFunction &CGF, 3069 PrePostActionTy &) { 3070 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 3071 llvm::Value *ThreadID = RT.getThreadID(CGF, Loc); 3072 // Build calls: 3073 // __kmpc_serialized_parallel(&Loc, GTid); 3074 llvm::Value *Args[] = {RTLoc, ThreadID}; 3075 CGF.EmitRuntimeCall( 3076 RT.createRuntimeFunction(OMPRTL__kmpc_serialized_parallel), Args); 3077 3078 // OutlinedFn(>id, &zero, CapturedStruct); 3079 Address ZeroAddr = CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty, 3080 /*Name*/ ".zero.addr"); 3081 CGF.InitTempAlloca(ZeroAddr, CGF.Builder.getInt32(/*C*/ 0)); 3082 llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs; 3083 // ThreadId for serialized parallels is 0. 3084 OutlinedFnArgs.push_back(ZeroAddr.getPointer()); 3085 OutlinedFnArgs.push_back(ZeroAddr.getPointer()); 3086 OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end()); 3087 RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs); 3088 3089 // __kmpc_end_serialized_parallel(&Loc, GTid); 3090 llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID}; 3091 CGF.EmitRuntimeCall( 3092 RT.createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel), 3093 EndArgs); 3094 }; 3095 if (IfCond) { 3096 emitOMPIfClause(CGF, IfCond, ThenGen, ElseGen); 3097 } else { 3098 RegionCodeGenTy ThenRCG(ThenGen); 3099 ThenRCG(CGF); 3100 } 3101 } 3102 3103 // If we're inside an (outlined) parallel region, use the region info's 3104 // thread-ID variable (it is passed in a first argument of the outlined function 3105 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in 3106 // regular serial code region, get thread ID by calling kmp_int32 3107 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and 3108 // return the address of that temp. 3109 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF, 3110 SourceLocation Loc) { 3111 if (auto *OMPRegionInfo = 3112 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 3113 if (OMPRegionInfo->getThreadIDVariable()) 3114 return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(); 3115 3116 llvm::Value *ThreadID = getThreadID(CGF, Loc); 3117 QualType Int32Ty = 3118 CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true); 3119 Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp."); 3120 CGF.EmitStoreOfScalar(ThreadID, 3121 CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty)); 3122 3123 return ThreadIDTemp; 3124 } 3125 3126 llvm::Constant *CGOpenMPRuntime::getOrCreateInternalVariable( 3127 llvm::Type *Ty, const llvm::Twine &Name, unsigned AddressSpace) { 3128 SmallString<256> Buffer; 3129 llvm::raw_svector_ostream Out(Buffer); 3130 Out << Name; 3131 StringRef RuntimeName = Out.str(); 3132 auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first; 3133 if (Elem.second) { 3134 assert(Elem.second->getType()->getPointerElementType() == Ty && 3135 "OMP internal variable has different type than requested"); 3136 return &*Elem.second; 3137 } 3138 3139 return Elem.second = new llvm::GlobalVariable( 3140 CGM.getModule(), Ty, /*IsConstant*/ false, 3141 llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty), 3142 Elem.first(), /*InsertBefore=*/nullptr, 3143 llvm::GlobalValue::NotThreadLocal, AddressSpace); 3144 } 3145 3146 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) { 3147 std::string Prefix = Twine("gomp_critical_user_", CriticalName).str(); 3148 std::string Name = getName({Prefix, "var"}); 3149 return getOrCreateInternalVariable(KmpCriticalNameTy, Name); 3150 } 3151 3152 namespace { 3153 /// Common pre(post)-action for different OpenMP constructs. 3154 class CommonActionTy final : public PrePostActionTy { 3155 llvm::FunctionCallee EnterCallee; 3156 ArrayRef<llvm::Value *> EnterArgs; 3157 llvm::FunctionCallee ExitCallee; 3158 ArrayRef<llvm::Value *> ExitArgs; 3159 bool Conditional; 3160 llvm::BasicBlock *ContBlock = nullptr; 3161 3162 public: 3163 CommonActionTy(llvm::FunctionCallee EnterCallee, 3164 ArrayRef<llvm::Value *> EnterArgs, 3165 llvm::FunctionCallee ExitCallee, 3166 ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false) 3167 : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee), 3168 ExitArgs(ExitArgs), Conditional(Conditional) {} 3169 void Enter(CodeGenFunction &CGF) override { 3170 llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs); 3171 if (Conditional) { 3172 llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes); 3173 auto *ThenBlock = CGF.createBasicBlock("omp_if.then"); 3174 ContBlock = CGF.createBasicBlock("omp_if.end"); 3175 // Generate the branch (If-stmt) 3176 CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock); 3177 CGF.EmitBlock(ThenBlock); 3178 } 3179 } 3180 void Done(CodeGenFunction &CGF) { 3181 // Emit the rest of blocks/branches 3182 CGF.EmitBranch(ContBlock); 3183 CGF.EmitBlock(ContBlock, true); 3184 } 3185 void Exit(CodeGenFunction &CGF) override { 3186 CGF.EmitRuntimeCall(ExitCallee, ExitArgs); 3187 } 3188 }; 3189 } // anonymous namespace 3190 3191 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF, 3192 StringRef CriticalName, 3193 const RegionCodeGenTy &CriticalOpGen, 3194 SourceLocation Loc, const Expr *Hint) { 3195 // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]); 3196 // CriticalOpGen(); 3197 // __kmpc_end_critical(ident_t *, gtid, Lock); 3198 // Prepare arguments and build a call to __kmpc_critical 3199 if (!CGF.HaveInsertPoint()) 3200 return; 3201 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3202 getCriticalRegionLock(CriticalName)}; 3203 llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args), 3204 std::end(Args)); 3205 if (Hint) { 3206 EnterArgs.push_back(CGF.Builder.CreateIntCast( 3207 CGF.EmitScalarExpr(Hint), CGM.IntPtrTy, /*isSigned=*/false)); 3208 } 3209 CommonActionTy Action( 3210 createRuntimeFunction(Hint ? OMPRTL__kmpc_critical_with_hint 3211 : OMPRTL__kmpc_critical), 3212 EnterArgs, createRuntimeFunction(OMPRTL__kmpc_end_critical), Args); 3213 CriticalOpGen.setAction(Action); 3214 emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen); 3215 } 3216 3217 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF, 3218 const RegionCodeGenTy &MasterOpGen, 3219 SourceLocation Loc) { 3220 if (!CGF.HaveInsertPoint()) 3221 return; 3222 // if(__kmpc_master(ident_t *, gtid)) { 3223 // MasterOpGen(); 3224 // __kmpc_end_master(ident_t *, gtid); 3225 // } 3226 // Prepare arguments and build a call to __kmpc_master 3227 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3228 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_master), Args, 3229 createRuntimeFunction(OMPRTL__kmpc_end_master), Args, 3230 /*Conditional=*/true); 3231 MasterOpGen.setAction(Action); 3232 emitInlinedDirective(CGF, OMPD_master, MasterOpGen); 3233 Action.Done(CGF); 3234 } 3235 3236 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 3237 SourceLocation Loc) { 3238 if (!CGF.HaveInsertPoint()) 3239 return; 3240 // Build call __kmpc_omp_taskyield(loc, thread_id, 0); 3241 llvm::Value *Args[] = { 3242 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3243 llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)}; 3244 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield), Args); 3245 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 3246 Region->emitUntiedSwitch(CGF); 3247 } 3248 3249 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF, 3250 const RegionCodeGenTy &TaskgroupOpGen, 3251 SourceLocation Loc) { 3252 if (!CGF.HaveInsertPoint()) 3253 return; 3254 // __kmpc_taskgroup(ident_t *, gtid); 3255 // TaskgroupOpGen(); 3256 // __kmpc_end_taskgroup(ident_t *, gtid); 3257 // Prepare arguments and build a call to __kmpc_taskgroup 3258 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3259 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_taskgroup), Args, 3260 createRuntimeFunction(OMPRTL__kmpc_end_taskgroup), 3261 Args); 3262 TaskgroupOpGen.setAction(Action); 3263 emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen); 3264 } 3265 3266 /// Given an array of pointers to variables, project the address of a 3267 /// given variable. 3268 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array, 3269 unsigned Index, const VarDecl *Var) { 3270 // Pull out the pointer to the variable. 3271 Address PtrAddr = CGF.Builder.CreateConstArrayGEP(Array, Index); 3272 llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr); 3273 3274 Address Addr = Address(Ptr, CGF.getContext().getDeclAlign(Var)); 3275 Addr = CGF.Builder.CreateElementBitCast( 3276 Addr, CGF.ConvertTypeForMem(Var->getType())); 3277 return Addr; 3278 } 3279 3280 static llvm::Value *emitCopyprivateCopyFunction( 3281 CodeGenModule &CGM, llvm::Type *ArgsType, 3282 ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs, 3283 ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps, 3284 SourceLocation Loc) { 3285 ASTContext &C = CGM.getContext(); 3286 // void copy_func(void *LHSArg, void *RHSArg); 3287 FunctionArgList Args; 3288 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 3289 ImplicitParamDecl::Other); 3290 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 3291 ImplicitParamDecl::Other); 3292 Args.push_back(&LHSArg); 3293 Args.push_back(&RHSArg); 3294 const auto &CGFI = 3295 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 3296 std::string Name = 3297 CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"}); 3298 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 3299 llvm::GlobalValue::InternalLinkage, Name, 3300 &CGM.getModule()); 3301 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 3302 Fn->setDoesNotRecurse(); 3303 CodeGenFunction CGF(CGM); 3304 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 3305 // Dest = (void*[n])(LHSArg); 3306 // Src = (void*[n])(RHSArg); 3307 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3308 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 3309 ArgsType), CGF.getPointerAlign()); 3310 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3311 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 3312 ArgsType), CGF.getPointerAlign()); 3313 // *(Type0*)Dst[0] = *(Type0*)Src[0]; 3314 // *(Type1*)Dst[1] = *(Type1*)Src[1]; 3315 // ... 3316 // *(Typen*)Dst[n] = *(Typen*)Src[n]; 3317 for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) { 3318 const auto *DestVar = 3319 cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl()); 3320 Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar); 3321 3322 const auto *SrcVar = 3323 cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl()); 3324 Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar); 3325 3326 const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl(); 3327 QualType Type = VD->getType(); 3328 CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]); 3329 } 3330 CGF.FinishFunction(); 3331 return Fn; 3332 } 3333 3334 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF, 3335 const RegionCodeGenTy &SingleOpGen, 3336 SourceLocation Loc, 3337 ArrayRef<const Expr *> CopyprivateVars, 3338 ArrayRef<const Expr *> SrcExprs, 3339 ArrayRef<const Expr *> DstExprs, 3340 ArrayRef<const Expr *> AssignmentOps) { 3341 if (!CGF.HaveInsertPoint()) 3342 return; 3343 assert(CopyprivateVars.size() == SrcExprs.size() && 3344 CopyprivateVars.size() == DstExprs.size() && 3345 CopyprivateVars.size() == AssignmentOps.size()); 3346 ASTContext &C = CGM.getContext(); 3347 // int32 did_it = 0; 3348 // if(__kmpc_single(ident_t *, gtid)) { 3349 // SingleOpGen(); 3350 // __kmpc_end_single(ident_t *, gtid); 3351 // did_it = 1; 3352 // } 3353 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 3354 // <copy_func>, did_it); 3355 3356 Address DidIt = Address::invalid(); 3357 if (!CopyprivateVars.empty()) { 3358 // int32 did_it = 0; 3359 QualType KmpInt32Ty = 3360 C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 3361 DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it"); 3362 CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt); 3363 } 3364 // Prepare arguments and build a call to __kmpc_single 3365 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3366 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_single), Args, 3367 createRuntimeFunction(OMPRTL__kmpc_end_single), Args, 3368 /*Conditional=*/true); 3369 SingleOpGen.setAction(Action); 3370 emitInlinedDirective(CGF, OMPD_single, SingleOpGen); 3371 if (DidIt.isValid()) { 3372 // did_it = 1; 3373 CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt); 3374 } 3375 Action.Done(CGF); 3376 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 3377 // <copy_func>, did_it); 3378 if (DidIt.isValid()) { 3379 llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size()); 3380 QualType CopyprivateArrayTy = C.getConstantArrayType( 3381 C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 3382 /*IndexTypeQuals=*/0); 3383 // Create a list of all private variables for copyprivate. 3384 Address CopyprivateList = 3385 CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list"); 3386 for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) { 3387 Address Elem = CGF.Builder.CreateConstArrayGEP(CopyprivateList, I); 3388 CGF.Builder.CreateStore( 3389 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3390 CGF.EmitLValue(CopyprivateVars[I]).getPointer(), CGF.VoidPtrTy), 3391 Elem); 3392 } 3393 // Build function that copies private values from single region to all other 3394 // threads in the corresponding parallel region. 3395 llvm::Value *CpyFn = emitCopyprivateCopyFunction( 3396 CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(), 3397 CopyprivateVars, SrcExprs, DstExprs, AssignmentOps, Loc); 3398 llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy); 3399 Address CL = 3400 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList, 3401 CGF.VoidPtrTy); 3402 llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt); 3403 llvm::Value *Args[] = { 3404 emitUpdateLocation(CGF, Loc), // ident_t *<loc> 3405 getThreadID(CGF, Loc), // i32 <gtid> 3406 BufSize, // size_t <buf_size> 3407 CL.getPointer(), // void *<copyprivate list> 3408 CpyFn, // void (*) (void *, void *) <copy_func> 3409 DidItVal // i32 did_it 3410 }; 3411 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_copyprivate), Args); 3412 } 3413 } 3414 3415 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF, 3416 const RegionCodeGenTy &OrderedOpGen, 3417 SourceLocation Loc, bool IsThreads) { 3418 if (!CGF.HaveInsertPoint()) 3419 return; 3420 // __kmpc_ordered(ident_t *, gtid); 3421 // OrderedOpGen(); 3422 // __kmpc_end_ordered(ident_t *, gtid); 3423 // Prepare arguments and build a call to __kmpc_ordered 3424 if (IsThreads) { 3425 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3426 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_ordered), Args, 3427 createRuntimeFunction(OMPRTL__kmpc_end_ordered), 3428 Args); 3429 OrderedOpGen.setAction(Action); 3430 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 3431 return; 3432 } 3433 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 3434 } 3435 3436 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) { 3437 unsigned Flags; 3438 if (Kind == OMPD_for) 3439 Flags = OMP_IDENT_BARRIER_IMPL_FOR; 3440 else if (Kind == OMPD_sections) 3441 Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS; 3442 else if (Kind == OMPD_single) 3443 Flags = OMP_IDENT_BARRIER_IMPL_SINGLE; 3444 else if (Kind == OMPD_barrier) 3445 Flags = OMP_IDENT_BARRIER_EXPL; 3446 else 3447 Flags = OMP_IDENT_BARRIER_IMPL; 3448 return Flags; 3449 } 3450 3451 void CGOpenMPRuntime::getDefaultScheduleAndChunk( 3452 CodeGenFunction &CGF, const OMPLoopDirective &S, 3453 OpenMPScheduleClauseKind &ScheduleKind, const Expr *&ChunkExpr) const { 3454 // Check if the loop directive is actually a doacross loop directive. In this 3455 // case choose static, 1 schedule. 3456 if (llvm::any_of( 3457 S.getClausesOfKind<OMPOrderedClause>(), 3458 [](const OMPOrderedClause *C) { return C->getNumForLoops(); })) { 3459 ScheduleKind = OMPC_SCHEDULE_static; 3460 // Chunk size is 1 in this case. 3461 llvm::APInt ChunkSize(32, 1); 3462 ChunkExpr = IntegerLiteral::Create( 3463 CGF.getContext(), ChunkSize, 3464 CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/0), 3465 SourceLocation()); 3466 } 3467 } 3468 3469 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc, 3470 OpenMPDirectiveKind Kind, bool EmitChecks, 3471 bool ForceSimpleCall) { 3472 if (!CGF.HaveInsertPoint()) 3473 return; 3474 // Build call __kmpc_cancel_barrier(loc, thread_id); 3475 // Build call __kmpc_barrier(loc, thread_id); 3476 unsigned Flags = getDefaultFlagsForBarriers(Kind); 3477 // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc, 3478 // thread_id); 3479 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags), 3480 getThreadID(CGF, Loc)}; 3481 if (auto *OMPRegionInfo = 3482 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 3483 if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) { 3484 llvm::Value *Result = CGF.EmitRuntimeCall( 3485 createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args); 3486 if (EmitChecks) { 3487 // if (__kmpc_cancel_barrier()) { 3488 // exit from construct; 3489 // } 3490 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 3491 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 3492 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 3493 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 3494 CGF.EmitBlock(ExitBB); 3495 // exit from construct; 3496 CodeGenFunction::JumpDest CancelDestination = 3497 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 3498 CGF.EmitBranchThroughCleanup(CancelDestination); 3499 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 3500 } 3501 return; 3502 } 3503 } 3504 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_barrier), Args); 3505 } 3506 3507 /// Map the OpenMP loop schedule to the runtime enumeration. 3508 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind, 3509 bool Chunked, bool Ordered) { 3510 switch (ScheduleKind) { 3511 case OMPC_SCHEDULE_static: 3512 return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked) 3513 : (Ordered ? OMP_ord_static : OMP_sch_static); 3514 case OMPC_SCHEDULE_dynamic: 3515 return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked; 3516 case OMPC_SCHEDULE_guided: 3517 return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked; 3518 case OMPC_SCHEDULE_runtime: 3519 return Ordered ? OMP_ord_runtime : OMP_sch_runtime; 3520 case OMPC_SCHEDULE_auto: 3521 return Ordered ? OMP_ord_auto : OMP_sch_auto; 3522 case OMPC_SCHEDULE_unknown: 3523 assert(!Chunked && "chunk was specified but schedule kind not known"); 3524 return Ordered ? OMP_ord_static : OMP_sch_static; 3525 } 3526 llvm_unreachable("Unexpected runtime schedule"); 3527 } 3528 3529 /// Map the OpenMP distribute schedule to the runtime enumeration. 3530 static OpenMPSchedType 3531 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) { 3532 // only static is allowed for dist_schedule 3533 return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static; 3534 } 3535 3536 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind, 3537 bool Chunked) const { 3538 OpenMPSchedType Schedule = 3539 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 3540 return Schedule == OMP_sch_static; 3541 } 3542 3543 bool CGOpenMPRuntime::isStaticNonchunked( 3544 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 3545 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 3546 return Schedule == OMP_dist_sch_static; 3547 } 3548 3549 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind, 3550 bool Chunked) const { 3551 OpenMPSchedType Schedule = 3552 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 3553 return Schedule == OMP_sch_static_chunked; 3554 } 3555 3556 bool CGOpenMPRuntime::isStaticChunked( 3557 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 3558 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 3559 return Schedule == OMP_dist_sch_static_chunked; 3560 } 3561 3562 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const { 3563 OpenMPSchedType Schedule = 3564 getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false); 3565 assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here"); 3566 return Schedule != OMP_sch_static; 3567 } 3568 3569 static int addMonoNonMonoModifier(CodeGenModule &CGM, OpenMPSchedType Schedule, 3570 OpenMPScheduleClauseModifier M1, 3571 OpenMPScheduleClauseModifier M2) { 3572 int Modifier = 0; 3573 switch (M1) { 3574 case OMPC_SCHEDULE_MODIFIER_monotonic: 3575 Modifier = OMP_sch_modifier_monotonic; 3576 break; 3577 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 3578 Modifier = OMP_sch_modifier_nonmonotonic; 3579 break; 3580 case OMPC_SCHEDULE_MODIFIER_simd: 3581 if (Schedule == OMP_sch_static_chunked) 3582 Schedule = OMP_sch_static_balanced_chunked; 3583 break; 3584 case OMPC_SCHEDULE_MODIFIER_last: 3585 case OMPC_SCHEDULE_MODIFIER_unknown: 3586 break; 3587 } 3588 switch (M2) { 3589 case OMPC_SCHEDULE_MODIFIER_monotonic: 3590 Modifier = OMP_sch_modifier_monotonic; 3591 break; 3592 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 3593 Modifier = OMP_sch_modifier_nonmonotonic; 3594 break; 3595 case OMPC_SCHEDULE_MODIFIER_simd: 3596 if (Schedule == OMP_sch_static_chunked) 3597 Schedule = OMP_sch_static_balanced_chunked; 3598 break; 3599 case OMPC_SCHEDULE_MODIFIER_last: 3600 case OMPC_SCHEDULE_MODIFIER_unknown: 3601 break; 3602 } 3603 // OpenMP 5.0, 2.9.2 Worksharing-Loop Construct, Desription. 3604 // If the static schedule kind is specified or if the ordered clause is 3605 // specified, and if the nonmonotonic modifier is not specified, the effect is 3606 // as if the monotonic modifier is specified. Otherwise, unless the monotonic 3607 // modifier is specified, the effect is as if the nonmonotonic modifier is 3608 // specified. 3609 if (CGM.getLangOpts().OpenMP >= 50 && Modifier == 0) { 3610 if (!(Schedule == OMP_sch_static_chunked || Schedule == OMP_sch_static || 3611 Schedule == OMP_sch_static_balanced_chunked || 3612 Schedule == OMP_ord_static_chunked || Schedule == OMP_ord_static)) 3613 Modifier = OMP_sch_modifier_nonmonotonic; 3614 } 3615 return Schedule | Modifier; 3616 } 3617 3618 void CGOpenMPRuntime::emitForDispatchInit( 3619 CodeGenFunction &CGF, SourceLocation Loc, 3620 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 3621 bool Ordered, const DispatchRTInput &DispatchValues) { 3622 if (!CGF.HaveInsertPoint()) 3623 return; 3624 OpenMPSchedType Schedule = getRuntimeSchedule( 3625 ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered); 3626 assert(Ordered || 3627 (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked && 3628 Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked && 3629 Schedule != OMP_sch_static_balanced_chunked)); 3630 // Call __kmpc_dispatch_init( 3631 // ident_t *loc, kmp_int32 tid, kmp_int32 schedule, 3632 // kmp_int[32|64] lower, kmp_int[32|64] upper, 3633 // kmp_int[32|64] stride, kmp_int[32|64] chunk); 3634 3635 // If the Chunk was not specified in the clause - use default value 1. 3636 llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk 3637 : CGF.Builder.getIntN(IVSize, 1); 3638 llvm::Value *Args[] = { 3639 emitUpdateLocation(CGF, Loc), 3640 getThreadID(CGF, Loc), 3641 CGF.Builder.getInt32(addMonoNonMonoModifier( 3642 CGM, Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type 3643 DispatchValues.LB, // Lower 3644 DispatchValues.UB, // Upper 3645 CGF.Builder.getIntN(IVSize, 1), // Stride 3646 Chunk // Chunk 3647 }; 3648 CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args); 3649 } 3650 3651 static void emitForStaticInitCall( 3652 CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId, 3653 llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule, 3654 OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2, 3655 const CGOpenMPRuntime::StaticRTInput &Values) { 3656 if (!CGF.HaveInsertPoint()) 3657 return; 3658 3659 assert(!Values.Ordered); 3660 assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked || 3661 Schedule == OMP_sch_static_balanced_chunked || 3662 Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked || 3663 Schedule == OMP_dist_sch_static || 3664 Schedule == OMP_dist_sch_static_chunked); 3665 3666 // Call __kmpc_for_static_init( 3667 // ident_t *loc, kmp_int32 tid, kmp_int32 schedtype, 3668 // kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower, 3669 // kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride, 3670 // kmp_int[32|64] incr, kmp_int[32|64] chunk); 3671 llvm::Value *Chunk = Values.Chunk; 3672 if (Chunk == nullptr) { 3673 assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static || 3674 Schedule == OMP_dist_sch_static) && 3675 "expected static non-chunked schedule"); 3676 // If the Chunk was not specified in the clause - use default value 1. 3677 Chunk = CGF.Builder.getIntN(Values.IVSize, 1); 3678 } else { 3679 assert((Schedule == OMP_sch_static_chunked || 3680 Schedule == OMP_sch_static_balanced_chunked || 3681 Schedule == OMP_ord_static_chunked || 3682 Schedule == OMP_dist_sch_static_chunked) && 3683 "expected static chunked schedule"); 3684 } 3685 llvm::Value *Args[] = { 3686 UpdateLocation, 3687 ThreadId, 3688 CGF.Builder.getInt32(addMonoNonMonoModifier(CGF.CGM, Schedule, M1, 3689 M2)), // Schedule type 3690 Values.IL.getPointer(), // &isLastIter 3691 Values.LB.getPointer(), // &LB 3692 Values.UB.getPointer(), // &UB 3693 Values.ST.getPointer(), // &Stride 3694 CGF.Builder.getIntN(Values.IVSize, 1), // Incr 3695 Chunk // Chunk 3696 }; 3697 CGF.EmitRuntimeCall(ForStaticInitFunction, Args); 3698 } 3699 3700 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF, 3701 SourceLocation Loc, 3702 OpenMPDirectiveKind DKind, 3703 const OpenMPScheduleTy &ScheduleKind, 3704 const StaticRTInput &Values) { 3705 OpenMPSchedType ScheduleNum = getRuntimeSchedule( 3706 ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered); 3707 assert(isOpenMPWorksharingDirective(DKind) && 3708 "Expected loop-based or sections-based directive."); 3709 llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc, 3710 isOpenMPLoopDirective(DKind) 3711 ? OMP_IDENT_WORK_LOOP 3712 : OMP_IDENT_WORK_SECTIONS); 3713 llvm::Value *ThreadId = getThreadID(CGF, Loc); 3714 llvm::FunctionCallee StaticInitFunction = 3715 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 3716 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 3717 ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values); 3718 } 3719 3720 void CGOpenMPRuntime::emitDistributeStaticInit( 3721 CodeGenFunction &CGF, SourceLocation Loc, 3722 OpenMPDistScheduleClauseKind SchedKind, 3723 const CGOpenMPRuntime::StaticRTInput &Values) { 3724 OpenMPSchedType ScheduleNum = 3725 getRuntimeSchedule(SchedKind, Values.Chunk != nullptr); 3726 llvm::Value *UpdatedLocation = 3727 emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE); 3728 llvm::Value *ThreadId = getThreadID(CGF, Loc); 3729 llvm::FunctionCallee StaticInitFunction = 3730 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 3731 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 3732 ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown, 3733 OMPC_SCHEDULE_MODIFIER_unknown, Values); 3734 } 3735 3736 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF, 3737 SourceLocation Loc, 3738 OpenMPDirectiveKind DKind) { 3739 if (!CGF.HaveInsertPoint()) 3740 return; 3741 // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid); 3742 llvm::Value *Args[] = { 3743 emitUpdateLocation(CGF, Loc, 3744 isOpenMPDistributeDirective(DKind) 3745 ? OMP_IDENT_WORK_DISTRIBUTE 3746 : isOpenMPLoopDirective(DKind) 3747 ? OMP_IDENT_WORK_LOOP 3748 : OMP_IDENT_WORK_SECTIONS), 3749 getThreadID(CGF, Loc)}; 3750 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini), 3751 Args); 3752 } 3753 3754 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 3755 SourceLocation Loc, 3756 unsigned IVSize, 3757 bool IVSigned) { 3758 if (!CGF.HaveInsertPoint()) 3759 return; 3760 // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid); 3761 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3762 CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args); 3763 } 3764 3765 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF, 3766 SourceLocation Loc, unsigned IVSize, 3767 bool IVSigned, Address IL, 3768 Address LB, Address UB, 3769 Address ST) { 3770 // Call __kmpc_dispatch_next( 3771 // ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter, 3772 // kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper, 3773 // kmp_int[32|64] *p_stride); 3774 llvm::Value *Args[] = { 3775 emitUpdateLocation(CGF, Loc), 3776 getThreadID(CGF, Loc), 3777 IL.getPointer(), // &isLastIter 3778 LB.getPointer(), // &Lower 3779 UB.getPointer(), // &Upper 3780 ST.getPointer() // &Stride 3781 }; 3782 llvm::Value *Call = 3783 CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args); 3784 return CGF.EmitScalarConversion( 3785 Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1), 3786 CGF.getContext().BoolTy, Loc); 3787 } 3788 3789 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 3790 llvm::Value *NumThreads, 3791 SourceLocation Loc) { 3792 if (!CGF.HaveInsertPoint()) 3793 return; 3794 // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads) 3795 llvm::Value *Args[] = { 3796 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3797 CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)}; 3798 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads), 3799 Args); 3800 } 3801 3802 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF, 3803 OpenMPProcBindClauseKind ProcBind, 3804 SourceLocation Loc) { 3805 if (!CGF.HaveInsertPoint()) 3806 return; 3807 // Constants for proc bind value accepted by the runtime. 3808 enum ProcBindTy { 3809 ProcBindFalse = 0, 3810 ProcBindTrue, 3811 ProcBindMaster, 3812 ProcBindClose, 3813 ProcBindSpread, 3814 ProcBindIntel, 3815 ProcBindDefault 3816 } RuntimeProcBind; 3817 switch (ProcBind) { 3818 case OMPC_PROC_BIND_master: 3819 RuntimeProcBind = ProcBindMaster; 3820 break; 3821 case OMPC_PROC_BIND_close: 3822 RuntimeProcBind = ProcBindClose; 3823 break; 3824 case OMPC_PROC_BIND_spread: 3825 RuntimeProcBind = ProcBindSpread; 3826 break; 3827 case OMPC_PROC_BIND_unknown: 3828 llvm_unreachable("Unsupported proc_bind value."); 3829 } 3830 // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind) 3831 llvm::Value *Args[] = { 3832 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3833 llvm::ConstantInt::get(CGM.IntTy, RuntimeProcBind, /*isSigned=*/true)}; 3834 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_proc_bind), Args); 3835 } 3836 3837 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>, 3838 SourceLocation Loc) { 3839 if (!CGF.HaveInsertPoint()) 3840 return; 3841 // Build call void __kmpc_flush(ident_t *loc) 3842 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush), 3843 emitUpdateLocation(CGF, Loc)); 3844 } 3845 3846 namespace { 3847 /// Indexes of fields for type kmp_task_t. 3848 enum KmpTaskTFields { 3849 /// List of shared variables. 3850 KmpTaskTShareds, 3851 /// Task routine. 3852 KmpTaskTRoutine, 3853 /// Partition id for the untied tasks. 3854 KmpTaskTPartId, 3855 /// Function with call of destructors for private variables. 3856 Data1, 3857 /// Task priority. 3858 Data2, 3859 /// (Taskloops only) Lower bound. 3860 KmpTaskTLowerBound, 3861 /// (Taskloops only) Upper bound. 3862 KmpTaskTUpperBound, 3863 /// (Taskloops only) Stride. 3864 KmpTaskTStride, 3865 /// (Taskloops only) Is last iteration flag. 3866 KmpTaskTLastIter, 3867 /// (Taskloops only) Reduction data. 3868 KmpTaskTReductions, 3869 }; 3870 } // anonymous namespace 3871 3872 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const { 3873 return OffloadEntriesTargetRegion.empty() && 3874 OffloadEntriesDeviceGlobalVar.empty(); 3875 } 3876 3877 /// Initialize target region entry. 3878 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3879 initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3880 StringRef ParentName, unsigned LineNum, 3881 unsigned Order) { 3882 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3883 "only required for the device " 3884 "code generation."); 3885 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = 3886 OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr, 3887 OMPTargetRegionEntryTargetRegion); 3888 ++OffloadingEntriesNum; 3889 } 3890 3891 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3892 registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3893 StringRef ParentName, unsigned LineNum, 3894 llvm::Constant *Addr, llvm::Constant *ID, 3895 OMPTargetRegionEntryKind Flags) { 3896 // If we are emitting code for a target, the entry is already initialized, 3897 // only has to be registered. 3898 if (CGM.getLangOpts().OpenMPIsDevice) { 3899 if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) { 3900 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3901 DiagnosticsEngine::Error, 3902 "Unable to find target region on line '%0' in the device code."); 3903 CGM.getDiags().Report(DiagID) << LineNum; 3904 return; 3905 } 3906 auto &Entry = 3907 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum]; 3908 assert(Entry.isValid() && "Entry not initialized!"); 3909 Entry.setAddress(Addr); 3910 Entry.setID(ID); 3911 Entry.setFlags(Flags); 3912 } else { 3913 OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags); 3914 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry; 3915 ++OffloadingEntriesNum; 3916 } 3917 } 3918 3919 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo( 3920 unsigned DeviceID, unsigned FileID, StringRef ParentName, 3921 unsigned LineNum) const { 3922 auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID); 3923 if (PerDevice == OffloadEntriesTargetRegion.end()) 3924 return false; 3925 auto PerFile = PerDevice->second.find(FileID); 3926 if (PerFile == PerDevice->second.end()) 3927 return false; 3928 auto PerParentName = PerFile->second.find(ParentName); 3929 if (PerParentName == PerFile->second.end()) 3930 return false; 3931 auto PerLine = PerParentName->second.find(LineNum); 3932 if (PerLine == PerParentName->second.end()) 3933 return false; 3934 // Fail if this entry is already registered. 3935 if (PerLine->second.getAddress() || PerLine->second.getID()) 3936 return false; 3937 return true; 3938 } 3939 3940 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo( 3941 const OffloadTargetRegionEntryInfoActTy &Action) { 3942 // Scan all target region entries and perform the provided action. 3943 for (const auto &D : OffloadEntriesTargetRegion) 3944 for (const auto &F : D.second) 3945 for (const auto &P : F.second) 3946 for (const auto &L : P.second) 3947 Action(D.first, F.first, P.first(), L.first, L.second); 3948 } 3949 3950 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3951 initializeDeviceGlobalVarEntryInfo(StringRef Name, 3952 OMPTargetGlobalVarEntryKind Flags, 3953 unsigned Order) { 3954 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3955 "only required for the device " 3956 "code generation."); 3957 OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags); 3958 ++OffloadingEntriesNum; 3959 } 3960 3961 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3962 registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr, 3963 CharUnits VarSize, 3964 OMPTargetGlobalVarEntryKind Flags, 3965 llvm::GlobalValue::LinkageTypes Linkage) { 3966 if (CGM.getLangOpts().OpenMPIsDevice) { 3967 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 3968 assert(Entry.isValid() && Entry.getFlags() == Flags && 3969 "Entry not initialized!"); 3970 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 3971 "Resetting with the new address."); 3972 if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) { 3973 if (Entry.getVarSize().isZero()) { 3974 Entry.setVarSize(VarSize); 3975 Entry.setLinkage(Linkage); 3976 } 3977 return; 3978 } 3979 Entry.setVarSize(VarSize); 3980 Entry.setLinkage(Linkage); 3981 Entry.setAddress(Addr); 3982 } else { 3983 if (hasDeviceGlobalVarEntryInfo(VarName)) { 3984 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 3985 assert(Entry.isValid() && Entry.getFlags() == Flags && 3986 "Entry not initialized!"); 3987 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 3988 "Resetting with the new address."); 3989 if (Entry.getVarSize().isZero()) { 3990 Entry.setVarSize(VarSize); 3991 Entry.setLinkage(Linkage); 3992 } 3993 return; 3994 } 3995 OffloadEntriesDeviceGlobalVar.try_emplace( 3996 VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage); 3997 ++OffloadingEntriesNum; 3998 } 3999 } 4000 4001 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 4002 actOnDeviceGlobalVarEntriesInfo( 4003 const OffloadDeviceGlobalVarEntryInfoActTy &Action) { 4004 // Scan all target region entries and perform the provided action. 4005 for (const auto &E : OffloadEntriesDeviceGlobalVar) 4006 Action(E.getKey(), E.getValue()); 4007 } 4008 4009 llvm::Function * 4010 CGOpenMPRuntime::createOffloadingBinaryDescriptorRegistration() { 4011 // If we don't have entries or if we are emitting code for the device, we 4012 // don't need to do anything. 4013 if (CGM.getLangOpts().OpenMPIsDevice || OffloadEntriesInfoManager.empty()) 4014 return nullptr; 4015 4016 llvm::Module &M = CGM.getModule(); 4017 ASTContext &C = CGM.getContext(); 4018 4019 // Get list of devices we care about 4020 const std::vector<llvm::Triple> &Devices = CGM.getLangOpts().OMPTargetTriples; 4021 4022 // We should be creating an offloading descriptor only if there are devices 4023 // specified. 4024 assert(!Devices.empty() && "No OpenMP offloading devices??"); 4025 4026 // Create the external variables that will point to the begin and end of the 4027 // host entries section. These will be defined by the linker. 4028 llvm::Type *OffloadEntryTy = 4029 CGM.getTypes().ConvertTypeForMem(getTgtOffloadEntryQTy()); 4030 auto *HostEntriesBegin = new llvm::GlobalVariable( 4031 M, OffloadEntryTy, /*isConstant=*/true, 4032 llvm::GlobalValue::ExternalLinkage, /*Initializer=*/nullptr, 4033 "__start_omp_offloading_entries"); 4034 HostEntriesBegin->setVisibility(llvm::GlobalValue::HiddenVisibility); 4035 auto *HostEntriesEnd = new llvm::GlobalVariable( 4036 M, OffloadEntryTy, /*isConstant=*/true, 4037 llvm::GlobalValue::ExternalLinkage, 4038 /*Initializer=*/nullptr, "__stop_omp_offloading_entries"); 4039 HostEntriesEnd->setVisibility(llvm::GlobalValue::HiddenVisibility); 4040 4041 // Create all device images 4042 auto *DeviceImageTy = cast<llvm::StructType>( 4043 CGM.getTypes().ConvertTypeForMem(getTgtDeviceImageQTy())); 4044 ConstantInitBuilder DeviceImagesBuilder(CGM); 4045 ConstantArrayBuilder DeviceImagesEntries = 4046 DeviceImagesBuilder.beginArray(DeviceImageTy); 4047 4048 for (const llvm::Triple &Device : Devices) { 4049 StringRef T = Device.getTriple(); 4050 std::string BeginName = getName({"omp_offloading", "img_start", ""}); 4051 auto *ImgBegin = new llvm::GlobalVariable( 4052 M, CGM.Int8Ty, /*isConstant=*/true, 4053 llvm::GlobalValue::ExternalWeakLinkage, 4054 /*Initializer=*/nullptr, Twine(BeginName).concat(T)); 4055 std::string EndName = getName({"omp_offloading", "img_end", ""}); 4056 auto *ImgEnd = new llvm::GlobalVariable( 4057 M, CGM.Int8Ty, /*isConstant=*/true, 4058 llvm::GlobalValue::ExternalWeakLinkage, 4059 /*Initializer=*/nullptr, Twine(EndName).concat(T)); 4060 4061 llvm::Constant *Data[] = {ImgBegin, ImgEnd, HostEntriesBegin, 4062 HostEntriesEnd}; 4063 createConstantGlobalStructAndAddToParent(CGM, getTgtDeviceImageQTy(), Data, 4064 DeviceImagesEntries); 4065 } 4066 4067 // Create device images global array. 4068 std::string ImagesName = getName({"omp_offloading", "device_images"}); 4069 llvm::GlobalVariable *DeviceImages = 4070 DeviceImagesEntries.finishAndCreateGlobal(ImagesName, 4071 CGM.getPointerAlign(), 4072 /*isConstant=*/true); 4073 DeviceImages->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 4074 4075 // This is a Zero array to be used in the creation of the constant expressions 4076 llvm::Constant *Index[] = {llvm::Constant::getNullValue(CGM.Int32Ty), 4077 llvm::Constant::getNullValue(CGM.Int32Ty)}; 4078 4079 // Create the target region descriptor. 4080 llvm::Constant *Data[] = { 4081 llvm::ConstantInt::get(CGM.Int32Ty, Devices.size()), 4082 llvm::ConstantExpr::getGetElementPtr(DeviceImages->getValueType(), 4083 DeviceImages, Index), 4084 HostEntriesBegin, HostEntriesEnd}; 4085 std::string Descriptor = getName({"omp_offloading", "descriptor"}); 4086 llvm::GlobalVariable *Desc = createGlobalStruct( 4087 CGM, getTgtBinaryDescriptorQTy(), /*IsConstant=*/true, Data, Descriptor); 4088 4089 // Emit code to register or unregister the descriptor at execution 4090 // startup or closing, respectively. 4091 4092 llvm::Function *UnRegFn; 4093 { 4094 FunctionArgList Args; 4095 ImplicitParamDecl DummyPtr(C, C.VoidPtrTy, ImplicitParamDecl::Other); 4096 Args.push_back(&DummyPtr); 4097 4098 CodeGenFunction CGF(CGM); 4099 // Disable debug info for global (de-)initializer because they are not part 4100 // of some particular construct. 4101 CGF.disableDebugInfo(); 4102 const auto &FI = 4103 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4104 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 4105 std::string UnregName = getName({"omp_offloading", "descriptor_unreg"}); 4106 UnRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, UnregName, FI); 4107 CGF.StartFunction(GlobalDecl(), C.VoidTy, UnRegFn, FI, Args); 4108 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_unregister_lib), 4109 Desc); 4110 CGF.FinishFunction(); 4111 } 4112 llvm::Function *RegFn; 4113 { 4114 CodeGenFunction CGF(CGM); 4115 // Disable debug info for global (de-)initializer because they are not part 4116 // of some particular construct. 4117 CGF.disableDebugInfo(); 4118 const auto &FI = CGM.getTypes().arrangeNullaryFunction(); 4119 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 4120 4121 // Encode offload target triples into the registration function name. It 4122 // will serve as a comdat key for the registration/unregistration code for 4123 // this particular combination of offloading targets. 4124 SmallVector<StringRef, 4U> RegFnNameParts(Devices.size() + 2U); 4125 RegFnNameParts[0] = "omp_offloading"; 4126 RegFnNameParts[1] = "descriptor_reg"; 4127 llvm::transform(Devices, std::next(RegFnNameParts.begin(), 2), 4128 [](const llvm::Triple &T) -> const std::string& { 4129 return T.getTriple(); 4130 }); 4131 llvm::sort(std::next(RegFnNameParts.begin(), 2), RegFnNameParts.end()); 4132 std::string Descriptor = getName(RegFnNameParts); 4133 RegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, Descriptor, FI); 4134 CGF.StartFunction(GlobalDecl(), C.VoidTy, RegFn, FI, FunctionArgList()); 4135 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_lib), Desc); 4136 // Create a variable to drive the registration and unregistration of the 4137 // descriptor, so we can reuse the logic that emits Ctors and Dtors. 4138 ImplicitParamDecl RegUnregVar(C, C.getTranslationUnitDecl(), 4139 SourceLocation(), nullptr, C.CharTy, 4140 ImplicitParamDecl::Other); 4141 CGM.getCXXABI().registerGlobalDtor(CGF, RegUnregVar, UnRegFn, Desc); 4142 CGF.FinishFunction(); 4143 } 4144 if (CGM.supportsCOMDAT()) { 4145 // It is sufficient to call registration function only once, so create a 4146 // COMDAT group for registration/unregistration functions and associated 4147 // data. That would reduce startup time and code size. Registration 4148 // function serves as a COMDAT group key. 4149 llvm::Comdat *ComdatKey = M.getOrInsertComdat(RegFn->getName()); 4150 RegFn->setLinkage(llvm::GlobalValue::LinkOnceAnyLinkage); 4151 RegFn->setVisibility(llvm::GlobalValue::HiddenVisibility); 4152 RegFn->setComdat(ComdatKey); 4153 UnRegFn->setComdat(ComdatKey); 4154 DeviceImages->setComdat(ComdatKey); 4155 Desc->setComdat(ComdatKey); 4156 } 4157 return RegFn; 4158 } 4159 4160 void CGOpenMPRuntime::createOffloadEntry( 4161 llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags, 4162 llvm::GlobalValue::LinkageTypes Linkage) { 4163 StringRef Name = Addr->getName(); 4164 llvm::Module &M = CGM.getModule(); 4165 llvm::LLVMContext &C = M.getContext(); 4166 4167 // Create constant string with the name. 4168 llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name); 4169 4170 std::string StringName = getName({"omp_offloading", "entry_name"}); 4171 auto *Str = new llvm::GlobalVariable( 4172 M, StrPtrInit->getType(), /*isConstant=*/true, 4173 llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName); 4174 Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 4175 4176 llvm::Constant *Data[] = {llvm::ConstantExpr::getBitCast(ID, CGM.VoidPtrTy), 4177 llvm::ConstantExpr::getBitCast(Str, CGM.Int8PtrTy), 4178 llvm::ConstantInt::get(CGM.SizeTy, Size), 4179 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 4180 llvm::ConstantInt::get(CGM.Int32Ty, 0)}; 4181 std::string EntryName = getName({"omp_offloading", "entry", ""}); 4182 llvm::GlobalVariable *Entry = createGlobalStruct( 4183 CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data, 4184 Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage); 4185 4186 // The entry has to be created in the section the linker expects it to be. 4187 Entry->setSection("omp_offloading_entries"); 4188 } 4189 4190 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() { 4191 // Emit the offloading entries and metadata so that the device codegen side 4192 // can easily figure out what to emit. The produced metadata looks like 4193 // this: 4194 // 4195 // !omp_offload.info = !{!1, ...} 4196 // 4197 // Right now we only generate metadata for function that contain target 4198 // regions. 4199 4200 // If we do not have entries, we don't need to do anything. 4201 if (OffloadEntriesInfoManager.empty()) 4202 return; 4203 4204 llvm::Module &M = CGM.getModule(); 4205 llvm::LLVMContext &C = M.getContext(); 4206 SmallVector<std::tuple<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *, 4207 SourceLocation, StringRef>, 4208 16> 4209 OrderedEntries(OffloadEntriesInfoManager.size()); 4210 llvm::SmallVector<StringRef, 16> ParentFunctions( 4211 OffloadEntriesInfoManager.size()); 4212 4213 // Auxiliary methods to create metadata values and strings. 4214 auto &&GetMDInt = [this](unsigned V) { 4215 return llvm::ConstantAsMetadata::get( 4216 llvm::ConstantInt::get(CGM.Int32Ty, V)); 4217 }; 4218 4219 auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); }; 4220 4221 // Create the offloading info metadata node. 4222 llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info"); 4223 4224 // Create function that emits metadata for each target region entry; 4225 auto &&TargetRegionMetadataEmitter = 4226 [this, &C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt, 4227 &GetMDString]( 4228 unsigned DeviceID, unsigned FileID, StringRef ParentName, 4229 unsigned Line, 4230 const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) { 4231 // Generate metadata for target regions. Each entry of this metadata 4232 // contains: 4233 // - Entry 0 -> Kind of this type of metadata (0). 4234 // - Entry 1 -> Device ID of the file where the entry was identified. 4235 // - Entry 2 -> File ID of the file where the entry was identified. 4236 // - Entry 3 -> Mangled name of the function where the entry was 4237 // identified. 4238 // - Entry 4 -> Line in the file where the entry was identified. 4239 // - Entry 5 -> Order the entry was created. 4240 // The first element of the metadata node is the kind. 4241 llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID), 4242 GetMDInt(FileID), GetMDString(ParentName), 4243 GetMDInt(Line), GetMDInt(E.getOrder())}; 4244 4245 SourceLocation Loc; 4246 for (auto I = CGM.getContext().getSourceManager().fileinfo_begin(), 4247 E = CGM.getContext().getSourceManager().fileinfo_end(); 4248 I != E; ++I) { 4249 if (I->getFirst()->getUniqueID().getDevice() == DeviceID && 4250 I->getFirst()->getUniqueID().getFile() == FileID) { 4251 Loc = CGM.getContext().getSourceManager().translateFileLineCol( 4252 I->getFirst(), Line, 1); 4253 break; 4254 } 4255 } 4256 // Save this entry in the right position of the ordered entries array. 4257 OrderedEntries[E.getOrder()] = std::make_tuple(&E, Loc, ParentName); 4258 ParentFunctions[E.getOrder()] = ParentName; 4259 4260 // Add metadata to the named metadata node. 4261 MD->addOperand(llvm::MDNode::get(C, Ops)); 4262 }; 4263 4264 OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo( 4265 TargetRegionMetadataEmitter); 4266 4267 // Create function that emits metadata for each device global variable entry; 4268 auto &&DeviceGlobalVarMetadataEmitter = 4269 [&C, &OrderedEntries, &GetMDInt, &GetMDString, 4270 MD](StringRef MangledName, 4271 const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar 4272 &E) { 4273 // Generate metadata for global variables. Each entry of this metadata 4274 // contains: 4275 // - Entry 0 -> Kind of this type of metadata (1). 4276 // - Entry 1 -> Mangled name of the variable. 4277 // - Entry 2 -> Declare target kind. 4278 // - Entry 3 -> Order the entry was created. 4279 // The first element of the metadata node is the kind. 4280 llvm::Metadata *Ops[] = { 4281 GetMDInt(E.getKind()), GetMDString(MangledName), 4282 GetMDInt(E.getFlags()), GetMDInt(E.getOrder())}; 4283 4284 // Save this entry in the right position of the ordered entries array. 4285 OrderedEntries[E.getOrder()] = 4286 std::make_tuple(&E, SourceLocation(), MangledName); 4287 4288 // Add metadata to the named metadata node. 4289 MD->addOperand(llvm::MDNode::get(C, Ops)); 4290 }; 4291 4292 OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo( 4293 DeviceGlobalVarMetadataEmitter); 4294 4295 for (const auto &E : OrderedEntries) { 4296 assert(std::get<0>(E) && "All ordered entries must exist!"); 4297 if (const auto *CE = 4298 dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>( 4299 std::get<0>(E))) { 4300 if (!CE->getID() || !CE->getAddress()) { 4301 // Do not blame the entry if the parent funtion is not emitted. 4302 StringRef FnName = ParentFunctions[CE->getOrder()]; 4303 if (!CGM.GetGlobalValue(FnName)) 4304 continue; 4305 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4306 DiagnosticsEngine::Error, 4307 "Offloading entry for target region in %0 is incorrect: either the " 4308 "address or the ID is invalid."); 4309 CGM.getDiags().Report(std::get<1>(E), DiagID) << FnName; 4310 continue; 4311 } 4312 createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0, 4313 CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage); 4314 } else if (const auto *CE = dyn_cast<OffloadEntriesInfoManagerTy:: 4315 OffloadEntryInfoDeviceGlobalVar>( 4316 std::get<0>(E))) { 4317 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags = 4318 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 4319 CE->getFlags()); 4320 switch (Flags) { 4321 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: { 4322 if (CGM.getLangOpts().OpenMPIsDevice && 4323 CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory()) 4324 continue; 4325 if (!CE->getAddress()) { 4326 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4327 DiagnosticsEngine::Error, "Offloading entry for declare target " 4328 "variable %0 is incorrect: the " 4329 "address is invalid."); 4330 CGM.getDiags().Report(std::get<1>(E), DiagID) << std::get<2>(E); 4331 continue; 4332 } 4333 // The vaiable has no definition - no need to add the entry. 4334 if (CE->getVarSize().isZero()) 4335 continue; 4336 break; 4337 } 4338 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink: 4339 assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) || 4340 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) && 4341 "Declaret target link address is set."); 4342 if (CGM.getLangOpts().OpenMPIsDevice) 4343 continue; 4344 if (!CE->getAddress()) { 4345 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4346 DiagnosticsEngine::Error, 4347 "Offloading entry for declare target variable is incorrect: the " 4348 "address is invalid."); 4349 CGM.getDiags().Report(DiagID); 4350 continue; 4351 } 4352 break; 4353 } 4354 createOffloadEntry(CE->getAddress(), CE->getAddress(), 4355 CE->getVarSize().getQuantity(), Flags, 4356 CE->getLinkage()); 4357 } else { 4358 llvm_unreachable("Unsupported entry kind."); 4359 } 4360 } 4361 } 4362 4363 /// Loads all the offload entries information from the host IR 4364 /// metadata. 4365 void CGOpenMPRuntime::loadOffloadInfoMetadata() { 4366 // If we are in target mode, load the metadata from the host IR. This code has 4367 // to match the metadaata creation in createOffloadEntriesAndInfoMetadata(). 4368 4369 if (!CGM.getLangOpts().OpenMPIsDevice) 4370 return; 4371 4372 if (CGM.getLangOpts().OMPHostIRFile.empty()) 4373 return; 4374 4375 auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile); 4376 if (auto EC = Buf.getError()) { 4377 CGM.getDiags().Report(diag::err_cannot_open_file) 4378 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 4379 return; 4380 } 4381 4382 llvm::LLVMContext C; 4383 auto ME = expectedToErrorOrAndEmitErrors( 4384 C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C)); 4385 4386 if (auto EC = ME.getError()) { 4387 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4388 DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'"); 4389 CGM.getDiags().Report(DiagID) 4390 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 4391 return; 4392 } 4393 4394 llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info"); 4395 if (!MD) 4396 return; 4397 4398 for (llvm::MDNode *MN : MD->operands()) { 4399 auto &&GetMDInt = [MN](unsigned Idx) { 4400 auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx)); 4401 return cast<llvm::ConstantInt>(V->getValue())->getZExtValue(); 4402 }; 4403 4404 auto &&GetMDString = [MN](unsigned Idx) { 4405 auto *V = cast<llvm::MDString>(MN->getOperand(Idx)); 4406 return V->getString(); 4407 }; 4408 4409 switch (GetMDInt(0)) { 4410 default: 4411 llvm_unreachable("Unexpected metadata!"); 4412 break; 4413 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 4414 OffloadingEntryInfoTargetRegion: 4415 OffloadEntriesInfoManager.initializeTargetRegionEntryInfo( 4416 /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2), 4417 /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4), 4418 /*Order=*/GetMDInt(5)); 4419 break; 4420 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 4421 OffloadingEntryInfoDeviceGlobalVar: 4422 OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo( 4423 /*MangledName=*/GetMDString(1), 4424 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 4425 /*Flags=*/GetMDInt(2)), 4426 /*Order=*/GetMDInt(3)); 4427 break; 4428 } 4429 } 4430 } 4431 4432 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) { 4433 if (!KmpRoutineEntryPtrTy) { 4434 // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type. 4435 ASTContext &C = CGM.getContext(); 4436 QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy}; 4437 FunctionProtoType::ExtProtoInfo EPI; 4438 KmpRoutineEntryPtrQTy = C.getPointerType( 4439 C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI)); 4440 KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy); 4441 } 4442 } 4443 4444 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() { 4445 // Make sure the type of the entry is already created. This is the type we 4446 // have to create: 4447 // struct __tgt_offload_entry{ 4448 // void *addr; // Pointer to the offload entry info. 4449 // // (function or global) 4450 // char *name; // Name of the function or global. 4451 // size_t size; // Size of the entry info (0 if it a function). 4452 // int32_t flags; // Flags associated with the entry, e.g. 'link'. 4453 // int32_t reserved; // Reserved, to use by the runtime library. 4454 // }; 4455 if (TgtOffloadEntryQTy.isNull()) { 4456 ASTContext &C = CGM.getContext(); 4457 RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry"); 4458 RD->startDefinition(); 4459 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4460 addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy)); 4461 addFieldToRecordDecl(C, RD, C.getSizeType()); 4462 addFieldToRecordDecl( 4463 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4464 addFieldToRecordDecl( 4465 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4466 RD->completeDefinition(); 4467 RD->addAttr(PackedAttr::CreateImplicit(C)); 4468 TgtOffloadEntryQTy = C.getRecordType(RD); 4469 } 4470 return TgtOffloadEntryQTy; 4471 } 4472 4473 QualType CGOpenMPRuntime::getTgtDeviceImageQTy() { 4474 // These are the types we need to build: 4475 // struct __tgt_device_image{ 4476 // void *ImageStart; // Pointer to the target code start. 4477 // void *ImageEnd; // Pointer to the target code end. 4478 // // We also add the host entries to the device image, as it may be useful 4479 // // for the target runtime to have access to that information. 4480 // __tgt_offload_entry *EntriesBegin; // Begin of the table with all 4481 // // the entries. 4482 // __tgt_offload_entry *EntriesEnd; // End of the table with all the 4483 // // entries (non inclusive). 4484 // }; 4485 if (TgtDeviceImageQTy.isNull()) { 4486 ASTContext &C = CGM.getContext(); 4487 RecordDecl *RD = C.buildImplicitRecord("__tgt_device_image"); 4488 RD->startDefinition(); 4489 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4490 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4491 addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy())); 4492 addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy())); 4493 RD->completeDefinition(); 4494 TgtDeviceImageQTy = C.getRecordType(RD); 4495 } 4496 return TgtDeviceImageQTy; 4497 } 4498 4499 QualType CGOpenMPRuntime::getTgtBinaryDescriptorQTy() { 4500 // struct __tgt_bin_desc{ 4501 // int32_t NumDevices; // Number of devices supported. 4502 // __tgt_device_image *DeviceImages; // Arrays of device images 4503 // // (one per device). 4504 // __tgt_offload_entry *EntriesBegin; // Begin of the table with all the 4505 // // entries. 4506 // __tgt_offload_entry *EntriesEnd; // End of the table with all the 4507 // // entries (non inclusive). 4508 // }; 4509 if (TgtBinaryDescriptorQTy.isNull()) { 4510 ASTContext &C = CGM.getContext(); 4511 RecordDecl *RD = C.buildImplicitRecord("__tgt_bin_desc"); 4512 RD->startDefinition(); 4513 addFieldToRecordDecl( 4514 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4515 addFieldToRecordDecl(C, RD, C.getPointerType(getTgtDeviceImageQTy())); 4516 addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy())); 4517 addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy())); 4518 RD->completeDefinition(); 4519 TgtBinaryDescriptorQTy = C.getRecordType(RD); 4520 } 4521 return TgtBinaryDescriptorQTy; 4522 } 4523 4524 namespace { 4525 struct PrivateHelpersTy { 4526 PrivateHelpersTy(const VarDecl *Original, const VarDecl *PrivateCopy, 4527 const VarDecl *PrivateElemInit) 4528 : Original(Original), PrivateCopy(PrivateCopy), 4529 PrivateElemInit(PrivateElemInit) {} 4530 const VarDecl *Original; 4531 const VarDecl *PrivateCopy; 4532 const VarDecl *PrivateElemInit; 4533 }; 4534 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy; 4535 } // anonymous namespace 4536 4537 static RecordDecl * 4538 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) { 4539 if (!Privates.empty()) { 4540 ASTContext &C = CGM.getContext(); 4541 // Build struct .kmp_privates_t. { 4542 // /* private vars */ 4543 // }; 4544 RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t"); 4545 RD->startDefinition(); 4546 for (const auto &Pair : Privates) { 4547 const VarDecl *VD = Pair.second.Original; 4548 QualType Type = VD->getType().getNonReferenceType(); 4549 FieldDecl *FD = addFieldToRecordDecl(C, RD, Type); 4550 if (VD->hasAttrs()) { 4551 for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()), 4552 E(VD->getAttrs().end()); 4553 I != E; ++I) 4554 FD->addAttr(*I); 4555 } 4556 } 4557 RD->completeDefinition(); 4558 return RD; 4559 } 4560 return nullptr; 4561 } 4562 4563 static RecordDecl * 4564 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind, 4565 QualType KmpInt32Ty, 4566 QualType KmpRoutineEntryPointerQTy) { 4567 ASTContext &C = CGM.getContext(); 4568 // Build struct kmp_task_t { 4569 // void * shareds; 4570 // kmp_routine_entry_t routine; 4571 // kmp_int32 part_id; 4572 // kmp_cmplrdata_t data1; 4573 // kmp_cmplrdata_t data2; 4574 // For taskloops additional fields: 4575 // kmp_uint64 lb; 4576 // kmp_uint64 ub; 4577 // kmp_int64 st; 4578 // kmp_int32 liter; 4579 // void * reductions; 4580 // }; 4581 RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union); 4582 UD->startDefinition(); 4583 addFieldToRecordDecl(C, UD, KmpInt32Ty); 4584 addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy); 4585 UD->completeDefinition(); 4586 QualType KmpCmplrdataTy = C.getRecordType(UD); 4587 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t"); 4588 RD->startDefinition(); 4589 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4590 addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy); 4591 addFieldToRecordDecl(C, RD, KmpInt32Ty); 4592 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 4593 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 4594 if (isOpenMPTaskLoopDirective(Kind)) { 4595 QualType KmpUInt64Ty = 4596 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0); 4597 QualType KmpInt64Ty = 4598 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 4599 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 4600 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 4601 addFieldToRecordDecl(C, RD, KmpInt64Ty); 4602 addFieldToRecordDecl(C, RD, KmpInt32Ty); 4603 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4604 } 4605 RD->completeDefinition(); 4606 return RD; 4607 } 4608 4609 static RecordDecl * 4610 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy, 4611 ArrayRef<PrivateDataTy> Privates) { 4612 ASTContext &C = CGM.getContext(); 4613 // Build struct kmp_task_t_with_privates { 4614 // kmp_task_t task_data; 4615 // .kmp_privates_t. privates; 4616 // }; 4617 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates"); 4618 RD->startDefinition(); 4619 addFieldToRecordDecl(C, RD, KmpTaskTQTy); 4620 if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates)) 4621 addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD)); 4622 RD->completeDefinition(); 4623 return RD; 4624 } 4625 4626 /// Emit a proxy function which accepts kmp_task_t as the second 4627 /// argument. 4628 /// \code 4629 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) { 4630 /// TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt, 4631 /// For taskloops: 4632 /// tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 4633 /// tt->reductions, tt->shareds); 4634 /// return 0; 4635 /// } 4636 /// \endcode 4637 static llvm::Function * 4638 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc, 4639 OpenMPDirectiveKind Kind, QualType KmpInt32Ty, 4640 QualType KmpTaskTWithPrivatesPtrQTy, 4641 QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy, 4642 QualType SharedsPtrTy, llvm::Function *TaskFunction, 4643 llvm::Value *TaskPrivatesMap) { 4644 ASTContext &C = CGM.getContext(); 4645 FunctionArgList Args; 4646 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 4647 ImplicitParamDecl::Other); 4648 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4649 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 4650 ImplicitParamDecl::Other); 4651 Args.push_back(&GtidArg); 4652 Args.push_back(&TaskTypeArg); 4653 const auto &TaskEntryFnInfo = 4654 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 4655 llvm::FunctionType *TaskEntryTy = 4656 CGM.getTypes().GetFunctionType(TaskEntryFnInfo); 4657 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""}); 4658 auto *TaskEntry = llvm::Function::Create( 4659 TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 4660 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo); 4661 TaskEntry->setDoesNotRecurse(); 4662 CodeGenFunction CGF(CGM); 4663 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args, 4664 Loc, Loc); 4665 4666 // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map, 4667 // tt, 4668 // For taskloops: 4669 // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 4670 // tt->task_data.shareds); 4671 llvm::Value *GtidParam = CGF.EmitLoadOfScalar( 4672 CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc); 4673 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4674 CGF.GetAddrOfLocalVar(&TaskTypeArg), 4675 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4676 const auto *KmpTaskTWithPrivatesQTyRD = 4677 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 4678 LValue Base = 4679 CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4680 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 4681 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 4682 LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI); 4683 llvm::Value *PartidParam = PartIdLVal.getPointer(); 4684 4685 auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds); 4686 LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI); 4687 llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4688 CGF.EmitLoadOfScalar(SharedsLVal, Loc), 4689 CGF.ConvertTypeForMem(SharedsPtrTy)); 4690 4691 auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 4692 llvm::Value *PrivatesParam; 4693 if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) { 4694 LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI); 4695 PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4696 PrivatesLVal.getPointer(), CGF.VoidPtrTy); 4697 } else { 4698 PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 4699 } 4700 4701 llvm::Value *CommonArgs[] = {GtidParam, PartidParam, PrivatesParam, 4702 TaskPrivatesMap, 4703 CGF.Builder 4704 .CreatePointerBitCastOrAddrSpaceCast( 4705 TDBase.getAddress(), CGF.VoidPtrTy) 4706 .getPointer()}; 4707 SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs), 4708 std::end(CommonArgs)); 4709 if (isOpenMPTaskLoopDirective(Kind)) { 4710 auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound); 4711 LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI); 4712 llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc); 4713 auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound); 4714 LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI); 4715 llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc); 4716 auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride); 4717 LValue StLVal = CGF.EmitLValueForField(Base, *StFI); 4718 llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc); 4719 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 4720 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 4721 llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc); 4722 auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions); 4723 LValue RLVal = CGF.EmitLValueForField(Base, *RFI); 4724 llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc); 4725 CallArgs.push_back(LBParam); 4726 CallArgs.push_back(UBParam); 4727 CallArgs.push_back(StParam); 4728 CallArgs.push_back(LIParam); 4729 CallArgs.push_back(RParam); 4730 } 4731 CallArgs.push_back(SharedsParam); 4732 4733 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction, 4734 CallArgs); 4735 CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)), 4736 CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty)); 4737 CGF.FinishFunction(); 4738 return TaskEntry; 4739 } 4740 4741 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM, 4742 SourceLocation Loc, 4743 QualType KmpInt32Ty, 4744 QualType KmpTaskTWithPrivatesPtrQTy, 4745 QualType KmpTaskTWithPrivatesQTy) { 4746 ASTContext &C = CGM.getContext(); 4747 FunctionArgList Args; 4748 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 4749 ImplicitParamDecl::Other); 4750 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4751 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 4752 ImplicitParamDecl::Other); 4753 Args.push_back(&GtidArg); 4754 Args.push_back(&TaskTypeArg); 4755 const auto &DestructorFnInfo = 4756 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 4757 llvm::FunctionType *DestructorFnTy = 4758 CGM.getTypes().GetFunctionType(DestructorFnInfo); 4759 std::string Name = 4760 CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""}); 4761 auto *DestructorFn = 4762 llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage, 4763 Name, &CGM.getModule()); 4764 CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn, 4765 DestructorFnInfo); 4766 DestructorFn->setDoesNotRecurse(); 4767 CodeGenFunction CGF(CGM); 4768 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo, 4769 Args, Loc, Loc); 4770 4771 LValue Base = CGF.EmitLoadOfPointerLValue( 4772 CGF.GetAddrOfLocalVar(&TaskTypeArg), 4773 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4774 const auto *KmpTaskTWithPrivatesQTyRD = 4775 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 4776 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4777 Base = CGF.EmitLValueForField(Base, *FI); 4778 for (const auto *Field : 4779 cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) { 4780 if (QualType::DestructionKind DtorKind = 4781 Field->getType().isDestructedType()) { 4782 LValue FieldLValue = CGF.EmitLValueForField(Base, Field); 4783 CGF.pushDestroy(DtorKind, FieldLValue.getAddress(), Field->getType()); 4784 } 4785 } 4786 CGF.FinishFunction(); 4787 return DestructorFn; 4788 } 4789 4790 /// Emit a privates mapping function for correct handling of private and 4791 /// firstprivate variables. 4792 /// \code 4793 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1> 4794 /// **noalias priv1,..., <tyn> **noalias privn) { 4795 /// *priv1 = &.privates.priv1; 4796 /// ...; 4797 /// *privn = &.privates.privn; 4798 /// } 4799 /// \endcode 4800 static llvm::Value * 4801 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc, 4802 ArrayRef<const Expr *> PrivateVars, 4803 ArrayRef<const Expr *> FirstprivateVars, 4804 ArrayRef<const Expr *> LastprivateVars, 4805 QualType PrivatesQTy, 4806 ArrayRef<PrivateDataTy> Privates) { 4807 ASTContext &C = CGM.getContext(); 4808 FunctionArgList Args; 4809 ImplicitParamDecl TaskPrivatesArg( 4810 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4811 C.getPointerType(PrivatesQTy).withConst().withRestrict(), 4812 ImplicitParamDecl::Other); 4813 Args.push_back(&TaskPrivatesArg); 4814 llvm::DenseMap<const VarDecl *, unsigned> PrivateVarsPos; 4815 unsigned Counter = 1; 4816 for (const Expr *E : PrivateVars) { 4817 Args.push_back(ImplicitParamDecl::Create( 4818 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4819 C.getPointerType(C.getPointerType(E->getType())) 4820 .withConst() 4821 .withRestrict(), 4822 ImplicitParamDecl::Other)); 4823 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4824 PrivateVarsPos[VD] = Counter; 4825 ++Counter; 4826 } 4827 for (const Expr *E : FirstprivateVars) { 4828 Args.push_back(ImplicitParamDecl::Create( 4829 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4830 C.getPointerType(C.getPointerType(E->getType())) 4831 .withConst() 4832 .withRestrict(), 4833 ImplicitParamDecl::Other)); 4834 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4835 PrivateVarsPos[VD] = Counter; 4836 ++Counter; 4837 } 4838 for (const Expr *E : LastprivateVars) { 4839 Args.push_back(ImplicitParamDecl::Create( 4840 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4841 C.getPointerType(C.getPointerType(E->getType())) 4842 .withConst() 4843 .withRestrict(), 4844 ImplicitParamDecl::Other)); 4845 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4846 PrivateVarsPos[VD] = Counter; 4847 ++Counter; 4848 } 4849 const auto &TaskPrivatesMapFnInfo = 4850 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4851 llvm::FunctionType *TaskPrivatesMapTy = 4852 CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo); 4853 std::string Name = 4854 CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""}); 4855 auto *TaskPrivatesMap = llvm::Function::Create( 4856 TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name, 4857 &CGM.getModule()); 4858 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap, 4859 TaskPrivatesMapFnInfo); 4860 if (CGM.getLangOpts().Optimize) { 4861 TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline); 4862 TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone); 4863 TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline); 4864 } 4865 CodeGenFunction CGF(CGM); 4866 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap, 4867 TaskPrivatesMapFnInfo, Args, Loc, Loc); 4868 4869 // *privi = &.privates.privi; 4870 LValue Base = CGF.EmitLoadOfPointerLValue( 4871 CGF.GetAddrOfLocalVar(&TaskPrivatesArg), 4872 TaskPrivatesArg.getType()->castAs<PointerType>()); 4873 const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl()); 4874 Counter = 0; 4875 for (const FieldDecl *Field : PrivatesQTyRD->fields()) { 4876 LValue FieldLVal = CGF.EmitLValueForField(Base, Field); 4877 const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]]; 4878 LValue RefLVal = 4879 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType()); 4880 LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue( 4881 RefLVal.getAddress(), RefLVal.getType()->castAs<PointerType>()); 4882 CGF.EmitStoreOfScalar(FieldLVal.getPointer(), RefLoadLVal); 4883 ++Counter; 4884 } 4885 CGF.FinishFunction(); 4886 return TaskPrivatesMap; 4887 } 4888 4889 /// Emit initialization for private variables in task-based directives. 4890 static void emitPrivatesInit(CodeGenFunction &CGF, 4891 const OMPExecutableDirective &D, 4892 Address KmpTaskSharedsPtr, LValue TDBase, 4893 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 4894 QualType SharedsTy, QualType SharedsPtrTy, 4895 const OMPTaskDataTy &Data, 4896 ArrayRef<PrivateDataTy> Privates, bool ForDup) { 4897 ASTContext &C = CGF.getContext(); 4898 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4899 LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI); 4900 OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind()) 4901 ? OMPD_taskloop 4902 : OMPD_task; 4903 const CapturedStmt &CS = *D.getCapturedStmt(Kind); 4904 CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS); 4905 LValue SrcBase; 4906 bool IsTargetTask = 4907 isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) || 4908 isOpenMPTargetExecutionDirective(D.getDirectiveKind()); 4909 // For target-based directives skip 3 firstprivate arrays BasePointersArray, 4910 // PointersArray and SizesArray. The original variables for these arrays are 4911 // not captured and we get their addresses explicitly. 4912 if ((!IsTargetTask && !Data.FirstprivateVars.empty()) || 4913 (IsTargetTask && KmpTaskSharedsPtr.isValid())) { 4914 SrcBase = CGF.MakeAddrLValue( 4915 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4916 KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)), 4917 SharedsTy); 4918 } 4919 FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin(); 4920 for (const PrivateDataTy &Pair : Privates) { 4921 const VarDecl *VD = Pair.second.PrivateCopy; 4922 const Expr *Init = VD->getAnyInitializer(); 4923 if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) && 4924 !CGF.isTrivialInitializer(Init)))) { 4925 LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI); 4926 if (const VarDecl *Elem = Pair.second.PrivateElemInit) { 4927 const VarDecl *OriginalVD = Pair.second.Original; 4928 // Check if the variable is the target-based BasePointersArray, 4929 // PointersArray or SizesArray. 4930 LValue SharedRefLValue; 4931 QualType Type = PrivateLValue.getType(); 4932 const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD); 4933 if (IsTargetTask && !SharedField) { 4934 assert(isa<ImplicitParamDecl>(OriginalVD) && 4935 isa<CapturedDecl>(OriginalVD->getDeclContext()) && 4936 cast<CapturedDecl>(OriginalVD->getDeclContext()) 4937 ->getNumParams() == 0 && 4938 isa<TranslationUnitDecl>( 4939 cast<CapturedDecl>(OriginalVD->getDeclContext()) 4940 ->getDeclContext()) && 4941 "Expected artificial target data variable."); 4942 SharedRefLValue = 4943 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type); 4944 } else { 4945 SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField); 4946 SharedRefLValue = CGF.MakeAddrLValue( 4947 Address(SharedRefLValue.getPointer(), C.getDeclAlign(OriginalVD)), 4948 SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl), 4949 SharedRefLValue.getTBAAInfo()); 4950 } 4951 if (Type->isArrayType()) { 4952 // Initialize firstprivate array. 4953 if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) { 4954 // Perform simple memcpy. 4955 CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type); 4956 } else { 4957 // Initialize firstprivate array using element-by-element 4958 // initialization. 4959 CGF.EmitOMPAggregateAssign( 4960 PrivateLValue.getAddress(), SharedRefLValue.getAddress(), Type, 4961 [&CGF, Elem, Init, &CapturesInfo](Address DestElement, 4962 Address SrcElement) { 4963 // Clean up any temporaries needed by the initialization. 4964 CodeGenFunction::OMPPrivateScope InitScope(CGF); 4965 InitScope.addPrivate( 4966 Elem, [SrcElement]() -> Address { return SrcElement; }); 4967 (void)InitScope.Privatize(); 4968 // Emit initialization for single element. 4969 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII( 4970 CGF, &CapturesInfo); 4971 CGF.EmitAnyExprToMem(Init, DestElement, 4972 Init->getType().getQualifiers(), 4973 /*IsInitializer=*/false); 4974 }); 4975 } 4976 } else { 4977 CodeGenFunction::OMPPrivateScope InitScope(CGF); 4978 InitScope.addPrivate(Elem, [SharedRefLValue]() -> Address { 4979 return SharedRefLValue.getAddress(); 4980 }); 4981 (void)InitScope.Privatize(); 4982 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo); 4983 CGF.EmitExprAsInit(Init, VD, PrivateLValue, 4984 /*capturedByInit=*/false); 4985 } 4986 } else { 4987 CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false); 4988 } 4989 } 4990 ++FI; 4991 } 4992 } 4993 4994 /// Check if duplication function is required for taskloops. 4995 static bool checkInitIsRequired(CodeGenFunction &CGF, 4996 ArrayRef<PrivateDataTy> Privates) { 4997 bool InitRequired = false; 4998 for (const PrivateDataTy &Pair : Privates) { 4999 const VarDecl *VD = Pair.second.PrivateCopy; 5000 const Expr *Init = VD->getAnyInitializer(); 5001 InitRequired = InitRequired || (Init && isa<CXXConstructExpr>(Init) && 5002 !CGF.isTrivialInitializer(Init)); 5003 if (InitRequired) 5004 break; 5005 } 5006 return InitRequired; 5007 } 5008 5009 5010 /// Emit task_dup function (for initialization of 5011 /// private/firstprivate/lastprivate vars and last_iter flag) 5012 /// \code 5013 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int 5014 /// lastpriv) { 5015 /// // setup lastprivate flag 5016 /// task_dst->last = lastpriv; 5017 /// // could be constructor calls here... 5018 /// } 5019 /// \endcode 5020 static llvm::Value * 5021 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc, 5022 const OMPExecutableDirective &D, 5023 QualType KmpTaskTWithPrivatesPtrQTy, 5024 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 5025 const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy, 5026 QualType SharedsPtrTy, const OMPTaskDataTy &Data, 5027 ArrayRef<PrivateDataTy> Privates, bool WithLastIter) { 5028 ASTContext &C = CGM.getContext(); 5029 FunctionArgList Args; 5030 ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 5031 KmpTaskTWithPrivatesPtrQTy, 5032 ImplicitParamDecl::Other); 5033 ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 5034 KmpTaskTWithPrivatesPtrQTy, 5035 ImplicitParamDecl::Other); 5036 ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy, 5037 ImplicitParamDecl::Other); 5038 Args.push_back(&DstArg); 5039 Args.push_back(&SrcArg); 5040 Args.push_back(&LastprivArg); 5041 const auto &TaskDupFnInfo = 5042 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5043 llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo); 5044 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""}); 5045 auto *TaskDup = llvm::Function::Create( 5046 TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 5047 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo); 5048 TaskDup->setDoesNotRecurse(); 5049 CodeGenFunction CGF(CGM); 5050 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc, 5051 Loc); 5052 5053 LValue TDBase = CGF.EmitLoadOfPointerLValue( 5054 CGF.GetAddrOfLocalVar(&DstArg), 5055 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 5056 // task_dst->liter = lastpriv; 5057 if (WithLastIter) { 5058 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 5059 LValue Base = CGF.EmitLValueForField( 5060 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 5061 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 5062 llvm::Value *Lastpriv = CGF.EmitLoadOfScalar( 5063 CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc); 5064 CGF.EmitStoreOfScalar(Lastpriv, LILVal); 5065 } 5066 5067 // Emit initial values for private copies (if any). 5068 assert(!Privates.empty()); 5069 Address KmpTaskSharedsPtr = Address::invalid(); 5070 if (!Data.FirstprivateVars.empty()) { 5071 LValue TDBase = CGF.EmitLoadOfPointerLValue( 5072 CGF.GetAddrOfLocalVar(&SrcArg), 5073 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 5074 LValue Base = CGF.EmitLValueForField( 5075 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 5076 KmpTaskSharedsPtr = Address( 5077 CGF.EmitLoadOfScalar(CGF.EmitLValueForField( 5078 Base, *std::next(KmpTaskTQTyRD->field_begin(), 5079 KmpTaskTShareds)), 5080 Loc), 5081 CGF.getNaturalTypeAlignment(SharedsTy)); 5082 } 5083 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD, 5084 SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true); 5085 CGF.FinishFunction(); 5086 return TaskDup; 5087 } 5088 5089 /// Checks if destructor function is required to be generated. 5090 /// \return true if cleanups are required, false otherwise. 5091 static bool 5092 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD) { 5093 bool NeedsCleanup = false; 5094 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 5095 const auto *PrivateRD = cast<RecordDecl>(FI->getType()->getAsTagDecl()); 5096 for (const FieldDecl *FD : PrivateRD->fields()) { 5097 NeedsCleanup = NeedsCleanup || FD->getType().isDestructedType(); 5098 if (NeedsCleanup) 5099 break; 5100 } 5101 return NeedsCleanup; 5102 } 5103 5104 CGOpenMPRuntime::TaskResultTy 5105 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc, 5106 const OMPExecutableDirective &D, 5107 llvm::Function *TaskFunction, QualType SharedsTy, 5108 Address Shareds, const OMPTaskDataTy &Data) { 5109 ASTContext &C = CGM.getContext(); 5110 llvm::SmallVector<PrivateDataTy, 4> Privates; 5111 // Aggregate privates and sort them by the alignment. 5112 auto I = Data.PrivateCopies.begin(); 5113 for (const Expr *E : Data.PrivateVars) { 5114 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 5115 Privates.emplace_back( 5116 C.getDeclAlign(VD), 5117 PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 5118 /*PrivateElemInit=*/nullptr)); 5119 ++I; 5120 } 5121 I = Data.FirstprivateCopies.begin(); 5122 auto IElemInitRef = Data.FirstprivateInits.begin(); 5123 for (const Expr *E : Data.FirstprivateVars) { 5124 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 5125 Privates.emplace_back( 5126 C.getDeclAlign(VD), 5127 PrivateHelpersTy( 5128 VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 5129 cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl()))); 5130 ++I; 5131 ++IElemInitRef; 5132 } 5133 I = Data.LastprivateCopies.begin(); 5134 for (const Expr *E : Data.LastprivateVars) { 5135 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 5136 Privates.emplace_back( 5137 C.getDeclAlign(VD), 5138 PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 5139 /*PrivateElemInit=*/nullptr)); 5140 ++I; 5141 } 5142 llvm::stable_sort(Privates, [](PrivateDataTy L, PrivateDataTy R) { 5143 return L.first > R.first; 5144 }); 5145 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 5146 // Build type kmp_routine_entry_t (if not built yet). 5147 emitKmpRoutineEntryT(KmpInt32Ty); 5148 // Build type kmp_task_t (if not built yet). 5149 if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) { 5150 if (SavedKmpTaskloopTQTy.isNull()) { 5151 SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl( 5152 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 5153 } 5154 KmpTaskTQTy = SavedKmpTaskloopTQTy; 5155 } else { 5156 assert((D.getDirectiveKind() == OMPD_task || 5157 isOpenMPTargetExecutionDirective(D.getDirectiveKind()) || 5158 isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) && 5159 "Expected taskloop, task or target directive"); 5160 if (SavedKmpTaskTQTy.isNull()) { 5161 SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl( 5162 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 5163 } 5164 KmpTaskTQTy = SavedKmpTaskTQTy; 5165 } 5166 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 5167 // Build particular struct kmp_task_t for the given task. 5168 const RecordDecl *KmpTaskTWithPrivatesQTyRD = 5169 createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates); 5170 QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD); 5171 QualType KmpTaskTWithPrivatesPtrQTy = 5172 C.getPointerType(KmpTaskTWithPrivatesQTy); 5173 llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy); 5174 llvm::Type *KmpTaskTWithPrivatesPtrTy = 5175 KmpTaskTWithPrivatesTy->getPointerTo(); 5176 llvm::Value *KmpTaskTWithPrivatesTySize = 5177 CGF.getTypeSize(KmpTaskTWithPrivatesQTy); 5178 QualType SharedsPtrTy = C.getPointerType(SharedsTy); 5179 5180 // Emit initial values for private copies (if any). 5181 llvm::Value *TaskPrivatesMap = nullptr; 5182 llvm::Type *TaskPrivatesMapTy = 5183 std::next(TaskFunction->arg_begin(), 3)->getType(); 5184 if (!Privates.empty()) { 5185 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 5186 TaskPrivatesMap = emitTaskPrivateMappingFunction( 5187 CGM, Loc, Data.PrivateVars, Data.FirstprivateVars, Data.LastprivateVars, 5188 FI->getType(), Privates); 5189 TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5190 TaskPrivatesMap, TaskPrivatesMapTy); 5191 } else { 5192 TaskPrivatesMap = llvm::ConstantPointerNull::get( 5193 cast<llvm::PointerType>(TaskPrivatesMapTy)); 5194 } 5195 // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid, 5196 // kmp_task_t *tt); 5197 llvm::Function *TaskEntry = emitProxyTaskFunction( 5198 CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 5199 KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction, 5200 TaskPrivatesMap); 5201 5202 // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 5203 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 5204 // kmp_routine_entry_t *task_entry); 5205 // Task flags. Format is taken from 5206 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h, 5207 // description of kmp_tasking_flags struct. 5208 enum { 5209 TiedFlag = 0x1, 5210 FinalFlag = 0x2, 5211 DestructorsFlag = 0x8, 5212 PriorityFlag = 0x20 5213 }; 5214 unsigned Flags = Data.Tied ? TiedFlag : 0; 5215 bool NeedsCleanup = false; 5216 if (!Privates.empty()) { 5217 NeedsCleanup = checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD); 5218 if (NeedsCleanup) 5219 Flags = Flags | DestructorsFlag; 5220 } 5221 if (Data.Priority.getInt()) 5222 Flags = Flags | PriorityFlag; 5223 llvm::Value *TaskFlags = 5224 Data.Final.getPointer() 5225 ? CGF.Builder.CreateSelect(Data.Final.getPointer(), 5226 CGF.Builder.getInt32(FinalFlag), 5227 CGF.Builder.getInt32(/*C=*/0)) 5228 : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0); 5229 TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags)); 5230 llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy)); 5231 SmallVector<llvm::Value *, 8> AllocArgs = {emitUpdateLocation(CGF, Loc), 5232 getThreadID(CGF, Loc), TaskFlags, KmpTaskTWithPrivatesTySize, 5233 SharedsSize, CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5234 TaskEntry, KmpRoutineEntryPtrTy)}; 5235 llvm::Value *NewTask; 5236 if (D.hasClausesOfKind<OMPNowaitClause>()) { 5237 // Check if we have any device clause associated with the directive. 5238 const Expr *Device = nullptr; 5239 if (auto *C = D.getSingleClause<OMPDeviceClause>()) 5240 Device = C->getDevice(); 5241 // Emit device ID if any otherwise use default value. 5242 llvm::Value *DeviceID; 5243 if (Device) 5244 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 5245 CGF.Int64Ty, /*isSigned=*/true); 5246 else 5247 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 5248 AllocArgs.push_back(DeviceID); 5249 NewTask = CGF.EmitRuntimeCall( 5250 createRuntimeFunction(OMPRTL__kmpc_omp_target_task_alloc), AllocArgs); 5251 } else { 5252 NewTask = CGF.EmitRuntimeCall( 5253 createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs); 5254 } 5255 llvm::Value *NewTaskNewTaskTTy = 5256 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5257 NewTask, KmpTaskTWithPrivatesPtrTy); 5258 LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy, 5259 KmpTaskTWithPrivatesQTy); 5260 LValue TDBase = 5261 CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin()); 5262 // Fill the data in the resulting kmp_task_t record. 5263 // Copy shareds if there are any. 5264 Address KmpTaskSharedsPtr = Address::invalid(); 5265 if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) { 5266 KmpTaskSharedsPtr = 5267 Address(CGF.EmitLoadOfScalar( 5268 CGF.EmitLValueForField( 5269 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), 5270 KmpTaskTShareds)), 5271 Loc), 5272 CGF.getNaturalTypeAlignment(SharedsTy)); 5273 LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy); 5274 LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy); 5275 CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap); 5276 } 5277 // Emit initial values for private copies (if any). 5278 TaskResultTy Result; 5279 if (!Privates.empty()) { 5280 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD, 5281 SharedsTy, SharedsPtrTy, Data, Privates, 5282 /*ForDup=*/false); 5283 if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) && 5284 (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) { 5285 Result.TaskDupFn = emitTaskDupFunction( 5286 CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD, 5287 KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates, 5288 /*WithLastIter=*/!Data.LastprivateVars.empty()); 5289 } 5290 } 5291 // Fields of union "kmp_cmplrdata_t" for destructors and priority. 5292 enum { Priority = 0, Destructors = 1 }; 5293 // Provide pointer to function with destructors for privates. 5294 auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1); 5295 const RecordDecl *KmpCmplrdataUD = 5296 (*FI)->getType()->getAsUnionType()->getDecl(); 5297 if (NeedsCleanup) { 5298 llvm::Value *DestructorFn = emitDestructorsFunction( 5299 CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 5300 KmpTaskTWithPrivatesQTy); 5301 LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI); 5302 LValue DestructorsLV = CGF.EmitLValueForField( 5303 Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors)); 5304 CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5305 DestructorFn, KmpRoutineEntryPtrTy), 5306 DestructorsLV); 5307 } 5308 // Set priority. 5309 if (Data.Priority.getInt()) { 5310 LValue Data2LV = CGF.EmitLValueForField( 5311 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2)); 5312 LValue PriorityLV = CGF.EmitLValueForField( 5313 Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority)); 5314 CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV); 5315 } 5316 Result.NewTask = NewTask; 5317 Result.TaskEntry = TaskEntry; 5318 Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy; 5319 Result.TDBase = TDBase; 5320 Result.KmpTaskTQTyRD = KmpTaskTQTyRD; 5321 return Result; 5322 } 5323 5324 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 5325 const OMPExecutableDirective &D, 5326 llvm::Function *TaskFunction, 5327 QualType SharedsTy, Address Shareds, 5328 const Expr *IfCond, 5329 const OMPTaskDataTy &Data) { 5330 if (!CGF.HaveInsertPoint()) 5331 return; 5332 5333 TaskResultTy Result = 5334 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5335 llvm::Value *NewTask = Result.NewTask; 5336 llvm::Function *TaskEntry = Result.TaskEntry; 5337 llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy; 5338 LValue TDBase = Result.TDBase; 5339 const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD; 5340 ASTContext &C = CGM.getContext(); 5341 // Process list of dependences. 5342 Address DependenciesArray = Address::invalid(); 5343 unsigned NumDependencies = Data.Dependences.size(); 5344 if (NumDependencies) { 5345 // Dependence kind for RTL. 5346 enum RTLDependenceKindTy { DepIn = 0x01, DepInOut = 0x3, DepMutexInOutSet = 0x4 }; 5347 enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags }; 5348 RecordDecl *KmpDependInfoRD; 5349 QualType FlagsTy = 5350 C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false); 5351 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 5352 if (KmpDependInfoTy.isNull()) { 5353 KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info"); 5354 KmpDependInfoRD->startDefinition(); 5355 addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType()); 5356 addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType()); 5357 addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy); 5358 KmpDependInfoRD->completeDefinition(); 5359 KmpDependInfoTy = C.getRecordType(KmpDependInfoRD); 5360 } else { 5361 KmpDependInfoRD = cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5362 } 5363 // Define type kmp_depend_info[<Dependences.size()>]; 5364 QualType KmpDependInfoArrayTy = C.getConstantArrayType( 5365 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies), 5366 nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 5367 // kmp_depend_info[<Dependences.size()>] deps; 5368 DependenciesArray = 5369 CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr"); 5370 for (unsigned I = 0; I < NumDependencies; ++I) { 5371 const Expr *E = Data.Dependences[I].second; 5372 LValue Addr = CGF.EmitLValue(E); 5373 llvm::Value *Size; 5374 QualType Ty = E->getType(); 5375 if (const auto *ASE = 5376 dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) { 5377 LValue UpAddrLVal = 5378 CGF.EmitOMPArraySectionExpr(ASE, /*IsLowerBound=*/false); 5379 llvm::Value *UpAddr = 5380 CGF.Builder.CreateConstGEP1_32(UpAddrLVal.getPointer(), /*Idx0=*/1); 5381 llvm::Value *LowIntPtr = 5382 CGF.Builder.CreatePtrToInt(Addr.getPointer(), CGM.SizeTy); 5383 llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGM.SizeTy); 5384 Size = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr); 5385 } else { 5386 Size = CGF.getTypeSize(Ty); 5387 } 5388 LValue Base = CGF.MakeAddrLValue( 5389 CGF.Builder.CreateConstArrayGEP(DependenciesArray, I), 5390 KmpDependInfoTy); 5391 // deps[i].base_addr = &<Dependences[i].second>; 5392 LValue BaseAddrLVal = CGF.EmitLValueForField( 5393 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 5394 CGF.EmitStoreOfScalar( 5395 CGF.Builder.CreatePtrToInt(Addr.getPointer(), CGF.IntPtrTy), 5396 BaseAddrLVal); 5397 // deps[i].len = sizeof(<Dependences[i].second>); 5398 LValue LenLVal = CGF.EmitLValueForField( 5399 Base, *std::next(KmpDependInfoRD->field_begin(), Len)); 5400 CGF.EmitStoreOfScalar(Size, LenLVal); 5401 // deps[i].flags = <Dependences[i].first>; 5402 RTLDependenceKindTy DepKind; 5403 switch (Data.Dependences[I].first) { 5404 case OMPC_DEPEND_in: 5405 DepKind = DepIn; 5406 break; 5407 // Out and InOut dependencies must use the same code. 5408 case OMPC_DEPEND_out: 5409 case OMPC_DEPEND_inout: 5410 DepKind = DepInOut; 5411 break; 5412 case OMPC_DEPEND_mutexinoutset: 5413 DepKind = DepMutexInOutSet; 5414 break; 5415 case OMPC_DEPEND_source: 5416 case OMPC_DEPEND_sink: 5417 case OMPC_DEPEND_unknown: 5418 llvm_unreachable("Unknown task dependence type"); 5419 } 5420 LValue FlagsLVal = CGF.EmitLValueForField( 5421 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 5422 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 5423 FlagsLVal); 5424 } 5425 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5426 CGF.Builder.CreateConstArrayGEP(DependenciesArray, 0), CGF.VoidPtrTy); 5427 } 5428 5429 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5430 // libcall. 5431 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 5432 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 5433 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence 5434 // list is not empty 5435 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5436 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5437 llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask }; 5438 llvm::Value *DepTaskArgs[7]; 5439 if (NumDependencies) { 5440 DepTaskArgs[0] = UpLoc; 5441 DepTaskArgs[1] = ThreadID; 5442 DepTaskArgs[2] = NewTask; 5443 DepTaskArgs[3] = CGF.Builder.getInt32(NumDependencies); 5444 DepTaskArgs[4] = DependenciesArray.getPointer(); 5445 DepTaskArgs[5] = CGF.Builder.getInt32(0); 5446 DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5447 } 5448 auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, NumDependencies, 5449 &TaskArgs, 5450 &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) { 5451 if (!Data.Tied) { 5452 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 5453 LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI); 5454 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal); 5455 } 5456 if (NumDependencies) { 5457 CGF.EmitRuntimeCall( 5458 createRuntimeFunction(OMPRTL__kmpc_omp_task_with_deps), DepTaskArgs); 5459 } else { 5460 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), 5461 TaskArgs); 5462 } 5463 // Check if parent region is untied and build return for untied task; 5464 if (auto *Region = 5465 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 5466 Region->emitUntiedSwitch(CGF); 5467 }; 5468 5469 llvm::Value *DepWaitTaskArgs[6]; 5470 if (NumDependencies) { 5471 DepWaitTaskArgs[0] = UpLoc; 5472 DepWaitTaskArgs[1] = ThreadID; 5473 DepWaitTaskArgs[2] = CGF.Builder.getInt32(NumDependencies); 5474 DepWaitTaskArgs[3] = DependenciesArray.getPointer(); 5475 DepWaitTaskArgs[4] = CGF.Builder.getInt32(0); 5476 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5477 } 5478 auto &&ElseCodeGen = [&TaskArgs, ThreadID, NewTaskNewTaskTTy, TaskEntry, 5479 NumDependencies, &DepWaitTaskArgs, 5480 Loc](CodeGenFunction &CGF, PrePostActionTy &) { 5481 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5482 CodeGenFunction::RunCleanupsScope LocalScope(CGF); 5483 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 5484 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 5485 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info 5486 // is specified. 5487 if (NumDependencies) 5488 CGF.EmitRuntimeCall(RT.createRuntimeFunction(OMPRTL__kmpc_omp_wait_deps), 5489 DepWaitTaskArgs); 5490 // Call proxy_task_entry(gtid, new_task); 5491 auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy, 5492 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 5493 Action.Enter(CGF); 5494 llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy}; 5495 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry, 5496 OutlinedFnArgs); 5497 }; 5498 5499 // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 5500 // kmp_task_t *new_task); 5501 // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 5502 // kmp_task_t *new_task); 5503 RegionCodeGenTy RCG(CodeGen); 5504 CommonActionTy Action( 5505 RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_begin_if0), TaskArgs, 5506 RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_complete_if0), TaskArgs); 5507 RCG.setAction(Action); 5508 RCG(CGF); 5509 }; 5510 5511 if (IfCond) { 5512 emitOMPIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen); 5513 } else { 5514 RegionCodeGenTy ThenRCG(ThenCodeGen); 5515 ThenRCG(CGF); 5516 } 5517 } 5518 5519 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc, 5520 const OMPLoopDirective &D, 5521 llvm::Function *TaskFunction, 5522 QualType SharedsTy, Address Shareds, 5523 const Expr *IfCond, 5524 const OMPTaskDataTy &Data) { 5525 if (!CGF.HaveInsertPoint()) 5526 return; 5527 TaskResultTy Result = 5528 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5529 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5530 // libcall. 5531 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 5532 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 5533 // sched, kmp_uint64 grainsize, void *task_dup); 5534 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5535 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5536 llvm::Value *IfVal; 5537 if (IfCond) { 5538 IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy, 5539 /*isSigned=*/true); 5540 } else { 5541 IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1); 5542 } 5543 5544 LValue LBLVal = CGF.EmitLValueForField( 5545 Result.TDBase, 5546 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound)); 5547 const auto *LBVar = 5548 cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl()); 5549 CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(), LBLVal.getQuals(), 5550 /*IsInitializer=*/true); 5551 LValue UBLVal = CGF.EmitLValueForField( 5552 Result.TDBase, 5553 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound)); 5554 const auto *UBVar = 5555 cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl()); 5556 CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(), UBLVal.getQuals(), 5557 /*IsInitializer=*/true); 5558 LValue StLVal = CGF.EmitLValueForField( 5559 Result.TDBase, 5560 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride)); 5561 const auto *StVar = 5562 cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl()); 5563 CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(), StLVal.getQuals(), 5564 /*IsInitializer=*/true); 5565 // Store reductions address. 5566 LValue RedLVal = CGF.EmitLValueForField( 5567 Result.TDBase, 5568 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions)); 5569 if (Data.Reductions) { 5570 CGF.EmitStoreOfScalar(Data.Reductions, RedLVal); 5571 } else { 5572 CGF.EmitNullInitialization(RedLVal.getAddress(), 5573 CGF.getContext().VoidPtrTy); 5574 } 5575 enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 }; 5576 llvm::Value *TaskArgs[] = { 5577 UpLoc, 5578 ThreadID, 5579 Result.NewTask, 5580 IfVal, 5581 LBLVal.getPointer(), 5582 UBLVal.getPointer(), 5583 CGF.EmitLoadOfScalar(StLVal, Loc), 5584 llvm::ConstantInt::getSigned( 5585 CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler 5586 llvm::ConstantInt::getSigned( 5587 CGF.IntTy, Data.Schedule.getPointer() 5588 ? Data.Schedule.getInt() ? NumTasks : Grainsize 5589 : NoSchedule), 5590 Data.Schedule.getPointer() 5591 ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty, 5592 /*isSigned=*/false) 5593 : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0), 5594 Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5595 Result.TaskDupFn, CGF.VoidPtrTy) 5596 : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)}; 5597 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_taskloop), TaskArgs); 5598 } 5599 5600 /// Emit reduction operation for each element of array (required for 5601 /// array sections) LHS op = RHS. 5602 /// \param Type Type of array. 5603 /// \param LHSVar Variable on the left side of the reduction operation 5604 /// (references element of array in original variable). 5605 /// \param RHSVar Variable on the right side of the reduction operation 5606 /// (references element of array in original variable). 5607 /// \param RedOpGen Generator of reduction operation with use of LHSVar and 5608 /// RHSVar. 5609 static void EmitOMPAggregateReduction( 5610 CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar, 5611 const VarDecl *RHSVar, 5612 const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *, 5613 const Expr *, const Expr *)> &RedOpGen, 5614 const Expr *XExpr = nullptr, const Expr *EExpr = nullptr, 5615 const Expr *UpExpr = nullptr) { 5616 // Perform element-by-element initialization. 5617 QualType ElementTy; 5618 Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar); 5619 Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar); 5620 5621 // Drill down to the base element type on both arrays. 5622 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 5623 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr); 5624 5625 llvm::Value *RHSBegin = RHSAddr.getPointer(); 5626 llvm::Value *LHSBegin = LHSAddr.getPointer(); 5627 // Cast from pointer to array type to pointer to single element. 5628 llvm::Value *LHSEnd = CGF.Builder.CreateGEP(LHSBegin, NumElements); 5629 // The basic structure here is a while-do loop. 5630 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body"); 5631 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done"); 5632 llvm::Value *IsEmpty = 5633 CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty"); 5634 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5635 5636 // Enter the loop body, making that address the current address. 5637 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5638 CGF.EmitBlock(BodyBB); 5639 5640 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 5641 5642 llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI( 5643 RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast"); 5644 RHSElementPHI->addIncoming(RHSBegin, EntryBB); 5645 Address RHSElementCurrent = 5646 Address(RHSElementPHI, 5647 RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5648 5649 llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI( 5650 LHSBegin->getType(), 2, "omp.arraycpy.destElementPast"); 5651 LHSElementPHI->addIncoming(LHSBegin, EntryBB); 5652 Address LHSElementCurrent = 5653 Address(LHSElementPHI, 5654 LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5655 5656 // Emit copy. 5657 CodeGenFunction::OMPPrivateScope Scope(CGF); 5658 Scope.addPrivate(LHSVar, [=]() { return LHSElementCurrent; }); 5659 Scope.addPrivate(RHSVar, [=]() { return RHSElementCurrent; }); 5660 Scope.Privatize(); 5661 RedOpGen(CGF, XExpr, EExpr, UpExpr); 5662 Scope.ForceCleanup(); 5663 5664 // Shift the address forward by one element. 5665 llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32( 5666 LHSElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 5667 llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32( 5668 RHSElementPHI, /*Idx0=*/1, "omp.arraycpy.src.element"); 5669 // Check whether we've reached the end. 5670 llvm::Value *Done = 5671 CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done"); 5672 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 5673 LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock()); 5674 RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock()); 5675 5676 // Done. 5677 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5678 } 5679 5680 /// Emit reduction combiner. If the combiner is a simple expression emit it as 5681 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of 5682 /// UDR combiner function. 5683 static void emitReductionCombiner(CodeGenFunction &CGF, 5684 const Expr *ReductionOp) { 5685 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 5686 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 5687 if (const auto *DRE = 5688 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 5689 if (const auto *DRD = 5690 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) { 5691 std::pair<llvm::Function *, llvm::Function *> Reduction = 5692 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 5693 RValue Func = RValue::get(Reduction.first); 5694 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 5695 CGF.EmitIgnoredExpr(ReductionOp); 5696 return; 5697 } 5698 CGF.EmitIgnoredExpr(ReductionOp); 5699 } 5700 5701 llvm::Function *CGOpenMPRuntime::emitReductionFunction( 5702 SourceLocation Loc, llvm::Type *ArgsType, ArrayRef<const Expr *> Privates, 5703 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 5704 ArrayRef<const Expr *> ReductionOps) { 5705 ASTContext &C = CGM.getContext(); 5706 5707 // void reduction_func(void *LHSArg, void *RHSArg); 5708 FunctionArgList Args; 5709 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5710 ImplicitParamDecl::Other); 5711 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5712 ImplicitParamDecl::Other); 5713 Args.push_back(&LHSArg); 5714 Args.push_back(&RHSArg); 5715 const auto &CGFI = 5716 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5717 std::string Name = getName({"omp", "reduction", "reduction_func"}); 5718 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 5719 llvm::GlobalValue::InternalLinkage, Name, 5720 &CGM.getModule()); 5721 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 5722 Fn->setDoesNotRecurse(); 5723 CodeGenFunction CGF(CGM); 5724 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 5725 5726 // Dst = (void*[n])(LHSArg); 5727 // Src = (void*[n])(RHSArg); 5728 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5729 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 5730 ArgsType), CGF.getPointerAlign()); 5731 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5732 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 5733 ArgsType), CGF.getPointerAlign()); 5734 5735 // ... 5736 // *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]); 5737 // ... 5738 CodeGenFunction::OMPPrivateScope Scope(CGF); 5739 auto IPriv = Privates.begin(); 5740 unsigned Idx = 0; 5741 for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) { 5742 const auto *RHSVar = 5743 cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl()); 5744 Scope.addPrivate(RHSVar, [&CGF, RHS, Idx, RHSVar]() { 5745 return emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar); 5746 }); 5747 const auto *LHSVar = 5748 cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl()); 5749 Scope.addPrivate(LHSVar, [&CGF, LHS, Idx, LHSVar]() { 5750 return emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar); 5751 }); 5752 QualType PrivTy = (*IPriv)->getType(); 5753 if (PrivTy->isVariablyModifiedType()) { 5754 // Get array size and emit VLA type. 5755 ++Idx; 5756 Address Elem = CGF.Builder.CreateConstArrayGEP(LHS, Idx); 5757 llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem); 5758 const VariableArrayType *VLA = 5759 CGF.getContext().getAsVariableArrayType(PrivTy); 5760 const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr()); 5761 CodeGenFunction::OpaqueValueMapping OpaqueMap( 5762 CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy))); 5763 CGF.EmitVariablyModifiedType(PrivTy); 5764 } 5765 } 5766 Scope.Privatize(); 5767 IPriv = Privates.begin(); 5768 auto ILHS = LHSExprs.begin(); 5769 auto IRHS = RHSExprs.begin(); 5770 for (const Expr *E : ReductionOps) { 5771 if ((*IPriv)->getType()->isArrayType()) { 5772 // Emit reduction for array section. 5773 const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5774 const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5775 EmitOMPAggregateReduction( 5776 CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5777 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5778 emitReductionCombiner(CGF, E); 5779 }); 5780 } else { 5781 // Emit reduction for array subscript or single variable. 5782 emitReductionCombiner(CGF, E); 5783 } 5784 ++IPriv; 5785 ++ILHS; 5786 ++IRHS; 5787 } 5788 Scope.ForceCleanup(); 5789 CGF.FinishFunction(); 5790 return Fn; 5791 } 5792 5793 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF, 5794 const Expr *ReductionOp, 5795 const Expr *PrivateRef, 5796 const DeclRefExpr *LHS, 5797 const DeclRefExpr *RHS) { 5798 if (PrivateRef->getType()->isArrayType()) { 5799 // Emit reduction for array section. 5800 const auto *LHSVar = cast<VarDecl>(LHS->getDecl()); 5801 const auto *RHSVar = cast<VarDecl>(RHS->getDecl()); 5802 EmitOMPAggregateReduction( 5803 CGF, PrivateRef->getType(), LHSVar, RHSVar, 5804 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5805 emitReductionCombiner(CGF, ReductionOp); 5806 }); 5807 } else { 5808 // Emit reduction for array subscript or single variable. 5809 emitReductionCombiner(CGF, ReductionOp); 5810 } 5811 } 5812 5813 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc, 5814 ArrayRef<const Expr *> Privates, 5815 ArrayRef<const Expr *> LHSExprs, 5816 ArrayRef<const Expr *> RHSExprs, 5817 ArrayRef<const Expr *> ReductionOps, 5818 ReductionOptionsTy Options) { 5819 if (!CGF.HaveInsertPoint()) 5820 return; 5821 5822 bool WithNowait = Options.WithNowait; 5823 bool SimpleReduction = Options.SimpleReduction; 5824 5825 // Next code should be emitted for reduction: 5826 // 5827 // static kmp_critical_name lock = { 0 }; 5828 // 5829 // void reduce_func(void *lhs[<n>], void *rhs[<n>]) { 5830 // *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]); 5831 // ... 5832 // *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1], 5833 // *(Type<n>-1*)rhs[<n>-1]); 5834 // } 5835 // 5836 // ... 5837 // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]}; 5838 // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5839 // RedList, reduce_func, &<lock>)) { 5840 // case 1: 5841 // ... 5842 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5843 // ... 5844 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5845 // break; 5846 // case 2: 5847 // ... 5848 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5849 // ... 5850 // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);] 5851 // break; 5852 // default:; 5853 // } 5854 // 5855 // if SimpleReduction is true, only the next code is generated: 5856 // ... 5857 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5858 // ... 5859 5860 ASTContext &C = CGM.getContext(); 5861 5862 if (SimpleReduction) { 5863 CodeGenFunction::RunCleanupsScope Scope(CGF); 5864 auto IPriv = Privates.begin(); 5865 auto ILHS = LHSExprs.begin(); 5866 auto IRHS = RHSExprs.begin(); 5867 for (const Expr *E : ReductionOps) { 5868 emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5869 cast<DeclRefExpr>(*IRHS)); 5870 ++IPriv; 5871 ++ILHS; 5872 ++IRHS; 5873 } 5874 return; 5875 } 5876 5877 // 1. Build a list of reduction variables. 5878 // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]}; 5879 auto Size = RHSExprs.size(); 5880 for (const Expr *E : Privates) { 5881 if (E->getType()->isVariablyModifiedType()) 5882 // Reserve place for array size. 5883 ++Size; 5884 } 5885 llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size); 5886 QualType ReductionArrayTy = 5887 C.getConstantArrayType(C.VoidPtrTy, ArraySize, nullptr, ArrayType::Normal, 5888 /*IndexTypeQuals=*/0); 5889 Address ReductionList = 5890 CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list"); 5891 auto IPriv = Privates.begin(); 5892 unsigned Idx = 0; 5893 for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) { 5894 Address Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 5895 CGF.Builder.CreateStore( 5896 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5897 CGF.EmitLValue(RHSExprs[I]).getPointer(), CGF.VoidPtrTy), 5898 Elem); 5899 if ((*IPriv)->getType()->isVariablyModifiedType()) { 5900 // Store array size. 5901 ++Idx; 5902 Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx); 5903 llvm::Value *Size = CGF.Builder.CreateIntCast( 5904 CGF.getVLASize( 5905 CGF.getContext().getAsVariableArrayType((*IPriv)->getType())) 5906 .NumElts, 5907 CGF.SizeTy, /*isSigned=*/false); 5908 CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy), 5909 Elem); 5910 } 5911 } 5912 5913 // 2. Emit reduce_func(). 5914 llvm::Function *ReductionFn = emitReductionFunction( 5915 Loc, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), Privates, 5916 LHSExprs, RHSExprs, ReductionOps); 5917 5918 // 3. Create static kmp_critical_name lock = { 0 }; 5919 std::string Name = getName({"reduction"}); 5920 llvm::Value *Lock = getCriticalRegionLock(Name); 5921 5922 // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5923 // RedList, reduce_func, &<lock>); 5924 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE); 5925 llvm::Value *ThreadId = getThreadID(CGF, Loc); 5926 llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy); 5927 llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5928 ReductionList.getPointer(), CGF.VoidPtrTy); 5929 llvm::Value *Args[] = { 5930 IdentTLoc, // ident_t *<loc> 5931 ThreadId, // i32 <gtid> 5932 CGF.Builder.getInt32(RHSExprs.size()), // i32 <n> 5933 ReductionArrayTySize, // size_type sizeof(RedList) 5934 RL, // void *RedList 5935 ReductionFn, // void (*) (void *, void *) <reduce_func> 5936 Lock // kmp_critical_name *&<lock> 5937 }; 5938 llvm::Value *Res = CGF.EmitRuntimeCall( 5939 createRuntimeFunction(WithNowait ? OMPRTL__kmpc_reduce_nowait 5940 : OMPRTL__kmpc_reduce), 5941 Args); 5942 5943 // 5. Build switch(res) 5944 llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default"); 5945 llvm::SwitchInst *SwInst = 5946 CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2); 5947 5948 // 6. Build case 1: 5949 // ... 5950 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5951 // ... 5952 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5953 // break; 5954 llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1"); 5955 SwInst->addCase(CGF.Builder.getInt32(1), Case1BB); 5956 CGF.EmitBlock(Case1BB); 5957 5958 // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5959 llvm::Value *EndArgs[] = { 5960 IdentTLoc, // ident_t *<loc> 5961 ThreadId, // i32 <gtid> 5962 Lock // kmp_critical_name *&<lock> 5963 }; 5964 auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps]( 5965 CodeGenFunction &CGF, PrePostActionTy &Action) { 5966 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5967 auto IPriv = Privates.begin(); 5968 auto ILHS = LHSExprs.begin(); 5969 auto IRHS = RHSExprs.begin(); 5970 for (const Expr *E : ReductionOps) { 5971 RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5972 cast<DeclRefExpr>(*IRHS)); 5973 ++IPriv; 5974 ++ILHS; 5975 ++IRHS; 5976 } 5977 }; 5978 RegionCodeGenTy RCG(CodeGen); 5979 CommonActionTy Action( 5980 nullptr, llvm::None, 5981 createRuntimeFunction(WithNowait ? OMPRTL__kmpc_end_reduce_nowait 5982 : OMPRTL__kmpc_end_reduce), 5983 EndArgs); 5984 RCG.setAction(Action); 5985 RCG(CGF); 5986 5987 CGF.EmitBranch(DefaultBB); 5988 5989 // 7. Build case 2: 5990 // ... 5991 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5992 // ... 5993 // break; 5994 llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2"); 5995 SwInst->addCase(CGF.Builder.getInt32(2), Case2BB); 5996 CGF.EmitBlock(Case2BB); 5997 5998 auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps]( 5999 CodeGenFunction &CGF, PrePostActionTy &Action) { 6000 auto ILHS = LHSExprs.begin(); 6001 auto IRHS = RHSExprs.begin(); 6002 auto IPriv = Privates.begin(); 6003 for (const Expr *E : ReductionOps) { 6004 const Expr *XExpr = nullptr; 6005 const Expr *EExpr = nullptr; 6006 const Expr *UpExpr = nullptr; 6007 BinaryOperatorKind BO = BO_Comma; 6008 if (const auto *BO = dyn_cast<BinaryOperator>(E)) { 6009 if (BO->getOpcode() == BO_Assign) { 6010 XExpr = BO->getLHS(); 6011 UpExpr = BO->getRHS(); 6012 } 6013 } 6014 // Try to emit update expression as a simple atomic. 6015 const Expr *RHSExpr = UpExpr; 6016 if (RHSExpr) { 6017 // Analyze RHS part of the whole expression. 6018 if (const auto *ACO = dyn_cast<AbstractConditionalOperator>( 6019 RHSExpr->IgnoreParenImpCasts())) { 6020 // If this is a conditional operator, analyze its condition for 6021 // min/max reduction operator. 6022 RHSExpr = ACO->getCond(); 6023 } 6024 if (const auto *BORHS = 6025 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) { 6026 EExpr = BORHS->getRHS(); 6027 BO = BORHS->getOpcode(); 6028 } 6029 } 6030 if (XExpr) { 6031 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 6032 auto &&AtomicRedGen = [BO, VD, 6033 Loc](CodeGenFunction &CGF, const Expr *XExpr, 6034 const Expr *EExpr, const Expr *UpExpr) { 6035 LValue X = CGF.EmitLValue(XExpr); 6036 RValue E; 6037 if (EExpr) 6038 E = CGF.EmitAnyExpr(EExpr); 6039 CGF.EmitOMPAtomicSimpleUpdateExpr( 6040 X, E, BO, /*IsXLHSInRHSPart=*/true, 6041 llvm::AtomicOrdering::Monotonic, Loc, 6042 [&CGF, UpExpr, VD, Loc](RValue XRValue) { 6043 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 6044 PrivateScope.addPrivate( 6045 VD, [&CGF, VD, XRValue, Loc]() { 6046 Address LHSTemp = CGF.CreateMemTemp(VD->getType()); 6047 CGF.emitOMPSimpleStore( 6048 CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue, 6049 VD->getType().getNonReferenceType(), Loc); 6050 return LHSTemp; 6051 }); 6052 (void)PrivateScope.Privatize(); 6053 return CGF.EmitAnyExpr(UpExpr); 6054 }); 6055 }; 6056 if ((*IPriv)->getType()->isArrayType()) { 6057 // Emit atomic reduction for array section. 6058 const auto *RHSVar = 6059 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 6060 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar, 6061 AtomicRedGen, XExpr, EExpr, UpExpr); 6062 } else { 6063 // Emit atomic reduction for array subscript or single variable. 6064 AtomicRedGen(CGF, XExpr, EExpr, UpExpr); 6065 } 6066 } else { 6067 // Emit as a critical region. 6068 auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *, 6069 const Expr *, const Expr *) { 6070 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6071 std::string Name = RT.getName({"atomic_reduction"}); 6072 RT.emitCriticalRegion( 6073 CGF, Name, 6074 [=](CodeGenFunction &CGF, PrePostActionTy &Action) { 6075 Action.Enter(CGF); 6076 emitReductionCombiner(CGF, E); 6077 }, 6078 Loc); 6079 }; 6080 if ((*IPriv)->getType()->isArrayType()) { 6081 const auto *LHSVar = 6082 cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 6083 const auto *RHSVar = 6084 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 6085 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar, 6086 CritRedGen); 6087 } else { 6088 CritRedGen(CGF, nullptr, nullptr, nullptr); 6089 } 6090 } 6091 ++ILHS; 6092 ++IRHS; 6093 ++IPriv; 6094 } 6095 }; 6096 RegionCodeGenTy AtomicRCG(AtomicCodeGen); 6097 if (!WithNowait) { 6098 // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>); 6099 llvm::Value *EndArgs[] = { 6100 IdentTLoc, // ident_t *<loc> 6101 ThreadId, // i32 <gtid> 6102 Lock // kmp_critical_name *&<lock> 6103 }; 6104 CommonActionTy Action(nullptr, llvm::None, 6105 createRuntimeFunction(OMPRTL__kmpc_end_reduce), 6106 EndArgs); 6107 AtomicRCG.setAction(Action); 6108 AtomicRCG(CGF); 6109 } else { 6110 AtomicRCG(CGF); 6111 } 6112 6113 CGF.EmitBranch(DefaultBB); 6114 CGF.EmitBlock(DefaultBB, /*IsFinished=*/true); 6115 } 6116 6117 /// Generates unique name for artificial threadprivate variables. 6118 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>" 6119 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix, 6120 const Expr *Ref) { 6121 SmallString<256> Buffer; 6122 llvm::raw_svector_ostream Out(Buffer); 6123 const clang::DeclRefExpr *DE; 6124 const VarDecl *D = ::getBaseDecl(Ref, DE); 6125 if (!D) 6126 D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl()); 6127 D = D->getCanonicalDecl(); 6128 std::string Name = CGM.getOpenMPRuntime().getName( 6129 {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)}); 6130 Out << Prefix << Name << "_" 6131 << D->getCanonicalDecl()->getBeginLoc().getRawEncoding(); 6132 return Out.str(); 6133 } 6134 6135 /// Emits reduction initializer function: 6136 /// \code 6137 /// void @.red_init(void* %arg) { 6138 /// %0 = bitcast void* %arg to <type>* 6139 /// store <type> <init>, <type>* %0 6140 /// ret void 6141 /// } 6142 /// \endcode 6143 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM, 6144 SourceLocation Loc, 6145 ReductionCodeGen &RCG, unsigned N) { 6146 ASTContext &C = CGM.getContext(); 6147 FunctionArgList Args; 6148 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6149 ImplicitParamDecl::Other); 6150 Args.emplace_back(&Param); 6151 const auto &FnInfo = 6152 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6153 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6154 std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""}); 6155 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6156 Name, &CGM.getModule()); 6157 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6158 Fn->setDoesNotRecurse(); 6159 CodeGenFunction CGF(CGM); 6160 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6161 Address PrivateAddr = CGF.EmitLoadOfPointer( 6162 CGF.GetAddrOfLocalVar(&Param), 6163 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6164 llvm::Value *Size = nullptr; 6165 // If the size of the reduction item is non-constant, load it from global 6166 // threadprivate variable. 6167 if (RCG.getSizes(N).second) { 6168 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6169 CGF, CGM.getContext().getSizeType(), 6170 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6171 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6172 CGM.getContext().getSizeType(), Loc); 6173 } 6174 RCG.emitAggregateType(CGF, N, Size); 6175 LValue SharedLVal; 6176 // If initializer uses initializer from declare reduction construct, emit a 6177 // pointer to the address of the original reduction item (reuired by reduction 6178 // initializer) 6179 if (RCG.usesReductionInitializer(N)) { 6180 Address SharedAddr = 6181 CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6182 CGF, CGM.getContext().VoidPtrTy, 6183 generateUniqueName(CGM, "reduction", RCG.getRefExpr(N))); 6184 SharedAddr = CGF.EmitLoadOfPointer( 6185 SharedAddr, 6186 CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr()); 6187 SharedLVal = CGF.MakeAddrLValue(SharedAddr, CGM.getContext().VoidPtrTy); 6188 } else { 6189 SharedLVal = CGF.MakeNaturalAlignAddrLValue( 6190 llvm::ConstantPointerNull::get(CGM.VoidPtrTy), 6191 CGM.getContext().VoidPtrTy); 6192 } 6193 // Emit the initializer: 6194 // %0 = bitcast void* %arg to <type>* 6195 // store <type> <init>, <type>* %0 6196 RCG.emitInitialization(CGF, N, PrivateAddr, SharedLVal, 6197 [](CodeGenFunction &) { return false; }); 6198 CGF.FinishFunction(); 6199 return Fn; 6200 } 6201 6202 /// Emits reduction combiner function: 6203 /// \code 6204 /// void @.red_comb(void* %arg0, void* %arg1) { 6205 /// %lhs = bitcast void* %arg0 to <type>* 6206 /// %rhs = bitcast void* %arg1 to <type>* 6207 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs) 6208 /// store <type> %2, <type>* %lhs 6209 /// ret void 6210 /// } 6211 /// \endcode 6212 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM, 6213 SourceLocation Loc, 6214 ReductionCodeGen &RCG, unsigned N, 6215 const Expr *ReductionOp, 6216 const Expr *LHS, const Expr *RHS, 6217 const Expr *PrivateRef) { 6218 ASTContext &C = CGM.getContext(); 6219 const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl()); 6220 const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl()); 6221 FunctionArgList Args; 6222 ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 6223 C.VoidPtrTy, ImplicitParamDecl::Other); 6224 ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6225 ImplicitParamDecl::Other); 6226 Args.emplace_back(&ParamInOut); 6227 Args.emplace_back(&ParamIn); 6228 const auto &FnInfo = 6229 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6230 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6231 std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""}); 6232 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6233 Name, &CGM.getModule()); 6234 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6235 Fn->setDoesNotRecurse(); 6236 CodeGenFunction CGF(CGM); 6237 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6238 llvm::Value *Size = nullptr; 6239 // If the size of the reduction item is non-constant, load it from global 6240 // threadprivate variable. 6241 if (RCG.getSizes(N).second) { 6242 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6243 CGF, CGM.getContext().getSizeType(), 6244 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6245 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6246 CGM.getContext().getSizeType(), Loc); 6247 } 6248 RCG.emitAggregateType(CGF, N, Size); 6249 // Remap lhs and rhs variables to the addresses of the function arguments. 6250 // %lhs = bitcast void* %arg0 to <type>* 6251 // %rhs = bitcast void* %arg1 to <type>* 6252 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 6253 PrivateScope.addPrivate(LHSVD, [&C, &CGF, &ParamInOut, LHSVD]() { 6254 // Pull out the pointer to the variable. 6255 Address PtrAddr = CGF.EmitLoadOfPointer( 6256 CGF.GetAddrOfLocalVar(&ParamInOut), 6257 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6258 return CGF.Builder.CreateElementBitCast( 6259 PtrAddr, CGF.ConvertTypeForMem(LHSVD->getType())); 6260 }); 6261 PrivateScope.addPrivate(RHSVD, [&C, &CGF, &ParamIn, RHSVD]() { 6262 // Pull out the pointer to the variable. 6263 Address PtrAddr = CGF.EmitLoadOfPointer( 6264 CGF.GetAddrOfLocalVar(&ParamIn), 6265 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6266 return CGF.Builder.CreateElementBitCast( 6267 PtrAddr, CGF.ConvertTypeForMem(RHSVD->getType())); 6268 }); 6269 PrivateScope.Privatize(); 6270 // Emit the combiner body: 6271 // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs) 6272 // store <type> %2, <type>* %lhs 6273 CGM.getOpenMPRuntime().emitSingleReductionCombiner( 6274 CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS), 6275 cast<DeclRefExpr>(RHS)); 6276 CGF.FinishFunction(); 6277 return Fn; 6278 } 6279 6280 /// Emits reduction finalizer function: 6281 /// \code 6282 /// void @.red_fini(void* %arg) { 6283 /// %0 = bitcast void* %arg to <type>* 6284 /// <destroy>(<type>* %0) 6285 /// ret void 6286 /// } 6287 /// \endcode 6288 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM, 6289 SourceLocation Loc, 6290 ReductionCodeGen &RCG, unsigned N) { 6291 if (!RCG.needCleanups(N)) 6292 return nullptr; 6293 ASTContext &C = CGM.getContext(); 6294 FunctionArgList Args; 6295 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6296 ImplicitParamDecl::Other); 6297 Args.emplace_back(&Param); 6298 const auto &FnInfo = 6299 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6300 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6301 std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""}); 6302 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6303 Name, &CGM.getModule()); 6304 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6305 Fn->setDoesNotRecurse(); 6306 CodeGenFunction CGF(CGM); 6307 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6308 Address PrivateAddr = CGF.EmitLoadOfPointer( 6309 CGF.GetAddrOfLocalVar(&Param), 6310 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6311 llvm::Value *Size = nullptr; 6312 // If the size of the reduction item is non-constant, load it from global 6313 // threadprivate variable. 6314 if (RCG.getSizes(N).second) { 6315 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6316 CGF, CGM.getContext().getSizeType(), 6317 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6318 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6319 CGM.getContext().getSizeType(), Loc); 6320 } 6321 RCG.emitAggregateType(CGF, N, Size); 6322 // Emit the finalizer body: 6323 // <destroy>(<type>* %0) 6324 RCG.emitCleanups(CGF, N, PrivateAddr); 6325 CGF.FinishFunction(); 6326 return Fn; 6327 } 6328 6329 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit( 6330 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 6331 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 6332 if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty()) 6333 return nullptr; 6334 6335 // Build typedef struct: 6336 // kmp_task_red_input { 6337 // void *reduce_shar; // shared reduction item 6338 // size_t reduce_size; // size of data item 6339 // void *reduce_init; // data initialization routine 6340 // void *reduce_fini; // data finalization routine 6341 // void *reduce_comb; // data combiner routine 6342 // kmp_task_red_flags_t flags; // flags for additional info from compiler 6343 // } kmp_task_red_input_t; 6344 ASTContext &C = CGM.getContext(); 6345 RecordDecl *RD = C.buildImplicitRecord("kmp_task_red_input_t"); 6346 RD->startDefinition(); 6347 const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6348 const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType()); 6349 const FieldDecl *InitFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6350 const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6351 const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6352 const FieldDecl *FlagsFD = addFieldToRecordDecl( 6353 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false)); 6354 RD->completeDefinition(); 6355 QualType RDType = C.getRecordType(RD); 6356 unsigned Size = Data.ReductionVars.size(); 6357 llvm::APInt ArraySize(/*numBits=*/64, Size); 6358 QualType ArrayRDType = C.getConstantArrayType( 6359 RDType, ArraySize, nullptr, ArrayType::Normal, /*IndexTypeQuals=*/0); 6360 // kmp_task_red_input_t .rd_input.[Size]; 6361 Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input."); 6362 ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionCopies, 6363 Data.ReductionOps); 6364 for (unsigned Cnt = 0; Cnt < Size; ++Cnt) { 6365 // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt]; 6366 llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0), 6367 llvm::ConstantInt::get(CGM.SizeTy, Cnt)}; 6368 llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP( 6369 TaskRedInput.getPointer(), Idxs, 6370 /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc, 6371 ".rd_input.gep."); 6372 LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType); 6373 // ElemLVal.reduce_shar = &Shareds[Cnt]; 6374 LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD); 6375 RCG.emitSharedLValue(CGF, Cnt); 6376 llvm::Value *CastedShared = 6377 CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer()); 6378 CGF.EmitStoreOfScalar(CastedShared, SharedLVal); 6379 RCG.emitAggregateType(CGF, Cnt); 6380 llvm::Value *SizeValInChars; 6381 llvm::Value *SizeVal; 6382 std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt); 6383 // We use delayed creation/initialization for VLAs, array sections and 6384 // custom reduction initializations. It is required because runtime does not 6385 // provide the way to pass the sizes of VLAs/array sections to 6386 // initializer/combiner/finalizer functions and does not pass the pointer to 6387 // original reduction item to the initializer. Instead threadprivate global 6388 // variables are used to store these values and use them in the functions. 6389 bool DelayedCreation = !!SizeVal; 6390 SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy, 6391 /*isSigned=*/false); 6392 LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD); 6393 CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal); 6394 // ElemLVal.reduce_init = init; 6395 LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD); 6396 llvm::Value *InitAddr = 6397 CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt)); 6398 CGF.EmitStoreOfScalar(InitAddr, InitLVal); 6399 DelayedCreation = DelayedCreation || RCG.usesReductionInitializer(Cnt); 6400 // ElemLVal.reduce_fini = fini; 6401 LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD); 6402 llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt); 6403 llvm::Value *FiniAddr = Fini 6404 ? CGF.EmitCastToVoidPtr(Fini) 6405 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy); 6406 CGF.EmitStoreOfScalar(FiniAddr, FiniLVal); 6407 // ElemLVal.reduce_comb = comb; 6408 LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD); 6409 llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction( 6410 CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt], 6411 RHSExprs[Cnt], Data.ReductionCopies[Cnt])); 6412 CGF.EmitStoreOfScalar(CombAddr, CombLVal); 6413 // ElemLVal.flags = 0; 6414 LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD); 6415 if (DelayedCreation) { 6416 CGF.EmitStoreOfScalar( 6417 llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*isSigned=*/true), 6418 FlagsLVal); 6419 } else 6420 CGF.EmitNullInitialization(FlagsLVal.getAddress(), FlagsLVal.getType()); 6421 } 6422 // Build call void *__kmpc_task_reduction_init(int gtid, int num_data, void 6423 // *data); 6424 llvm::Value *Args[] = { 6425 CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy, 6426 /*isSigned=*/true), 6427 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true), 6428 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(), 6429 CGM.VoidPtrTy)}; 6430 return CGF.EmitRuntimeCall( 6431 createRuntimeFunction(OMPRTL__kmpc_task_reduction_init), Args); 6432 } 6433 6434 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 6435 SourceLocation Loc, 6436 ReductionCodeGen &RCG, 6437 unsigned N) { 6438 auto Sizes = RCG.getSizes(N); 6439 // Emit threadprivate global variable if the type is non-constant 6440 // (Sizes.second = nullptr). 6441 if (Sizes.second) { 6442 llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy, 6443 /*isSigned=*/false); 6444 Address SizeAddr = getAddrOfArtificialThreadPrivate( 6445 CGF, CGM.getContext().getSizeType(), 6446 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6447 CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false); 6448 } 6449 // Store address of the original reduction item if custom initializer is used. 6450 if (RCG.usesReductionInitializer(N)) { 6451 Address SharedAddr = getAddrOfArtificialThreadPrivate( 6452 CGF, CGM.getContext().VoidPtrTy, 6453 generateUniqueName(CGM, "reduction", RCG.getRefExpr(N))); 6454 CGF.Builder.CreateStore( 6455 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6456 RCG.getSharedLValue(N).getPointer(), CGM.VoidPtrTy), 6457 SharedAddr, /*IsVolatile=*/false); 6458 } 6459 } 6460 6461 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF, 6462 SourceLocation Loc, 6463 llvm::Value *ReductionsPtr, 6464 LValue SharedLVal) { 6465 // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 6466 // *d); 6467 llvm::Value *Args[] = { 6468 CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy, 6469 /*isSigned=*/true), 6470 ReductionsPtr, 6471 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(SharedLVal.getPointer(), 6472 CGM.VoidPtrTy)}; 6473 return Address( 6474 CGF.EmitRuntimeCall( 6475 createRuntimeFunction(OMPRTL__kmpc_task_reduction_get_th_data), Args), 6476 SharedLVal.getAlignment()); 6477 } 6478 6479 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 6480 SourceLocation Loc) { 6481 if (!CGF.HaveInsertPoint()) 6482 return; 6483 // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 6484 // global_tid); 6485 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 6486 // Ignore return result until untied tasks are supported. 6487 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskwait), Args); 6488 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 6489 Region->emitUntiedSwitch(CGF); 6490 } 6491 6492 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF, 6493 OpenMPDirectiveKind InnerKind, 6494 const RegionCodeGenTy &CodeGen, 6495 bool HasCancel) { 6496 if (!CGF.HaveInsertPoint()) 6497 return; 6498 InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel); 6499 CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr); 6500 } 6501 6502 namespace { 6503 enum RTCancelKind { 6504 CancelNoreq = 0, 6505 CancelParallel = 1, 6506 CancelLoop = 2, 6507 CancelSections = 3, 6508 CancelTaskgroup = 4 6509 }; 6510 } // anonymous namespace 6511 6512 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) { 6513 RTCancelKind CancelKind = CancelNoreq; 6514 if (CancelRegion == OMPD_parallel) 6515 CancelKind = CancelParallel; 6516 else if (CancelRegion == OMPD_for) 6517 CancelKind = CancelLoop; 6518 else if (CancelRegion == OMPD_sections) 6519 CancelKind = CancelSections; 6520 else { 6521 assert(CancelRegion == OMPD_taskgroup); 6522 CancelKind = CancelTaskgroup; 6523 } 6524 return CancelKind; 6525 } 6526 6527 void CGOpenMPRuntime::emitCancellationPointCall( 6528 CodeGenFunction &CGF, SourceLocation Loc, 6529 OpenMPDirectiveKind CancelRegion) { 6530 if (!CGF.HaveInsertPoint()) 6531 return; 6532 // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 6533 // global_tid, kmp_int32 cncl_kind); 6534 if (auto *OMPRegionInfo = 6535 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6536 // For 'cancellation point taskgroup', the task region info may not have a 6537 // cancel. This may instead happen in another adjacent task. 6538 if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) { 6539 llvm::Value *Args[] = { 6540 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 6541 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6542 // Ignore return result until untied tasks are supported. 6543 llvm::Value *Result = CGF.EmitRuntimeCall( 6544 createRuntimeFunction(OMPRTL__kmpc_cancellationpoint), Args); 6545 // if (__kmpc_cancellationpoint()) { 6546 // exit from construct; 6547 // } 6548 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6549 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6550 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6551 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6552 CGF.EmitBlock(ExitBB); 6553 // exit from construct; 6554 CodeGenFunction::JumpDest CancelDest = 6555 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6556 CGF.EmitBranchThroughCleanup(CancelDest); 6557 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6558 } 6559 } 6560 } 6561 6562 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc, 6563 const Expr *IfCond, 6564 OpenMPDirectiveKind CancelRegion) { 6565 if (!CGF.HaveInsertPoint()) 6566 return; 6567 // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 6568 // kmp_int32 cncl_kind); 6569 if (auto *OMPRegionInfo = 6570 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6571 auto &&ThenGen = [Loc, CancelRegion, OMPRegionInfo](CodeGenFunction &CGF, 6572 PrePostActionTy &) { 6573 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6574 llvm::Value *Args[] = { 6575 RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc), 6576 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6577 // Ignore return result until untied tasks are supported. 6578 llvm::Value *Result = CGF.EmitRuntimeCall( 6579 RT.createRuntimeFunction(OMPRTL__kmpc_cancel), Args); 6580 // if (__kmpc_cancel()) { 6581 // exit from construct; 6582 // } 6583 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6584 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6585 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6586 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6587 CGF.EmitBlock(ExitBB); 6588 // exit from construct; 6589 CodeGenFunction::JumpDest CancelDest = 6590 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6591 CGF.EmitBranchThroughCleanup(CancelDest); 6592 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6593 }; 6594 if (IfCond) { 6595 emitOMPIfClause(CGF, IfCond, ThenGen, 6596 [](CodeGenFunction &, PrePostActionTy &) {}); 6597 } else { 6598 RegionCodeGenTy ThenRCG(ThenGen); 6599 ThenRCG(CGF); 6600 } 6601 } 6602 } 6603 6604 void CGOpenMPRuntime::emitTargetOutlinedFunction( 6605 const OMPExecutableDirective &D, StringRef ParentName, 6606 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6607 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6608 assert(!ParentName.empty() && "Invalid target region parent name!"); 6609 HasEmittedTargetRegion = true; 6610 emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID, 6611 IsOffloadEntry, CodeGen); 6612 } 6613 6614 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper( 6615 const OMPExecutableDirective &D, StringRef ParentName, 6616 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6617 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6618 // Create a unique name for the entry function using the source location 6619 // information of the current target region. The name will be something like: 6620 // 6621 // __omp_offloading_DD_FFFF_PP_lBB 6622 // 6623 // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the 6624 // mangled name of the function that encloses the target region and BB is the 6625 // line number of the target region. 6626 6627 unsigned DeviceID; 6628 unsigned FileID; 6629 unsigned Line; 6630 getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID, 6631 Line); 6632 SmallString<64> EntryFnName; 6633 { 6634 llvm::raw_svector_ostream OS(EntryFnName); 6635 OS << "__omp_offloading" << llvm::format("_%x", DeviceID) 6636 << llvm::format("_%x_", FileID) << ParentName << "_l" << Line; 6637 } 6638 6639 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 6640 6641 CodeGenFunction CGF(CGM, true); 6642 CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName); 6643 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6644 6645 OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS); 6646 6647 // If this target outline function is not an offload entry, we don't need to 6648 // register it. 6649 if (!IsOffloadEntry) 6650 return; 6651 6652 // The target region ID is used by the runtime library to identify the current 6653 // target region, so it only has to be unique and not necessarily point to 6654 // anything. It could be the pointer to the outlined function that implements 6655 // the target region, but we aren't using that so that the compiler doesn't 6656 // need to keep that, and could therefore inline the host function if proven 6657 // worthwhile during optimization. In the other hand, if emitting code for the 6658 // device, the ID has to be the function address so that it can retrieved from 6659 // the offloading entry and launched by the runtime library. We also mark the 6660 // outlined function to have external linkage in case we are emitting code for 6661 // the device, because these functions will be entry points to the device. 6662 6663 if (CGM.getLangOpts().OpenMPIsDevice) { 6664 OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy); 6665 OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 6666 OutlinedFn->setDSOLocal(false); 6667 } else { 6668 std::string Name = getName({EntryFnName, "region_id"}); 6669 OutlinedFnID = new llvm::GlobalVariable( 6670 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 6671 llvm::GlobalValue::WeakAnyLinkage, 6672 llvm::Constant::getNullValue(CGM.Int8Ty), Name); 6673 } 6674 6675 // Register the information for the entry associated with this target region. 6676 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 6677 DeviceID, FileID, ParentName, Line, OutlinedFn, OutlinedFnID, 6678 OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion); 6679 } 6680 6681 /// Checks if the expression is constant or does not have non-trivial function 6682 /// calls. 6683 static bool isTrivial(ASTContext &Ctx, const Expr * E) { 6684 // We can skip constant expressions. 6685 // We can skip expressions with trivial calls or simple expressions. 6686 return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) || 6687 !E->hasNonTrivialCall(Ctx)) && 6688 !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true); 6689 } 6690 6691 const Stmt *CGOpenMPRuntime::getSingleCompoundChild(ASTContext &Ctx, 6692 const Stmt *Body) { 6693 const Stmt *Child = Body->IgnoreContainers(); 6694 while (const auto *C = dyn_cast_or_null<CompoundStmt>(Child)) { 6695 Child = nullptr; 6696 for (const Stmt *S : C->body()) { 6697 if (const auto *E = dyn_cast<Expr>(S)) { 6698 if (isTrivial(Ctx, E)) 6699 continue; 6700 } 6701 // Some of the statements can be ignored. 6702 if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) || 6703 isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S)) 6704 continue; 6705 // Analyze declarations. 6706 if (const auto *DS = dyn_cast<DeclStmt>(S)) { 6707 if (llvm::all_of(DS->decls(), [&Ctx](const Decl *D) { 6708 if (isa<EmptyDecl>(D) || isa<DeclContext>(D) || 6709 isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) || 6710 isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) || 6711 isa<UsingDirectiveDecl>(D) || 6712 isa<OMPDeclareReductionDecl>(D) || 6713 isa<OMPThreadPrivateDecl>(D) || isa<OMPAllocateDecl>(D)) 6714 return true; 6715 const auto *VD = dyn_cast<VarDecl>(D); 6716 if (!VD) 6717 return false; 6718 return VD->isConstexpr() || 6719 ((VD->getType().isTrivialType(Ctx) || 6720 VD->getType()->isReferenceType()) && 6721 (!VD->hasInit() || isTrivial(Ctx, VD->getInit()))); 6722 })) 6723 continue; 6724 } 6725 // Found multiple children - cannot get the one child only. 6726 if (Child) 6727 return nullptr; 6728 Child = S; 6729 } 6730 if (Child) 6731 Child = Child->IgnoreContainers(); 6732 } 6733 return Child; 6734 } 6735 6736 /// Emit the number of teams for a target directive. Inspect the num_teams 6737 /// clause associated with a teams construct combined or closely nested 6738 /// with the target directive. 6739 /// 6740 /// Emit a team of size one for directives such as 'target parallel' that 6741 /// have no associated teams construct. 6742 /// 6743 /// Otherwise, return nullptr. 6744 static llvm::Value * 6745 emitNumTeamsForTargetDirective(CodeGenFunction &CGF, 6746 const OMPExecutableDirective &D) { 6747 assert(!CGF.getLangOpts().OpenMPIsDevice && 6748 "Clauses associated with the teams directive expected to be emitted " 6749 "only for the host!"); 6750 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6751 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 6752 "Expected target-based executable directive."); 6753 CGBuilderTy &Bld = CGF.Builder; 6754 switch (DirectiveKind) { 6755 case OMPD_target: { 6756 const auto *CS = D.getInnermostCapturedStmt(); 6757 const auto *Body = 6758 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 6759 const Stmt *ChildStmt = 6760 CGOpenMPRuntime::getSingleCompoundChild(CGF.getContext(), Body); 6761 if (const auto *NestedDir = 6762 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 6763 if (isOpenMPTeamsDirective(NestedDir->getDirectiveKind())) { 6764 if (NestedDir->hasClausesOfKind<OMPNumTeamsClause>()) { 6765 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6766 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6767 const Expr *NumTeams = 6768 NestedDir->getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6769 llvm::Value *NumTeamsVal = 6770 CGF.EmitScalarExpr(NumTeams, 6771 /*IgnoreResultAssign*/ true); 6772 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6773 /*isSigned=*/true); 6774 } 6775 return Bld.getInt32(0); 6776 } 6777 if (isOpenMPParallelDirective(NestedDir->getDirectiveKind()) || 6778 isOpenMPSimdDirective(NestedDir->getDirectiveKind())) 6779 return Bld.getInt32(1); 6780 return Bld.getInt32(0); 6781 } 6782 return nullptr; 6783 } 6784 case OMPD_target_teams: 6785 case OMPD_target_teams_distribute: 6786 case OMPD_target_teams_distribute_simd: 6787 case OMPD_target_teams_distribute_parallel_for: 6788 case OMPD_target_teams_distribute_parallel_for_simd: { 6789 if (D.hasClausesOfKind<OMPNumTeamsClause>()) { 6790 CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF); 6791 const Expr *NumTeams = 6792 D.getSingleClause<OMPNumTeamsClause>()->getNumTeams(); 6793 llvm::Value *NumTeamsVal = 6794 CGF.EmitScalarExpr(NumTeams, 6795 /*IgnoreResultAssign*/ true); 6796 return Bld.CreateIntCast(NumTeamsVal, CGF.Int32Ty, 6797 /*isSigned=*/true); 6798 } 6799 return Bld.getInt32(0); 6800 } 6801 case OMPD_target_parallel: 6802 case OMPD_target_parallel_for: 6803 case OMPD_target_parallel_for_simd: 6804 case OMPD_target_simd: 6805 return Bld.getInt32(1); 6806 case OMPD_parallel: 6807 case OMPD_for: 6808 case OMPD_parallel_for: 6809 case OMPD_parallel_sections: 6810 case OMPD_for_simd: 6811 case OMPD_parallel_for_simd: 6812 case OMPD_cancel: 6813 case OMPD_cancellation_point: 6814 case OMPD_ordered: 6815 case OMPD_threadprivate: 6816 case OMPD_allocate: 6817 case OMPD_task: 6818 case OMPD_simd: 6819 case OMPD_sections: 6820 case OMPD_section: 6821 case OMPD_single: 6822 case OMPD_master: 6823 case OMPD_critical: 6824 case OMPD_taskyield: 6825 case OMPD_barrier: 6826 case OMPD_taskwait: 6827 case OMPD_taskgroup: 6828 case OMPD_atomic: 6829 case OMPD_flush: 6830 case OMPD_teams: 6831 case OMPD_target_data: 6832 case OMPD_target_exit_data: 6833 case OMPD_target_enter_data: 6834 case OMPD_distribute: 6835 case OMPD_distribute_simd: 6836 case OMPD_distribute_parallel_for: 6837 case OMPD_distribute_parallel_for_simd: 6838 case OMPD_teams_distribute: 6839 case OMPD_teams_distribute_simd: 6840 case OMPD_teams_distribute_parallel_for: 6841 case OMPD_teams_distribute_parallel_for_simd: 6842 case OMPD_target_update: 6843 case OMPD_declare_simd: 6844 case OMPD_declare_variant: 6845 case OMPD_declare_target: 6846 case OMPD_end_declare_target: 6847 case OMPD_declare_reduction: 6848 case OMPD_declare_mapper: 6849 case OMPD_taskloop: 6850 case OMPD_taskloop_simd: 6851 case OMPD_master_taskloop: 6852 case OMPD_requires: 6853 case OMPD_unknown: 6854 break; 6855 } 6856 llvm_unreachable("Unexpected directive kind."); 6857 } 6858 6859 static llvm::Value *getNumThreads(CodeGenFunction &CGF, const CapturedStmt *CS, 6860 llvm::Value *DefaultThreadLimitVal) { 6861 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6862 CGF.getContext(), CS->getCapturedStmt()); 6863 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6864 if (isOpenMPParallelDirective(Dir->getDirectiveKind())) { 6865 llvm::Value *NumThreads = nullptr; 6866 llvm::Value *CondVal = nullptr; 6867 // Handle if clause. If if clause present, the number of threads is 6868 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 6869 if (Dir->hasClausesOfKind<OMPIfClause>()) { 6870 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6871 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6872 const OMPIfClause *IfClause = nullptr; 6873 for (const auto *C : Dir->getClausesOfKind<OMPIfClause>()) { 6874 if (C->getNameModifier() == OMPD_unknown || 6875 C->getNameModifier() == OMPD_parallel) { 6876 IfClause = C; 6877 break; 6878 } 6879 } 6880 if (IfClause) { 6881 const Expr *Cond = IfClause->getCondition(); 6882 bool Result; 6883 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 6884 if (!Result) 6885 return CGF.Builder.getInt32(1); 6886 } else { 6887 CodeGenFunction::LexicalScope Scope(CGF, Cond->getSourceRange()); 6888 if (const auto *PreInit = 6889 cast_or_null<DeclStmt>(IfClause->getPreInitStmt())) { 6890 for (const auto *I : PreInit->decls()) { 6891 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6892 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6893 } else { 6894 CodeGenFunction::AutoVarEmission Emission = 6895 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6896 CGF.EmitAutoVarCleanups(Emission); 6897 } 6898 } 6899 } 6900 CondVal = CGF.EvaluateExprAsBool(Cond); 6901 } 6902 } 6903 } 6904 // Check the value of num_threads clause iff if clause was not specified 6905 // or is not evaluated to false. 6906 if (Dir->hasClausesOfKind<OMPNumThreadsClause>()) { 6907 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6908 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6909 const auto *NumThreadsClause = 6910 Dir->getSingleClause<OMPNumThreadsClause>(); 6911 CodeGenFunction::LexicalScope Scope( 6912 CGF, NumThreadsClause->getNumThreads()->getSourceRange()); 6913 if (const auto *PreInit = 6914 cast_or_null<DeclStmt>(NumThreadsClause->getPreInitStmt())) { 6915 for (const auto *I : PreInit->decls()) { 6916 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6917 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6918 } else { 6919 CodeGenFunction::AutoVarEmission Emission = 6920 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6921 CGF.EmitAutoVarCleanups(Emission); 6922 } 6923 } 6924 } 6925 NumThreads = CGF.EmitScalarExpr(NumThreadsClause->getNumThreads()); 6926 NumThreads = CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, 6927 /*isSigned=*/false); 6928 if (DefaultThreadLimitVal) 6929 NumThreads = CGF.Builder.CreateSelect( 6930 CGF.Builder.CreateICmpULT(DefaultThreadLimitVal, NumThreads), 6931 DefaultThreadLimitVal, NumThreads); 6932 } else { 6933 NumThreads = DefaultThreadLimitVal ? DefaultThreadLimitVal 6934 : CGF.Builder.getInt32(0); 6935 } 6936 // Process condition of the if clause. 6937 if (CondVal) { 6938 NumThreads = CGF.Builder.CreateSelect(CondVal, NumThreads, 6939 CGF.Builder.getInt32(1)); 6940 } 6941 return NumThreads; 6942 } 6943 if (isOpenMPSimdDirective(Dir->getDirectiveKind())) 6944 return CGF.Builder.getInt32(1); 6945 return DefaultThreadLimitVal; 6946 } 6947 return DefaultThreadLimitVal ? DefaultThreadLimitVal 6948 : CGF.Builder.getInt32(0); 6949 } 6950 6951 /// Emit the number of threads for a target directive. Inspect the 6952 /// thread_limit clause associated with a teams construct combined or closely 6953 /// nested with the target directive. 6954 /// 6955 /// Emit the num_threads clause for directives such as 'target parallel' that 6956 /// have no associated teams construct. 6957 /// 6958 /// Otherwise, return nullptr. 6959 static llvm::Value * 6960 emitNumThreadsForTargetDirective(CodeGenFunction &CGF, 6961 const OMPExecutableDirective &D) { 6962 assert(!CGF.getLangOpts().OpenMPIsDevice && 6963 "Clauses associated with the teams directive expected to be emitted " 6964 "only for the host!"); 6965 OpenMPDirectiveKind DirectiveKind = D.getDirectiveKind(); 6966 assert(isOpenMPTargetExecutionDirective(DirectiveKind) && 6967 "Expected target-based executable directive."); 6968 CGBuilderTy &Bld = CGF.Builder; 6969 llvm::Value *ThreadLimitVal = nullptr; 6970 llvm::Value *NumThreadsVal = nullptr; 6971 switch (DirectiveKind) { 6972 case OMPD_target: { 6973 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 6974 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 6975 return NumThreads; 6976 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 6977 CGF.getContext(), CS->getCapturedStmt()); 6978 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 6979 if (Dir->hasClausesOfKind<OMPThreadLimitClause>()) { 6980 CGOpenMPInnerExprInfo CGInfo(CGF, *CS); 6981 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6982 const auto *ThreadLimitClause = 6983 Dir->getSingleClause<OMPThreadLimitClause>(); 6984 CodeGenFunction::LexicalScope Scope( 6985 CGF, ThreadLimitClause->getThreadLimit()->getSourceRange()); 6986 if (const auto *PreInit = 6987 cast_or_null<DeclStmt>(ThreadLimitClause->getPreInitStmt())) { 6988 for (const auto *I : PreInit->decls()) { 6989 if (!I->hasAttr<OMPCaptureNoInitAttr>()) { 6990 CGF.EmitVarDecl(cast<VarDecl>(*I)); 6991 } else { 6992 CodeGenFunction::AutoVarEmission Emission = 6993 CGF.EmitAutoVarAlloca(cast<VarDecl>(*I)); 6994 CGF.EmitAutoVarCleanups(Emission); 6995 } 6996 } 6997 } 6998 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 6999 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7000 ThreadLimitVal = 7001 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7002 } 7003 if (isOpenMPTeamsDirective(Dir->getDirectiveKind()) && 7004 !isOpenMPDistributeDirective(Dir->getDirectiveKind())) { 7005 CS = Dir->getInnermostCapturedStmt(); 7006 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7007 CGF.getContext(), CS->getCapturedStmt()); 7008 Dir = dyn_cast_or_null<OMPExecutableDirective>(Child); 7009 } 7010 if (Dir && isOpenMPDistributeDirective(Dir->getDirectiveKind()) && 7011 !isOpenMPSimdDirective(Dir->getDirectiveKind())) { 7012 CS = Dir->getInnermostCapturedStmt(); 7013 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7014 return NumThreads; 7015 } 7016 if (Dir && isOpenMPSimdDirective(Dir->getDirectiveKind())) 7017 return Bld.getInt32(1); 7018 } 7019 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 7020 } 7021 case OMPD_target_teams: { 7022 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7023 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7024 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7025 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7026 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7027 ThreadLimitVal = 7028 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7029 } 7030 const CapturedStmt *CS = D.getInnermostCapturedStmt(); 7031 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7032 return NumThreads; 7033 const Stmt *Child = CGOpenMPRuntime::getSingleCompoundChild( 7034 CGF.getContext(), CS->getCapturedStmt()); 7035 if (const auto *Dir = dyn_cast_or_null<OMPExecutableDirective>(Child)) { 7036 if (Dir->getDirectiveKind() == OMPD_distribute) { 7037 CS = Dir->getInnermostCapturedStmt(); 7038 if (llvm::Value *NumThreads = getNumThreads(CGF, CS, ThreadLimitVal)) 7039 return NumThreads; 7040 } 7041 } 7042 return ThreadLimitVal ? ThreadLimitVal : Bld.getInt32(0); 7043 } 7044 case OMPD_target_teams_distribute: 7045 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7046 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7047 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7048 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7049 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7050 ThreadLimitVal = 7051 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7052 } 7053 return getNumThreads(CGF, D.getInnermostCapturedStmt(), ThreadLimitVal); 7054 case OMPD_target_parallel: 7055 case OMPD_target_parallel_for: 7056 case OMPD_target_parallel_for_simd: 7057 case OMPD_target_teams_distribute_parallel_for: 7058 case OMPD_target_teams_distribute_parallel_for_simd: { 7059 llvm::Value *CondVal = nullptr; 7060 // Handle if clause. If if clause present, the number of threads is 7061 // calculated as <cond> ? (<numthreads> ? <numthreads> : 0 ) : 1. 7062 if (D.hasClausesOfKind<OMPIfClause>()) { 7063 const OMPIfClause *IfClause = nullptr; 7064 for (const auto *C : D.getClausesOfKind<OMPIfClause>()) { 7065 if (C->getNameModifier() == OMPD_unknown || 7066 C->getNameModifier() == OMPD_parallel) { 7067 IfClause = C; 7068 break; 7069 } 7070 } 7071 if (IfClause) { 7072 const Expr *Cond = IfClause->getCondition(); 7073 bool Result; 7074 if (Cond->EvaluateAsBooleanCondition(Result, CGF.getContext())) { 7075 if (!Result) 7076 return Bld.getInt32(1); 7077 } else { 7078 CodeGenFunction::RunCleanupsScope Scope(CGF); 7079 CondVal = CGF.EvaluateExprAsBool(Cond); 7080 } 7081 } 7082 } 7083 if (D.hasClausesOfKind<OMPThreadLimitClause>()) { 7084 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 7085 const auto *ThreadLimitClause = D.getSingleClause<OMPThreadLimitClause>(); 7086 llvm::Value *ThreadLimit = CGF.EmitScalarExpr( 7087 ThreadLimitClause->getThreadLimit(), /*IgnoreResultAssign=*/true); 7088 ThreadLimitVal = 7089 Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, /*isSigned=*/false); 7090 } 7091 if (D.hasClausesOfKind<OMPNumThreadsClause>()) { 7092 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF); 7093 const auto *NumThreadsClause = D.getSingleClause<OMPNumThreadsClause>(); 7094 llvm::Value *NumThreads = CGF.EmitScalarExpr( 7095 NumThreadsClause->getNumThreads(), /*IgnoreResultAssign=*/true); 7096 NumThreadsVal = 7097 Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned=*/false); 7098 ThreadLimitVal = ThreadLimitVal 7099 ? Bld.CreateSelect(Bld.CreateICmpULT(NumThreadsVal, 7100 ThreadLimitVal), 7101 NumThreadsVal, ThreadLimitVal) 7102 : NumThreadsVal; 7103 } 7104 if (!ThreadLimitVal) 7105 ThreadLimitVal = Bld.getInt32(0); 7106 if (CondVal) 7107 return Bld.CreateSelect(CondVal, ThreadLimitVal, Bld.getInt32(1)); 7108 return ThreadLimitVal; 7109 } 7110 case OMPD_target_teams_distribute_simd: 7111 case OMPD_target_simd: 7112 return Bld.getInt32(1); 7113 case OMPD_parallel: 7114 case OMPD_for: 7115 case OMPD_parallel_for: 7116 case OMPD_parallel_sections: 7117 case OMPD_for_simd: 7118 case OMPD_parallel_for_simd: 7119 case OMPD_cancel: 7120 case OMPD_cancellation_point: 7121 case OMPD_ordered: 7122 case OMPD_threadprivate: 7123 case OMPD_allocate: 7124 case OMPD_task: 7125 case OMPD_simd: 7126 case OMPD_sections: 7127 case OMPD_section: 7128 case OMPD_single: 7129 case OMPD_master: 7130 case OMPD_critical: 7131 case OMPD_taskyield: 7132 case OMPD_barrier: 7133 case OMPD_taskwait: 7134 case OMPD_taskgroup: 7135 case OMPD_atomic: 7136 case OMPD_flush: 7137 case OMPD_teams: 7138 case OMPD_target_data: 7139 case OMPD_target_exit_data: 7140 case OMPD_target_enter_data: 7141 case OMPD_distribute: 7142 case OMPD_distribute_simd: 7143 case OMPD_distribute_parallel_for: 7144 case OMPD_distribute_parallel_for_simd: 7145 case OMPD_teams_distribute: 7146 case OMPD_teams_distribute_simd: 7147 case OMPD_teams_distribute_parallel_for: 7148 case OMPD_teams_distribute_parallel_for_simd: 7149 case OMPD_target_update: 7150 case OMPD_declare_simd: 7151 case OMPD_declare_variant: 7152 case OMPD_declare_target: 7153 case OMPD_end_declare_target: 7154 case OMPD_declare_reduction: 7155 case OMPD_declare_mapper: 7156 case OMPD_taskloop: 7157 case OMPD_taskloop_simd: 7158 case OMPD_master_taskloop: 7159 case OMPD_requires: 7160 case OMPD_unknown: 7161 break; 7162 } 7163 llvm_unreachable("Unsupported directive kind."); 7164 } 7165 7166 namespace { 7167 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 7168 7169 // Utility to handle information from clauses associated with a given 7170 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause). 7171 // It provides a convenient interface to obtain the information and generate 7172 // code for that information. 7173 class MappableExprsHandler { 7174 public: 7175 /// Values for bit flags used to specify the mapping type for 7176 /// offloading. 7177 enum OpenMPOffloadMappingFlags : uint64_t { 7178 /// No flags 7179 OMP_MAP_NONE = 0x0, 7180 /// Allocate memory on the device and move data from host to device. 7181 OMP_MAP_TO = 0x01, 7182 /// Allocate memory on the device and move data from device to host. 7183 OMP_MAP_FROM = 0x02, 7184 /// Always perform the requested mapping action on the element, even 7185 /// if it was already mapped before. 7186 OMP_MAP_ALWAYS = 0x04, 7187 /// Delete the element from the device environment, ignoring the 7188 /// current reference count associated with the element. 7189 OMP_MAP_DELETE = 0x08, 7190 /// The element being mapped is a pointer-pointee pair; both the 7191 /// pointer and the pointee should be mapped. 7192 OMP_MAP_PTR_AND_OBJ = 0x10, 7193 /// This flags signals that the base address of an entry should be 7194 /// passed to the target kernel as an argument. 7195 OMP_MAP_TARGET_PARAM = 0x20, 7196 /// Signal that the runtime library has to return the device pointer 7197 /// in the current position for the data being mapped. Used when we have the 7198 /// use_device_ptr clause. 7199 OMP_MAP_RETURN_PARAM = 0x40, 7200 /// This flag signals that the reference being passed is a pointer to 7201 /// private data. 7202 OMP_MAP_PRIVATE = 0x80, 7203 /// Pass the element to the device by value. 7204 OMP_MAP_LITERAL = 0x100, 7205 /// Implicit map 7206 OMP_MAP_IMPLICIT = 0x200, 7207 /// Close is a hint to the runtime to allocate memory close to 7208 /// the target device. 7209 OMP_MAP_CLOSE = 0x400, 7210 /// The 16 MSBs of the flags indicate whether the entry is member of some 7211 /// struct/class. 7212 OMP_MAP_MEMBER_OF = 0xffff000000000000, 7213 LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF), 7214 }; 7215 7216 /// Get the offset of the OMP_MAP_MEMBER_OF field. 7217 static unsigned getFlagMemberOffset() { 7218 unsigned Offset = 0; 7219 for (uint64_t Remain = OMP_MAP_MEMBER_OF; !(Remain & 1); 7220 Remain = Remain >> 1) 7221 Offset++; 7222 return Offset; 7223 } 7224 7225 /// Class that associates information with a base pointer to be passed to the 7226 /// runtime library. 7227 class BasePointerInfo { 7228 /// The base pointer. 7229 llvm::Value *Ptr = nullptr; 7230 /// The base declaration that refers to this device pointer, or null if 7231 /// there is none. 7232 const ValueDecl *DevPtrDecl = nullptr; 7233 7234 public: 7235 BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr) 7236 : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {} 7237 llvm::Value *operator*() const { return Ptr; } 7238 const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; } 7239 void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; } 7240 }; 7241 7242 using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>; 7243 using MapValuesArrayTy = SmallVector<llvm::Value *, 4>; 7244 using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>; 7245 7246 /// Map between a struct and the its lowest & highest elements which have been 7247 /// mapped. 7248 /// [ValueDecl *] --> {LE(FieldIndex, Pointer), 7249 /// HE(FieldIndex, Pointer)} 7250 struct StructRangeInfoTy { 7251 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = { 7252 0, Address::invalid()}; 7253 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = { 7254 0, Address::invalid()}; 7255 Address Base = Address::invalid(); 7256 }; 7257 7258 private: 7259 /// Kind that defines how a device pointer has to be returned. 7260 struct MapInfo { 7261 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 7262 OpenMPMapClauseKind MapType = OMPC_MAP_unknown; 7263 ArrayRef<OpenMPMapModifierKind> MapModifiers; 7264 bool ReturnDevicePointer = false; 7265 bool IsImplicit = false; 7266 7267 MapInfo() = default; 7268 MapInfo( 7269 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7270 OpenMPMapClauseKind MapType, 7271 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7272 bool ReturnDevicePointer, bool IsImplicit) 7273 : Components(Components), MapType(MapType), MapModifiers(MapModifiers), 7274 ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit) {} 7275 }; 7276 7277 /// If use_device_ptr is used on a pointer which is a struct member and there 7278 /// is no map information about it, then emission of that entry is deferred 7279 /// until the whole struct has been processed. 7280 struct DeferredDevicePtrEntryTy { 7281 const Expr *IE = nullptr; 7282 const ValueDecl *VD = nullptr; 7283 7284 DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD) 7285 : IE(IE), VD(VD) {} 7286 }; 7287 7288 /// The target directive from where the mappable clauses were extracted. It 7289 /// is either a executable directive or a user-defined mapper directive. 7290 llvm::PointerUnion<const OMPExecutableDirective *, 7291 const OMPDeclareMapperDecl *> 7292 CurDir; 7293 7294 /// Function the directive is being generated for. 7295 CodeGenFunction &CGF; 7296 7297 /// Set of all first private variables in the current directive. 7298 /// bool data is set to true if the variable is implicitly marked as 7299 /// firstprivate, false otherwise. 7300 llvm::DenseMap<CanonicalDeclPtr<const VarDecl>, bool> FirstPrivateDecls; 7301 7302 /// Map between device pointer declarations and their expression components. 7303 /// The key value for declarations in 'this' is null. 7304 llvm::DenseMap< 7305 const ValueDecl *, 7306 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>> 7307 DevPointersMap; 7308 7309 llvm::Value *getExprTypeSize(const Expr *E) const { 7310 QualType ExprTy = E->getType().getCanonicalType(); 7311 7312 // Reference types are ignored for mapping purposes. 7313 if (const auto *RefTy = ExprTy->getAs<ReferenceType>()) 7314 ExprTy = RefTy->getPointeeType().getCanonicalType(); 7315 7316 // Given that an array section is considered a built-in type, we need to 7317 // do the calculation based on the length of the section instead of relying 7318 // on CGF.getTypeSize(E->getType()). 7319 if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) { 7320 QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType( 7321 OAE->getBase()->IgnoreParenImpCasts()) 7322 .getCanonicalType(); 7323 7324 // If there is no length associated with the expression and lower bound is 7325 // not specified too, that means we are using the whole length of the 7326 // base. 7327 if (!OAE->getLength() && OAE->getColonLoc().isValid() && 7328 !OAE->getLowerBound()) 7329 return CGF.getTypeSize(BaseTy); 7330 7331 llvm::Value *ElemSize; 7332 if (const auto *PTy = BaseTy->getAs<PointerType>()) { 7333 ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType()); 7334 } else { 7335 const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr()); 7336 assert(ATy && "Expecting array type if not a pointer type."); 7337 ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType()); 7338 } 7339 7340 // If we don't have a length at this point, that is because we have an 7341 // array section with a single element. 7342 if (!OAE->getLength() && OAE->getColonLoc().isInvalid()) 7343 return ElemSize; 7344 7345 if (const Expr *LenExpr = OAE->getLength()) { 7346 llvm::Value *LengthVal = CGF.EmitScalarExpr(LenExpr); 7347 LengthVal = CGF.EmitScalarConversion(LengthVal, LenExpr->getType(), 7348 CGF.getContext().getSizeType(), 7349 LenExpr->getExprLoc()); 7350 return CGF.Builder.CreateNUWMul(LengthVal, ElemSize); 7351 } 7352 assert(!OAE->getLength() && OAE->getColonLoc().isValid() && 7353 OAE->getLowerBound() && "expected array_section[lb:]."); 7354 // Size = sizetype - lb * elemtype; 7355 llvm::Value *LengthVal = CGF.getTypeSize(BaseTy); 7356 llvm::Value *LBVal = CGF.EmitScalarExpr(OAE->getLowerBound()); 7357 LBVal = CGF.EmitScalarConversion(LBVal, OAE->getLowerBound()->getType(), 7358 CGF.getContext().getSizeType(), 7359 OAE->getLowerBound()->getExprLoc()); 7360 LBVal = CGF.Builder.CreateNUWMul(LBVal, ElemSize); 7361 llvm::Value *Cmp = CGF.Builder.CreateICmpUGT(LengthVal, LBVal); 7362 llvm::Value *TrueVal = CGF.Builder.CreateNUWSub(LengthVal, LBVal); 7363 LengthVal = CGF.Builder.CreateSelect( 7364 Cmp, TrueVal, llvm::ConstantInt::get(CGF.SizeTy, 0)); 7365 return LengthVal; 7366 } 7367 return CGF.getTypeSize(ExprTy); 7368 } 7369 7370 /// Return the corresponding bits for a given map clause modifier. Add 7371 /// a flag marking the map as a pointer if requested. Add a flag marking the 7372 /// map as the first one of a series of maps that relate to the same map 7373 /// expression. 7374 OpenMPOffloadMappingFlags getMapTypeBits( 7375 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers, 7376 bool IsImplicit, bool AddPtrFlag, bool AddIsTargetParamFlag) const { 7377 OpenMPOffloadMappingFlags Bits = 7378 IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE; 7379 switch (MapType) { 7380 case OMPC_MAP_alloc: 7381 case OMPC_MAP_release: 7382 // alloc and release is the default behavior in the runtime library, i.e. 7383 // if we don't pass any bits alloc/release that is what the runtime is 7384 // going to do. Therefore, we don't need to signal anything for these two 7385 // type modifiers. 7386 break; 7387 case OMPC_MAP_to: 7388 Bits |= OMP_MAP_TO; 7389 break; 7390 case OMPC_MAP_from: 7391 Bits |= OMP_MAP_FROM; 7392 break; 7393 case OMPC_MAP_tofrom: 7394 Bits |= OMP_MAP_TO | OMP_MAP_FROM; 7395 break; 7396 case OMPC_MAP_delete: 7397 Bits |= OMP_MAP_DELETE; 7398 break; 7399 case OMPC_MAP_unknown: 7400 llvm_unreachable("Unexpected map type!"); 7401 } 7402 if (AddPtrFlag) 7403 Bits |= OMP_MAP_PTR_AND_OBJ; 7404 if (AddIsTargetParamFlag) 7405 Bits |= OMP_MAP_TARGET_PARAM; 7406 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_always) 7407 != MapModifiers.end()) 7408 Bits |= OMP_MAP_ALWAYS; 7409 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_close) 7410 != MapModifiers.end()) 7411 Bits |= OMP_MAP_CLOSE; 7412 return Bits; 7413 } 7414 7415 /// Return true if the provided expression is a final array section. A 7416 /// final array section, is one whose length can't be proved to be one. 7417 bool isFinalArraySectionExpression(const Expr *E) const { 7418 const auto *OASE = dyn_cast<OMPArraySectionExpr>(E); 7419 7420 // It is not an array section and therefore not a unity-size one. 7421 if (!OASE) 7422 return false; 7423 7424 // An array section with no colon always refer to a single element. 7425 if (OASE->getColonLoc().isInvalid()) 7426 return false; 7427 7428 const Expr *Length = OASE->getLength(); 7429 7430 // If we don't have a length we have to check if the array has size 1 7431 // for this dimension. Also, we should always expect a length if the 7432 // base type is pointer. 7433 if (!Length) { 7434 QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType( 7435 OASE->getBase()->IgnoreParenImpCasts()) 7436 .getCanonicalType(); 7437 if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr())) 7438 return ATy->getSize().getSExtValue() != 1; 7439 // If we don't have a constant dimension length, we have to consider 7440 // the current section as having any size, so it is not necessarily 7441 // unitary. If it happen to be unity size, that's user fault. 7442 return true; 7443 } 7444 7445 // Check if the length evaluates to 1. 7446 Expr::EvalResult Result; 7447 if (!Length->EvaluateAsInt(Result, CGF.getContext())) 7448 return true; // Can have more that size 1. 7449 7450 llvm::APSInt ConstLength = Result.Val.getInt(); 7451 return ConstLength.getSExtValue() != 1; 7452 } 7453 7454 /// Generate the base pointers, section pointers, sizes and map type 7455 /// bits for the provided map type, map modifier, and expression components. 7456 /// \a IsFirstComponent should be set to true if the provided set of 7457 /// components is the first associated with a capture. 7458 void generateInfoForComponentList( 7459 OpenMPMapClauseKind MapType, 7460 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7461 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 7462 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 7463 MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types, 7464 StructRangeInfoTy &PartialStruct, bool IsFirstComponentList, 7465 bool IsImplicit, 7466 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 7467 OverlappedElements = llvm::None) const { 7468 // The following summarizes what has to be generated for each map and the 7469 // types below. The generated information is expressed in this order: 7470 // base pointer, section pointer, size, flags 7471 // (to add to the ones that come from the map type and modifier). 7472 // 7473 // double d; 7474 // int i[100]; 7475 // float *p; 7476 // 7477 // struct S1 { 7478 // int i; 7479 // float f[50]; 7480 // } 7481 // struct S2 { 7482 // int i; 7483 // float f[50]; 7484 // S1 s; 7485 // double *p; 7486 // struct S2 *ps; 7487 // } 7488 // S2 s; 7489 // S2 *ps; 7490 // 7491 // map(d) 7492 // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM 7493 // 7494 // map(i) 7495 // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM 7496 // 7497 // map(i[1:23]) 7498 // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM 7499 // 7500 // map(p) 7501 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM 7502 // 7503 // map(p[1:24]) 7504 // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM 7505 // 7506 // map(s) 7507 // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM 7508 // 7509 // map(s.i) 7510 // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM 7511 // 7512 // map(s.s.f) 7513 // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7514 // 7515 // map(s.p) 7516 // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM 7517 // 7518 // map(to: s.p[:22]) 7519 // &s, &(s.p), sizeof(double*), TARGET_PARAM (*) 7520 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**) 7521 // &(s.p), &(s.p[0]), 22*sizeof(double), 7522 // MEMBER_OF(1) | PTR_AND_OBJ | TO (***) 7523 // (*) alloc space for struct members, only this is a target parameter 7524 // (**) map the pointer (nothing to be mapped in this example) (the compiler 7525 // optimizes this entry out, same in the examples below) 7526 // (***) map the pointee (map: to) 7527 // 7528 // map(s.ps) 7529 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7530 // 7531 // map(from: s.ps->s.i) 7532 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7533 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7534 // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7535 // 7536 // map(to: s.ps->ps) 7537 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7538 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7539 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | TO 7540 // 7541 // map(s.ps->ps->ps) 7542 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7543 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7544 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7545 // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7546 // 7547 // map(to: s.ps->ps->s.f[:22]) 7548 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 7549 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 7550 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7551 // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7552 // 7553 // map(ps) 7554 // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM 7555 // 7556 // map(ps->i) 7557 // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM 7558 // 7559 // map(ps->s.f) 7560 // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 7561 // 7562 // map(from: ps->p) 7563 // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM 7564 // 7565 // map(to: ps->p[:22]) 7566 // ps, &(ps->p), sizeof(double*), TARGET_PARAM 7567 // ps, &(ps->p), sizeof(double*), MEMBER_OF(1) 7568 // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO 7569 // 7570 // map(ps->ps) 7571 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM 7572 // 7573 // map(from: ps->ps->s.i) 7574 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7575 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7576 // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7577 // 7578 // map(from: ps->ps->ps) 7579 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7580 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7581 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7582 // 7583 // map(ps->ps->ps->ps) 7584 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7585 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7586 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7587 // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 7588 // 7589 // map(to: ps->ps->ps->s.f[:22]) 7590 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 7591 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 7592 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 7593 // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 7594 // 7595 // map(to: s.f[:22]) map(from: s.p[:33]) 7596 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) + 7597 // sizeof(double*) (**), TARGET_PARAM 7598 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO 7599 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) 7600 // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7601 // (*) allocate contiguous space needed to fit all mapped members even if 7602 // we allocate space for members not mapped (in this example, 7603 // s.f[22..49] and s.s are not mapped, yet we must allocate space for 7604 // them as well because they fall between &s.f[0] and &s.p) 7605 // 7606 // map(from: s.f[:22]) map(to: ps->p[:33]) 7607 // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM 7608 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7609 // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*) 7610 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO 7611 // (*) the struct this entry pertains to is the 2nd element in the list of 7612 // arguments, hence MEMBER_OF(2) 7613 // 7614 // map(from: s.f[:22], s.s) map(to: ps->p[:33]) 7615 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM 7616 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM 7617 // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM 7618 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7619 // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*) 7620 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO 7621 // (*) the struct this entry pertains to is the 4th element in the list 7622 // of arguments, hence MEMBER_OF(4) 7623 7624 // Track if the map information being generated is the first for a capture. 7625 bool IsCaptureFirstInfo = IsFirstComponentList; 7626 // When the variable is on a declare target link or in a to clause with 7627 // unified memory, a reference is needed to hold the host/device address 7628 // of the variable. 7629 bool RequiresReference = false; 7630 7631 // Scan the components from the base to the complete expression. 7632 auto CI = Components.rbegin(); 7633 auto CE = Components.rend(); 7634 auto I = CI; 7635 7636 // Track if the map information being generated is the first for a list of 7637 // components. 7638 bool IsExpressionFirstInfo = true; 7639 Address BP = Address::invalid(); 7640 const Expr *AssocExpr = I->getAssociatedExpression(); 7641 const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr); 7642 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 7643 7644 if (isa<MemberExpr>(AssocExpr)) { 7645 // The base is the 'this' pointer. The content of the pointer is going 7646 // to be the base of the field being mapped. 7647 BP = CGF.LoadCXXThisAddress(); 7648 } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) || 7649 (OASE && 7650 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) { 7651 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(); 7652 } else { 7653 // The base is the reference to the variable. 7654 // BP = &Var. 7655 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(); 7656 if (const auto *VD = 7657 dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) { 7658 if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 7659 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) { 7660 if ((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 7661 (*Res == OMPDeclareTargetDeclAttr::MT_To && 7662 CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory())) { 7663 RequiresReference = true; 7664 BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 7665 } 7666 } 7667 } 7668 7669 // If the variable is a pointer and is being dereferenced (i.e. is not 7670 // the last component), the base has to be the pointer itself, not its 7671 // reference. References are ignored for mapping purposes. 7672 QualType Ty = 7673 I->getAssociatedDeclaration()->getType().getNonReferenceType(); 7674 if (Ty->isAnyPointerType() && std::next(I) != CE) { 7675 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>()); 7676 7677 // We do not need to generate individual map information for the 7678 // pointer, it can be associated with the combined storage. 7679 ++I; 7680 } 7681 } 7682 7683 // Track whether a component of the list should be marked as MEMBER_OF some 7684 // combined entry (for partial structs). Only the first PTR_AND_OBJ entry 7685 // in a component list should be marked as MEMBER_OF, all subsequent entries 7686 // do not belong to the base struct. E.g. 7687 // struct S2 s; 7688 // s.ps->ps->ps->f[:] 7689 // (1) (2) (3) (4) 7690 // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a 7691 // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3) 7692 // is the pointee of ps(2) which is not member of struct s, so it should not 7693 // be marked as such (it is still PTR_AND_OBJ). 7694 // The variable is initialized to false so that PTR_AND_OBJ entries which 7695 // are not struct members are not considered (e.g. array of pointers to 7696 // data). 7697 bool ShouldBeMemberOf = false; 7698 7699 // Variable keeping track of whether or not we have encountered a component 7700 // in the component list which is a member expression. Useful when we have a 7701 // pointer or a final array section, in which case it is the previous 7702 // component in the list which tells us whether we have a member expression. 7703 // E.g. X.f[:] 7704 // While processing the final array section "[:]" it is "f" which tells us 7705 // whether we are dealing with a member of a declared struct. 7706 const MemberExpr *EncounteredME = nullptr; 7707 7708 for (; I != CE; ++I) { 7709 // If the current component is member of a struct (parent struct) mark it. 7710 if (!EncounteredME) { 7711 EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression()); 7712 // If we encounter a PTR_AND_OBJ entry from now on it should be marked 7713 // as MEMBER_OF the parent struct. 7714 if (EncounteredME) 7715 ShouldBeMemberOf = true; 7716 } 7717 7718 auto Next = std::next(I); 7719 7720 // We need to generate the addresses and sizes if this is the last 7721 // component, if the component is a pointer or if it is an array section 7722 // whose length can't be proved to be one. If this is a pointer, it 7723 // becomes the base address for the following components. 7724 7725 // A final array section, is one whose length can't be proved to be one. 7726 bool IsFinalArraySection = 7727 isFinalArraySectionExpression(I->getAssociatedExpression()); 7728 7729 // Get information on whether the element is a pointer. Have to do a 7730 // special treatment for array sections given that they are built-in 7731 // types. 7732 const auto *OASE = 7733 dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression()); 7734 bool IsPointer = 7735 (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE) 7736 .getCanonicalType() 7737 ->isAnyPointerType()) || 7738 I->getAssociatedExpression()->getType()->isAnyPointerType(); 7739 7740 if (Next == CE || IsPointer || IsFinalArraySection) { 7741 // If this is not the last component, we expect the pointer to be 7742 // associated with an array expression or member expression. 7743 assert((Next == CE || 7744 isa<MemberExpr>(Next->getAssociatedExpression()) || 7745 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) || 7746 isa<OMPArraySectionExpr>(Next->getAssociatedExpression())) && 7747 "Unexpected expression"); 7748 7749 Address LB = 7750 CGF.EmitOMPSharedLValue(I->getAssociatedExpression()).getAddress(); 7751 7752 // If this component is a pointer inside the base struct then we don't 7753 // need to create any entry for it - it will be combined with the object 7754 // it is pointing to into a single PTR_AND_OBJ entry. 7755 bool IsMemberPointer = 7756 IsPointer && EncounteredME && 7757 (dyn_cast<MemberExpr>(I->getAssociatedExpression()) == 7758 EncounteredME); 7759 if (!OverlappedElements.empty()) { 7760 // Handle base element with the info for overlapped elements. 7761 assert(!PartialStruct.Base.isValid() && "The base element is set."); 7762 assert(Next == CE && 7763 "Expected last element for the overlapped elements."); 7764 assert(!IsPointer && 7765 "Unexpected base element with the pointer type."); 7766 // Mark the whole struct as the struct that requires allocation on the 7767 // device. 7768 PartialStruct.LowestElem = {0, LB}; 7769 CharUnits TypeSize = CGF.getContext().getTypeSizeInChars( 7770 I->getAssociatedExpression()->getType()); 7771 Address HB = CGF.Builder.CreateConstGEP( 7772 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(LB, 7773 CGF.VoidPtrTy), 7774 TypeSize.getQuantity() - 1); 7775 PartialStruct.HighestElem = { 7776 std::numeric_limits<decltype( 7777 PartialStruct.HighestElem.first)>::max(), 7778 HB}; 7779 PartialStruct.Base = BP; 7780 // Emit data for non-overlapped data. 7781 OpenMPOffloadMappingFlags Flags = 7782 OMP_MAP_MEMBER_OF | 7783 getMapTypeBits(MapType, MapModifiers, IsImplicit, 7784 /*AddPtrFlag=*/false, 7785 /*AddIsTargetParamFlag=*/false); 7786 LB = BP; 7787 llvm::Value *Size = nullptr; 7788 // Do bitcopy of all non-overlapped structure elements. 7789 for (OMPClauseMappableExprCommon::MappableExprComponentListRef 7790 Component : OverlappedElements) { 7791 Address ComponentLB = Address::invalid(); 7792 for (const OMPClauseMappableExprCommon::MappableComponent &MC : 7793 Component) { 7794 if (MC.getAssociatedDeclaration()) { 7795 ComponentLB = 7796 CGF.EmitOMPSharedLValue(MC.getAssociatedExpression()) 7797 .getAddress(); 7798 Size = CGF.Builder.CreatePtrDiff( 7799 CGF.EmitCastToVoidPtr(ComponentLB.getPointer()), 7800 CGF.EmitCastToVoidPtr(LB.getPointer())); 7801 break; 7802 } 7803 } 7804 BasePointers.push_back(BP.getPointer()); 7805 Pointers.push_back(LB.getPointer()); 7806 Sizes.push_back(CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, 7807 /*isSigned=*/true)); 7808 Types.push_back(Flags); 7809 LB = CGF.Builder.CreateConstGEP(ComponentLB, 1); 7810 } 7811 BasePointers.push_back(BP.getPointer()); 7812 Pointers.push_back(LB.getPointer()); 7813 Size = CGF.Builder.CreatePtrDiff( 7814 CGF.EmitCastToVoidPtr( 7815 CGF.Builder.CreateConstGEP(HB, 1).getPointer()), 7816 CGF.EmitCastToVoidPtr(LB.getPointer())); 7817 Sizes.push_back( 7818 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 7819 Types.push_back(Flags); 7820 break; 7821 } 7822 llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression()); 7823 if (!IsMemberPointer) { 7824 BasePointers.push_back(BP.getPointer()); 7825 Pointers.push_back(LB.getPointer()); 7826 Sizes.push_back( 7827 CGF.Builder.CreateIntCast(Size, CGF.Int64Ty, /*isSigned=*/true)); 7828 7829 // We need to add a pointer flag for each map that comes from the 7830 // same expression except for the first one. We also need to signal 7831 // this map is the first one that relates with the current capture 7832 // (there is a set of entries for each capture). 7833 OpenMPOffloadMappingFlags Flags = getMapTypeBits( 7834 MapType, MapModifiers, IsImplicit, 7835 !IsExpressionFirstInfo || RequiresReference, 7836 IsCaptureFirstInfo && !RequiresReference); 7837 7838 if (!IsExpressionFirstInfo) { 7839 // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well, 7840 // then we reset the TO/FROM/ALWAYS/DELETE/CLOSE flags. 7841 if (IsPointer) 7842 Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS | 7843 OMP_MAP_DELETE | OMP_MAP_CLOSE); 7844 7845 if (ShouldBeMemberOf) { 7846 // Set placeholder value MEMBER_OF=FFFF to indicate that the flag 7847 // should be later updated with the correct value of MEMBER_OF. 7848 Flags |= OMP_MAP_MEMBER_OF; 7849 // From now on, all subsequent PTR_AND_OBJ entries should not be 7850 // marked as MEMBER_OF. 7851 ShouldBeMemberOf = false; 7852 } 7853 } 7854 7855 Types.push_back(Flags); 7856 } 7857 7858 // If we have encountered a member expression so far, keep track of the 7859 // mapped member. If the parent is "*this", then the value declaration 7860 // is nullptr. 7861 if (EncounteredME) { 7862 const auto *FD = dyn_cast<FieldDecl>(EncounteredME->getMemberDecl()); 7863 unsigned FieldIndex = FD->getFieldIndex(); 7864 7865 // Update info about the lowest and highest elements for this struct 7866 if (!PartialStruct.Base.isValid()) { 7867 PartialStruct.LowestElem = {FieldIndex, LB}; 7868 PartialStruct.HighestElem = {FieldIndex, LB}; 7869 PartialStruct.Base = BP; 7870 } else if (FieldIndex < PartialStruct.LowestElem.first) { 7871 PartialStruct.LowestElem = {FieldIndex, LB}; 7872 } else if (FieldIndex > PartialStruct.HighestElem.first) { 7873 PartialStruct.HighestElem = {FieldIndex, LB}; 7874 } 7875 } 7876 7877 // If we have a final array section, we are done with this expression. 7878 if (IsFinalArraySection) 7879 break; 7880 7881 // The pointer becomes the base for the next element. 7882 if (Next != CE) 7883 BP = LB; 7884 7885 IsExpressionFirstInfo = false; 7886 IsCaptureFirstInfo = false; 7887 } 7888 } 7889 } 7890 7891 /// Return the adjusted map modifiers if the declaration a capture refers to 7892 /// appears in a first-private clause. This is expected to be used only with 7893 /// directives that start with 'target'. 7894 MappableExprsHandler::OpenMPOffloadMappingFlags 7895 getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const { 7896 assert(Cap.capturesVariable() && "Expected capture by reference only!"); 7897 7898 // A first private variable captured by reference will use only the 7899 // 'private ptr' and 'map to' flag. Return the right flags if the captured 7900 // declaration is known as first-private in this handler. 7901 if (FirstPrivateDecls.count(Cap.getCapturedVar())) { 7902 if (Cap.getCapturedVar()->getType().isConstant(CGF.getContext()) && 7903 Cap.getCaptureKind() == CapturedStmt::VCK_ByRef) 7904 return MappableExprsHandler::OMP_MAP_ALWAYS | 7905 MappableExprsHandler::OMP_MAP_TO; 7906 if (Cap.getCapturedVar()->getType()->isAnyPointerType()) 7907 return MappableExprsHandler::OMP_MAP_TO | 7908 MappableExprsHandler::OMP_MAP_PTR_AND_OBJ; 7909 return MappableExprsHandler::OMP_MAP_PRIVATE | 7910 MappableExprsHandler::OMP_MAP_TO; 7911 } 7912 return MappableExprsHandler::OMP_MAP_TO | 7913 MappableExprsHandler::OMP_MAP_FROM; 7914 } 7915 7916 static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) { 7917 // Rotate by getFlagMemberOffset() bits. 7918 return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1) 7919 << getFlagMemberOffset()); 7920 } 7921 7922 static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags, 7923 OpenMPOffloadMappingFlags MemberOfFlag) { 7924 // If the entry is PTR_AND_OBJ but has not been marked with the special 7925 // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be 7926 // marked as MEMBER_OF. 7927 if ((Flags & OMP_MAP_PTR_AND_OBJ) && 7928 ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF)) 7929 return; 7930 7931 // Reset the placeholder value to prepare the flag for the assignment of the 7932 // proper MEMBER_OF value. 7933 Flags &= ~OMP_MAP_MEMBER_OF; 7934 Flags |= MemberOfFlag; 7935 } 7936 7937 void getPlainLayout(const CXXRecordDecl *RD, 7938 llvm::SmallVectorImpl<const FieldDecl *> &Layout, 7939 bool AsBase) const { 7940 const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD); 7941 7942 llvm::StructType *St = 7943 AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType(); 7944 7945 unsigned NumElements = St->getNumElements(); 7946 llvm::SmallVector< 7947 llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4> 7948 RecordLayout(NumElements); 7949 7950 // Fill bases. 7951 for (const auto &I : RD->bases()) { 7952 if (I.isVirtual()) 7953 continue; 7954 const auto *Base = I.getType()->getAsCXXRecordDecl(); 7955 // Ignore empty bases. 7956 if (Base->isEmpty() || CGF.getContext() 7957 .getASTRecordLayout(Base) 7958 .getNonVirtualSize() 7959 .isZero()) 7960 continue; 7961 7962 unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base); 7963 RecordLayout[FieldIndex] = Base; 7964 } 7965 // Fill in virtual bases. 7966 for (const auto &I : RD->vbases()) { 7967 const auto *Base = I.getType()->getAsCXXRecordDecl(); 7968 // Ignore empty bases. 7969 if (Base->isEmpty()) 7970 continue; 7971 unsigned FieldIndex = RL.getVirtualBaseIndex(Base); 7972 if (RecordLayout[FieldIndex]) 7973 continue; 7974 RecordLayout[FieldIndex] = Base; 7975 } 7976 // Fill in all the fields. 7977 assert(!RD->isUnion() && "Unexpected union."); 7978 for (const auto *Field : RD->fields()) { 7979 // Fill in non-bitfields. (Bitfields always use a zero pattern, which we 7980 // will fill in later.) 7981 if (!Field->isBitField() && !Field->isZeroSize(CGF.getContext())) { 7982 unsigned FieldIndex = RL.getLLVMFieldNo(Field); 7983 RecordLayout[FieldIndex] = Field; 7984 } 7985 } 7986 for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *> 7987 &Data : RecordLayout) { 7988 if (Data.isNull()) 7989 continue; 7990 if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>()) 7991 getPlainLayout(Base, Layout, /*AsBase=*/true); 7992 else 7993 Layout.push_back(Data.get<const FieldDecl *>()); 7994 } 7995 } 7996 7997 public: 7998 MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF) 7999 : CurDir(&Dir), CGF(CGF) { 8000 // Extract firstprivate clause information. 8001 for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>()) 8002 for (const auto *D : C->varlists()) 8003 FirstPrivateDecls.try_emplace( 8004 cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl()), C->isImplicit()); 8005 // Extract device pointer clause information. 8006 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>()) 8007 for (auto L : C->component_lists()) 8008 DevPointersMap[L.first].push_back(L.second); 8009 } 8010 8011 /// Constructor for the declare mapper directive. 8012 MappableExprsHandler(const OMPDeclareMapperDecl &Dir, CodeGenFunction &CGF) 8013 : CurDir(&Dir), CGF(CGF) {} 8014 8015 /// Generate code for the combined entry if we have a partially mapped struct 8016 /// and take care of the mapping flags of the arguments corresponding to 8017 /// individual struct members. 8018 void emitCombinedEntry(MapBaseValuesArrayTy &BasePointers, 8019 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 8020 MapFlagsArrayTy &Types, MapFlagsArrayTy &CurTypes, 8021 const StructRangeInfoTy &PartialStruct) const { 8022 // Base is the base of the struct 8023 BasePointers.push_back(PartialStruct.Base.getPointer()); 8024 // Pointer is the address of the lowest element 8025 llvm::Value *LB = PartialStruct.LowestElem.second.getPointer(); 8026 Pointers.push_back(LB); 8027 // Size is (addr of {highest+1} element) - (addr of lowest element) 8028 llvm::Value *HB = PartialStruct.HighestElem.second.getPointer(); 8029 llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(HB, /*Idx0=*/1); 8030 llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy); 8031 llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy); 8032 llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr); 8033 llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.Int64Ty, 8034 /*isSigned=*/false); 8035 Sizes.push_back(Size); 8036 // Map type is always TARGET_PARAM 8037 Types.push_back(OMP_MAP_TARGET_PARAM); 8038 // Remove TARGET_PARAM flag from the first element 8039 (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM; 8040 8041 // All other current entries will be MEMBER_OF the combined entry 8042 // (except for PTR_AND_OBJ entries which do not have a placeholder value 8043 // 0xFFFF in the MEMBER_OF field). 8044 OpenMPOffloadMappingFlags MemberOfFlag = 8045 getMemberOfFlag(BasePointers.size() - 1); 8046 for (auto &M : CurTypes) 8047 setCorrectMemberOfFlag(M, MemberOfFlag); 8048 } 8049 8050 /// Generate all the base pointers, section pointers, sizes and map 8051 /// types for the extracted mappable expressions. Also, for each item that 8052 /// relates with a device pointer, a pair of the relevant declaration and 8053 /// index where it occurs is appended to the device pointers info array. 8054 void generateAllInfo(MapBaseValuesArrayTy &BasePointers, 8055 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 8056 MapFlagsArrayTy &Types) const { 8057 // We have to process the component lists that relate with the same 8058 // declaration in a single chunk so that we can generate the map flags 8059 // correctly. Therefore, we organize all lists in a map. 8060 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 8061 8062 // Helper function to fill the information map for the different supported 8063 // clauses. 8064 auto &&InfoGen = [&Info]( 8065 const ValueDecl *D, 8066 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 8067 OpenMPMapClauseKind MapType, 8068 ArrayRef<OpenMPMapModifierKind> MapModifiers, 8069 bool ReturnDevicePointer, bool IsImplicit) { 8070 const ValueDecl *VD = 8071 D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr; 8072 Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer, 8073 IsImplicit); 8074 }; 8075 8076 assert(CurDir.is<const OMPExecutableDirective *>() && 8077 "Expect a executable directive"); 8078 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8079 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) 8080 for (const auto &L : C->component_lists()) { 8081 InfoGen(L.first, L.second, C->getMapType(), C->getMapTypeModifiers(), 8082 /*ReturnDevicePointer=*/false, C->isImplicit()); 8083 } 8084 for (const auto *C : CurExecDir->getClausesOfKind<OMPToClause>()) 8085 for (const auto &L : C->component_lists()) { 8086 InfoGen(L.first, L.second, OMPC_MAP_to, llvm::None, 8087 /*ReturnDevicePointer=*/false, C->isImplicit()); 8088 } 8089 for (const auto *C : CurExecDir->getClausesOfKind<OMPFromClause>()) 8090 for (const auto &L : C->component_lists()) { 8091 InfoGen(L.first, L.second, OMPC_MAP_from, llvm::None, 8092 /*ReturnDevicePointer=*/false, C->isImplicit()); 8093 } 8094 8095 // Look at the use_device_ptr clause information and mark the existing map 8096 // entries as such. If there is no map information for an entry in the 8097 // use_device_ptr list, we create one with map type 'alloc' and zero size 8098 // section. It is the user fault if that was not mapped before. If there is 8099 // no map information and the pointer is a struct member, then we defer the 8100 // emission of that entry until the whole struct has been processed. 8101 llvm::MapVector<const ValueDecl *, SmallVector<DeferredDevicePtrEntryTy, 4>> 8102 DeferredInfo; 8103 8104 for (const auto *C : 8105 CurExecDir->getClausesOfKind<OMPUseDevicePtrClause>()) { 8106 for (const auto &L : C->component_lists()) { 8107 assert(!L.second.empty() && "Not expecting empty list of components!"); 8108 const ValueDecl *VD = L.second.back().getAssociatedDeclaration(); 8109 VD = cast<ValueDecl>(VD->getCanonicalDecl()); 8110 const Expr *IE = L.second.back().getAssociatedExpression(); 8111 // If the first component is a member expression, we have to look into 8112 // 'this', which maps to null in the map of map information. Otherwise 8113 // look directly for the information. 8114 auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD); 8115 8116 // We potentially have map information for this declaration already. 8117 // Look for the first set of components that refer to it. 8118 if (It != Info.end()) { 8119 auto CI = std::find_if( 8120 It->second.begin(), It->second.end(), [VD](const MapInfo &MI) { 8121 return MI.Components.back().getAssociatedDeclaration() == VD; 8122 }); 8123 // If we found a map entry, signal that the pointer has to be returned 8124 // and move on to the next declaration. 8125 if (CI != It->second.end()) { 8126 CI->ReturnDevicePointer = true; 8127 continue; 8128 } 8129 } 8130 8131 // We didn't find any match in our map information - generate a zero 8132 // size array section - if the pointer is a struct member we defer this 8133 // action until the whole struct has been processed. 8134 if (isa<MemberExpr>(IE)) { 8135 // Insert the pointer into Info to be processed by 8136 // generateInfoForComponentList. Because it is a member pointer 8137 // without a pointee, no entry will be generated for it, therefore 8138 // we need to generate one after the whole struct has been processed. 8139 // Nonetheless, generateInfoForComponentList must be called to take 8140 // the pointer into account for the calculation of the range of the 8141 // partial struct. 8142 InfoGen(nullptr, L.second, OMPC_MAP_unknown, llvm::None, 8143 /*ReturnDevicePointer=*/false, C->isImplicit()); 8144 DeferredInfo[nullptr].emplace_back(IE, VD); 8145 } else { 8146 llvm::Value *Ptr = 8147 CGF.EmitLoadOfScalar(CGF.EmitLValue(IE), IE->getExprLoc()); 8148 BasePointers.emplace_back(Ptr, VD); 8149 Pointers.push_back(Ptr); 8150 Sizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty)); 8151 Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_TARGET_PARAM); 8152 } 8153 } 8154 } 8155 8156 for (const auto &M : Info) { 8157 // We need to know when we generate information for the first component 8158 // associated with a capture, because the mapping flags depend on it. 8159 bool IsFirstComponentList = true; 8160 8161 // Temporary versions of arrays 8162 MapBaseValuesArrayTy CurBasePointers; 8163 MapValuesArrayTy CurPointers; 8164 MapValuesArrayTy CurSizes; 8165 MapFlagsArrayTy CurTypes; 8166 StructRangeInfoTy PartialStruct; 8167 8168 for (const MapInfo &L : M.second) { 8169 assert(!L.Components.empty() && 8170 "Not expecting declaration with no component lists."); 8171 8172 // Remember the current base pointer index. 8173 unsigned CurrentBasePointersIdx = CurBasePointers.size(); 8174 generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components, 8175 CurBasePointers, CurPointers, CurSizes, 8176 CurTypes, PartialStruct, 8177 IsFirstComponentList, L.IsImplicit); 8178 8179 // If this entry relates with a device pointer, set the relevant 8180 // declaration and add the 'return pointer' flag. 8181 if (L.ReturnDevicePointer) { 8182 assert(CurBasePointers.size() > CurrentBasePointersIdx && 8183 "Unexpected number of mapped base pointers."); 8184 8185 const ValueDecl *RelevantVD = 8186 L.Components.back().getAssociatedDeclaration(); 8187 assert(RelevantVD && 8188 "No relevant declaration related with device pointer??"); 8189 8190 CurBasePointers[CurrentBasePointersIdx].setDevicePtrDecl(RelevantVD); 8191 CurTypes[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM; 8192 } 8193 IsFirstComponentList = false; 8194 } 8195 8196 // Append any pending zero-length pointers which are struct members and 8197 // used with use_device_ptr. 8198 auto CI = DeferredInfo.find(M.first); 8199 if (CI != DeferredInfo.end()) { 8200 for (const DeferredDevicePtrEntryTy &L : CI->second) { 8201 llvm::Value *BasePtr = this->CGF.EmitLValue(L.IE).getPointer(); 8202 llvm::Value *Ptr = this->CGF.EmitLoadOfScalar( 8203 this->CGF.EmitLValue(L.IE), L.IE->getExprLoc()); 8204 CurBasePointers.emplace_back(BasePtr, L.VD); 8205 CurPointers.push_back(Ptr); 8206 CurSizes.push_back(llvm::Constant::getNullValue(this->CGF.Int64Ty)); 8207 // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the placeholder 8208 // value MEMBER_OF=FFFF so that the entry is later updated with the 8209 // correct value of MEMBER_OF. 8210 CurTypes.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM | 8211 OMP_MAP_MEMBER_OF); 8212 } 8213 } 8214 8215 // If there is an entry in PartialStruct it means we have a struct with 8216 // individual members mapped. Emit an extra combined entry. 8217 if (PartialStruct.Base.isValid()) 8218 emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes, 8219 PartialStruct); 8220 8221 // We need to append the results of this capture to what we already have. 8222 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 8223 Pointers.append(CurPointers.begin(), CurPointers.end()); 8224 Sizes.append(CurSizes.begin(), CurSizes.end()); 8225 Types.append(CurTypes.begin(), CurTypes.end()); 8226 } 8227 } 8228 8229 /// Generate all the base pointers, section pointers, sizes and map types for 8230 /// the extracted map clauses of user-defined mapper. 8231 void generateAllInfoForMapper(MapBaseValuesArrayTy &BasePointers, 8232 MapValuesArrayTy &Pointers, 8233 MapValuesArrayTy &Sizes, 8234 MapFlagsArrayTy &Types) const { 8235 assert(CurDir.is<const OMPDeclareMapperDecl *>() && 8236 "Expect a declare mapper directive"); 8237 const auto *CurMapperDir = CurDir.get<const OMPDeclareMapperDecl *>(); 8238 // We have to process the component lists that relate with the same 8239 // declaration in a single chunk so that we can generate the map flags 8240 // correctly. Therefore, we organize all lists in a map. 8241 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 8242 8243 // Helper function to fill the information map for the different supported 8244 // clauses. 8245 auto &&InfoGen = [&Info]( 8246 const ValueDecl *D, 8247 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 8248 OpenMPMapClauseKind MapType, 8249 ArrayRef<OpenMPMapModifierKind> MapModifiers, 8250 bool ReturnDevicePointer, bool IsImplicit) { 8251 const ValueDecl *VD = 8252 D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr; 8253 Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer, 8254 IsImplicit); 8255 }; 8256 8257 for (const auto *C : CurMapperDir->clauselists()) { 8258 const auto *MC = cast<OMPMapClause>(C); 8259 for (const auto &L : MC->component_lists()) { 8260 InfoGen(L.first, L.second, MC->getMapType(), MC->getMapTypeModifiers(), 8261 /*ReturnDevicePointer=*/false, MC->isImplicit()); 8262 } 8263 } 8264 8265 for (const auto &M : Info) { 8266 // We need to know when we generate information for the first component 8267 // associated with a capture, because the mapping flags depend on it. 8268 bool IsFirstComponentList = true; 8269 8270 // Temporary versions of arrays 8271 MapBaseValuesArrayTy CurBasePointers; 8272 MapValuesArrayTy CurPointers; 8273 MapValuesArrayTy CurSizes; 8274 MapFlagsArrayTy CurTypes; 8275 StructRangeInfoTy PartialStruct; 8276 8277 for (const MapInfo &L : M.second) { 8278 assert(!L.Components.empty() && 8279 "Not expecting declaration with no component lists."); 8280 generateInfoForComponentList(L.MapType, L.MapModifiers, L.Components, 8281 CurBasePointers, CurPointers, CurSizes, 8282 CurTypes, PartialStruct, 8283 IsFirstComponentList, L.IsImplicit); 8284 IsFirstComponentList = false; 8285 } 8286 8287 // If there is an entry in PartialStruct it means we have a struct with 8288 // individual members mapped. Emit an extra combined entry. 8289 if (PartialStruct.Base.isValid()) 8290 emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes, 8291 PartialStruct); 8292 8293 // We need to append the results of this capture to what we already have. 8294 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 8295 Pointers.append(CurPointers.begin(), CurPointers.end()); 8296 Sizes.append(CurSizes.begin(), CurSizes.end()); 8297 Types.append(CurTypes.begin(), CurTypes.end()); 8298 } 8299 } 8300 8301 /// Emit capture info for lambdas for variables captured by reference. 8302 void generateInfoForLambdaCaptures( 8303 const ValueDecl *VD, llvm::Value *Arg, MapBaseValuesArrayTy &BasePointers, 8304 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 8305 MapFlagsArrayTy &Types, 8306 llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const { 8307 const auto *RD = VD->getType() 8308 .getCanonicalType() 8309 .getNonReferenceType() 8310 ->getAsCXXRecordDecl(); 8311 if (!RD || !RD->isLambda()) 8312 return; 8313 Address VDAddr = Address(Arg, CGF.getContext().getDeclAlign(VD)); 8314 LValue VDLVal = CGF.MakeAddrLValue( 8315 VDAddr, VD->getType().getCanonicalType().getNonReferenceType()); 8316 llvm::DenseMap<const VarDecl *, FieldDecl *> Captures; 8317 FieldDecl *ThisCapture = nullptr; 8318 RD->getCaptureFields(Captures, ThisCapture); 8319 if (ThisCapture) { 8320 LValue ThisLVal = 8321 CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture); 8322 LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture); 8323 LambdaPointers.try_emplace(ThisLVal.getPointer(), VDLVal.getPointer()); 8324 BasePointers.push_back(ThisLVal.getPointer()); 8325 Pointers.push_back(ThisLValVal.getPointer()); 8326 Sizes.push_back( 8327 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy), 8328 CGF.Int64Ty, /*isSigned=*/true)); 8329 Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8330 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 8331 } 8332 for (const LambdaCapture &LC : RD->captures()) { 8333 if (!LC.capturesVariable()) 8334 continue; 8335 const VarDecl *VD = LC.getCapturedVar(); 8336 if (LC.getCaptureKind() != LCK_ByRef && !VD->getType()->isPointerType()) 8337 continue; 8338 auto It = Captures.find(VD); 8339 assert(It != Captures.end() && "Found lambda capture without field."); 8340 LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second); 8341 if (LC.getCaptureKind() == LCK_ByRef) { 8342 LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second); 8343 LambdaPointers.try_emplace(VarLVal.getPointer(), VDLVal.getPointer()); 8344 BasePointers.push_back(VarLVal.getPointer()); 8345 Pointers.push_back(VarLValVal.getPointer()); 8346 Sizes.push_back(CGF.Builder.CreateIntCast( 8347 CGF.getTypeSize( 8348 VD->getType().getCanonicalType().getNonReferenceType()), 8349 CGF.Int64Ty, /*isSigned=*/true)); 8350 } else { 8351 RValue VarRVal = CGF.EmitLoadOfLValue(VarLVal, RD->getLocation()); 8352 LambdaPointers.try_emplace(VarLVal.getPointer(), VDLVal.getPointer()); 8353 BasePointers.push_back(VarLVal.getPointer()); 8354 Pointers.push_back(VarRVal.getScalarVal()); 8355 Sizes.push_back(llvm::ConstantInt::get(CGF.Int64Ty, 0)); 8356 } 8357 Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8358 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 8359 } 8360 } 8361 8362 /// Set correct indices for lambdas captures. 8363 void adjustMemberOfForLambdaCaptures( 8364 const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers, 8365 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 8366 MapFlagsArrayTy &Types) const { 8367 for (unsigned I = 0, E = Types.size(); I < E; ++I) { 8368 // Set correct member_of idx for all implicit lambda captures. 8369 if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 8370 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT)) 8371 continue; 8372 llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]); 8373 assert(BasePtr && "Unable to find base lambda address."); 8374 int TgtIdx = -1; 8375 for (unsigned J = I; J > 0; --J) { 8376 unsigned Idx = J - 1; 8377 if (Pointers[Idx] != BasePtr) 8378 continue; 8379 TgtIdx = Idx; 8380 break; 8381 } 8382 assert(TgtIdx != -1 && "Unable to find parent lambda."); 8383 // All other current entries will be MEMBER_OF the combined entry 8384 // (except for PTR_AND_OBJ entries which do not have a placeholder value 8385 // 0xFFFF in the MEMBER_OF field). 8386 OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx); 8387 setCorrectMemberOfFlag(Types[I], MemberOfFlag); 8388 } 8389 } 8390 8391 /// Generate the base pointers, section pointers, sizes and map types 8392 /// associated to a given capture. 8393 void generateInfoForCapture(const CapturedStmt::Capture *Cap, 8394 llvm::Value *Arg, 8395 MapBaseValuesArrayTy &BasePointers, 8396 MapValuesArrayTy &Pointers, 8397 MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types, 8398 StructRangeInfoTy &PartialStruct) const { 8399 assert(!Cap->capturesVariableArrayType() && 8400 "Not expecting to generate map info for a variable array type!"); 8401 8402 // We need to know when we generating information for the first component 8403 const ValueDecl *VD = Cap->capturesThis() 8404 ? nullptr 8405 : Cap->getCapturedVar()->getCanonicalDecl(); 8406 8407 // If this declaration appears in a is_device_ptr clause we just have to 8408 // pass the pointer by value. If it is a reference to a declaration, we just 8409 // pass its value. 8410 if (DevPointersMap.count(VD)) { 8411 BasePointers.emplace_back(Arg, VD); 8412 Pointers.push_back(Arg); 8413 Sizes.push_back( 8414 CGF.Builder.CreateIntCast(CGF.getTypeSize(CGF.getContext().VoidPtrTy), 8415 CGF.Int64Ty, /*isSigned=*/true)); 8416 Types.push_back(OMP_MAP_LITERAL | OMP_MAP_TARGET_PARAM); 8417 return; 8418 } 8419 8420 using MapData = 8421 std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef, 8422 OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool>; 8423 SmallVector<MapData, 4> DeclComponentLists; 8424 assert(CurDir.is<const OMPExecutableDirective *>() && 8425 "Expect a executable directive"); 8426 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8427 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) { 8428 for (const auto &L : C->decl_component_lists(VD)) { 8429 assert(L.first == VD && 8430 "We got information for the wrong declaration??"); 8431 assert(!L.second.empty() && 8432 "Not expecting declaration with no component lists."); 8433 DeclComponentLists.emplace_back(L.second, C->getMapType(), 8434 C->getMapTypeModifiers(), 8435 C->isImplicit()); 8436 } 8437 } 8438 8439 // Find overlapping elements (including the offset from the base element). 8440 llvm::SmallDenseMap< 8441 const MapData *, 8442 llvm::SmallVector< 8443 OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>, 8444 4> 8445 OverlappedData; 8446 size_t Count = 0; 8447 for (const MapData &L : DeclComponentLists) { 8448 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8449 OpenMPMapClauseKind MapType; 8450 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8451 bool IsImplicit; 8452 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8453 ++Count; 8454 for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) { 8455 OMPClauseMappableExprCommon::MappableExprComponentListRef Components1; 8456 std::tie(Components1, MapType, MapModifiers, IsImplicit) = L1; 8457 auto CI = Components.rbegin(); 8458 auto CE = Components.rend(); 8459 auto SI = Components1.rbegin(); 8460 auto SE = Components1.rend(); 8461 for (; CI != CE && SI != SE; ++CI, ++SI) { 8462 if (CI->getAssociatedExpression()->getStmtClass() != 8463 SI->getAssociatedExpression()->getStmtClass()) 8464 break; 8465 // Are we dealing with different variables/fields? 8466 if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration()) 8467 break; 8468 } 8469 // Found overlapping if, at least for one component, reached the head of 8470 // the components list. 8471 if (CI == CE || SI == SE) { 8472 assert((CI != CE || SI != SE) && 8473 "Unexpected full match of the mapping components."); 8474 const MapData &BaseData = CI == CE ? L : L1; 8475 OMPClauseMappableExprCommon::MappableExprComponentListRef SubData = 8476 SI == SE ? Components : Components1; 8477 auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData); 8478 OverlappedElements.getSecond().push_back(SubData); 8479 } 8480 } 8481 } 8482 // Sort the overlapped elements for each item. 8483 llvm::SmallVector<const FieldDecl *, 4> Layout; 8484 if (!OverlappedData.empty()) { 8485 if (const auto *CRD = 8486 VD->getType().getCanonicalType()->getAsCXXRecordDecl()) 8487 getPlainLayout(CRD, Layout, /*AsBase=*/false); 8488 else { 8489 const auto *RD = VD->getType().getCanonicalType()->getAsRecordDecl(); 8490 Layout.append(RD->field_begin(), RD->field_end()); 8491 } 8492 } 8493 for (auto &Pair : OverlappedData) { 8494 llvm::sort( 8495 Pair.getSecond(), 8496 [&Layout]( 8497 OMPClauseMappableExprCommon::MappableExprComponentListRef First, 8498 OMPClauseMappableExprCommon::MappableExprComponentListRef 8499 Second) { 8500 auto CI = First.rbegin(); 8501 auto CE = First.rend(); 8502 auto SI = Second.rbegin(); 8503 auto SE = Second.rend(); 8504 for (; CI != CE && SI != SE; ++CI, ++SI) { 8505 if (CI->getAssociatedExpression()->getStmtClass() != 8506 SI->getAssociatedExpression()->getStmtClass()) 8507 break; 8508 // Are we dealing with different variables/fields? 8509 if (CI->getAssociatedDeclaration() != 8510 SI->getAssociatedDeclaration()) 8511 break; 8512 } 8513 8514 // Lists contain the same elements. 8515 if (CI == CE && SI == SE) 8516 return false; 8517 8518 // List with less elements is less than list with more elements. 8519 if (CI == CE || SI == SE) 8520 return CI == CE; 8521 8522 const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration()); 8523 const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration()); 8524 if (FD1->getParent() == FD2->getParent()) 8525 return FD1->getFieldIndex() < FD2->getFieldIndex(); 8526 const auto It = 8527 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) { 8528 return FD == FD1 || FD == FD2; 8529 }); 8530 return *It == FD1; 8531 }); 8532 } 8533 8534 // Associated with a capture, because the mapping flags depend on it. 8535 // Go through all of the elements with the overlapped elements. 8536 for (const auto &Pair : OverlappedData) { 8537 const MapData &L = *Pair.getFirst(); 8538 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8539 OpenMPMapClauseKind MapType; 8540 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8541 bool IsImplicit; 8542 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8543 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 8544 OverlappedComponents = Pair.getSecond(); 8545 bool IsFirstComponentList = true; 8546 generateInfoForComponentList(MapType, MapModifiers, Components, 8547 BasePointers, Pointers, Sizes, Types, 8548 PartialStruct, IsFirstComponentList, 8549 IsImplicit, OverlappedComponents); 8550 } 8551 // Go through other elements without overlapped elements. 8552 bool IsFirstComponentList = OverlappedData.empty(); 8553 for (const MapData &L : DeclComponentLists) { 8554 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 8555 OpenMPMapClauseKind MapType; 8556 ArrayRef<OpenMPMapModifierKind> MapModifiers; 8557 bool IsImplicit; 8558 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 8559 auto It = OverlappedData.find(&L); 8560 if (It == OverlappedData.end()) 8561 generateInfoForComponentList(MapType, MapModifiers, Components, 8562 BasePointers, Pointers, Sizes, Types, 8563 PartialStruct, IsFirstComponentList, 8564 IsImplicit); 8565 IsFirstComponentList = false; 8566 } 8567 } 8568 8569 /// Generate the base pointers, section pointers, sizes and map types 8570 /// associated with the declare target link variables. 8571 void generateInfoForDeclareTargetLink(MapBaseValuesArrayTy &BasePointers, 8572 MapValuesArrayTy &Pointers, 8573 MapValuesArrayTy &Sizes, 8574 MapFlagsArrayTy &Types) const { 8575 assert(CurDir.is<const OMPExecutableDirective *>() && 8576 "Expect a executable directive"); 8577 const auto *CurExecDir = CurDir.get<const OMPExecutableDirective *>(); 8578 // Map other list items in the map clause which are not captured variables 8579 // but "declare target link" global variables. 8580 for (const auto *C : CurExecDir->getClausesOfKind<OMPMapClause>()) { 8581 for (const auto &L : C->component_lists()) { 8582 if (!L.first) 8583 continue; 8584 const auto *VD = dyn_cast<VarDecl>(L.first); 8585 if (!VD) 8586 continue; 8587 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 8588 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 8589 if (CGF.CGM.getOpenMPRuntime().hasRequiresUnifiedSharedMemory() || 8590 !Res || *Res != OMPDeclareTargetDeclAttr::MT_Link) 8591 continue; 8592 StructRangeInfoTy PartialStruct; 8593 generateInfoForComponentList( 8594 C->getMapType(), C->getMapTypeModifiers(), L.second, BasePointers, 8595 Pointers, Sizes, Types, PartialStruct, 8596 /*IsFirstComponentList=*/true, C->isImplicit()); 8597 assert(!PartialStruct.Base.isValid() && 8598 "No partial structs for declare target link expected."); 8599 } 8600 } 8601 } 8602 8603 /// Generate the default map information for a given capture \a CI, 8604 /// record field declaration \a RI and captured value \a CV. 8605 void generateDefaultMapInfo(const CapturedStmt::Capture &CI, 8606 const FieldDecl &RI, llvm::Value *CV, 8607 MapBaseValuesArrayTy &CurBasePointers, 8608 MapValuesArrayTy &CurPointers, 8609 MapValuesArrayTy &CurSizes, 8610 MapFlagsArrayTy &CurMapTypes) const { 8611 bool IsImplicit = true; 8612 // Do the default mapping. 8613 if (CI.capturesThis()) { 8614 CurBasePointers.push_back(CV); 8615 CurPointers.push_back(CV); 8616 const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr()); 8617 CurSizes.push_back( 8618 CGF.Builder.CreateIntCast(CGF.getTypeSize(PtrTy->getPointeeType()), 8619 CGF.Int64Ty, /*isSigned=*/true)); 8620 // Default map type. 8621 CurMapTypes.push_back(OMP_MAP_TO | OMP_MAP_FROM); 8622 } else if (CI.capturesVariableByCopy()) { 8623 CurBasePointers.push_back(CV); 8624 CurPointers.push_back(CV); 8625 if (!RI.getType()->isAnyPointerType()) { 8626 // We have to signal to the runtime captures passed by value that are 8627 // not pointers. 8628 CurMapTypes.push_back(OMP_MAP_LITERAL); 8629 CurSizes.push_back(CGF.Builder.CreateIntCast( 8630 CGF.getTypeSize(RI.getType()), CGF.Int64Ty, /*isSigned=*/true)); 8631 } else { 8632 // Pointers are implicitly mapped with a zero size and no flags 8633 // (other than first map that is added for all implicit maps). 8634 CurMapTypes.push_back(OMP_MAP_NONE); 8635 CurSizes.push_back(llvm::Constant::getNullValue(CGF.Int64Ty)); 8636 } 8637 const VarDecl *VD = CI.getCapturedVar(); 8638 auto I = FirstPrivateDecls.find(VD); 8639 if (I != FirstPrivateDecls.end()) 8640 IsImplicit = I->getSecond(); 8641 } else { 8642 assert(CI.capturesVariable() && "Expected captured reference."); 8643 const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr()); 8644 QualType ElementType = PtrTy->getPointeeType(); 8645 CurSizes.push_back(CGF.Builder.CreateIntCast( 8646 CGF.getTypeSize(ElementType), CGF.Int64Ty, /*isSigned=*/true)); 8647 // The default map type for a scalar/complex type is 'to' because by 8648 // default the value doesn't have to be retrieved. For an aggregate 8649 // type, the default is 'tofrom'. 8650 CurMapTypes.push_back(getMapModifiersForPrivateClauses(CI)); 8651 const VarDecl *VD = CI.getCapturedVar(); 8652 auto I = FirstPrivateDecls.find(VD); 8653 if (I != FirstPrivateDecls.end() && 8654 VD->getType().isConstant(CGF.getContext())) { 8655 llvm::Constant *Addr = 8656 CGF.CGM.getOpenMPRuntime().registerTargetFirstprivateCopy(CGF, VD); 8657 // Copy the value of the original variable to the new global copy. 8658 CGF.Builder.CreateMemCpy( 8659 CGF.MakeNaturalAlignAddrLValue(Addr, ElementType).getAddress(), 8660 Address(CV, CGF.getContext().getTypeAlignInChars(ElementType)), 8661 CurSizes.back(), /*IsVolatile=*/false); 8662 // Use new global variable as the base pointers. 8663 CurBasePointers.push_back(Addr); 8664 CurPointers.push_back(Addr); 8665 } else { 8666 CurBasePointers.push_back(CV); 8667 if (I != FirstPrivateDecls.end() && ElementType->isAnyPointerType()) { 8668 Address PtrAddr = CGF.EmitLoadOfReference(CGF.MakeAddrLValue( 8669 CV, ElementType, CGF.getContext().getDeclAlign(VD), 8670 AlignmentSource::Decl)); 8671 CurPointers.push_back(PtrAddr.getPointer()); 8672 } else { 8673 CurPointers.push_back(CV); 8674 } 8675 } 8676 if (I != FirstPrivateDecls.end()) 8677 IsImplicit = I->getSecond(); 8678 } 8679 // Every default map produces a single argument which is a target parameter. 8680 CurMapTypes.back() |= OMP_MAP_TARGET_PARAM; 8681 8682 // Add flag stating this is an implicit map. 8683 if (IsImplicit) 8684 CurMapTypes.back() |= OMP_MAP_IMPLICIT; 8685 } 8686 }; 8687 } // anonymous namespace 8688 8689 /// Emit the arrays used to pass the captures and map information to the 8690 /// offloading runtime library. If there is no map or capture information, 8691 /// return nullptr by reference. 8692 static void 8693 emitOffloadingArrays(CodeGenFunction &CGF, 8694 MappableExprsHandler::MapBaseValuesArrayTy &BasePointers, 8695 MappableExprsHandler::MapValuesArrayTy &Pointers, 8696 MappableExprsHandler::MapValuesArrayTy &Sizes, 8697 MappableExprsHandler::MapFlagsArrayTy &MapTypes, 8698 CGOpenMPRuntime::TargetDataInfo &Info) { 8699 CodeGenModule &CGM = CGF.CGM; 8700 ASTContext &Ctx = CGF.getContext(); 8701 8702 // Reset the array information. 8703 Info.clearArrayInfo(); 8704 Info.NumberOfPtrs = BasePointers.size(); 8705 8706 if (Info.NumberOfPtrs) { 8707 // Detect if we have any capture size requiring runtime evaluation of the 8708 // size so that a constant array could be eventually used. 8709 bool hasRuntimeEvaluationCaptureSize = false; 8710 for (llvm::Value *S : Sizes) 8711 if (!isa<llvm::Constant>(S)) { 8712 hasRuntimeEvaluationCaptureSize = true; 8713 break; 8714 } 8715 8716 llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true); 8717 QualType PointerArrayType = Ctx.getConstantArrayType( 8718 Ctx.VoidPtrTy, PointerNumAP, nullptr, ArrayType::Normal, 8719 /*IndexTypeQuals=*/0); 8720 8721 Info.BasePointersArray = 8722 CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer(); 8723 Info.PointersArray = 8724 CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer(); 8725 8726 // If we don't have any VLA types or other types that require runtime 8727 // evaluation, we can use a constant array for the map sizes, otherwise we 8728 // need to fill up the arrays as we do for the pointers. 8729 QualType Int64Ty = 8730 Ctx.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 8731 if (hasRuntimeEvaluationCaptureSize) { 8732 QualType SizeArrayType = Ctx.getConstantArrayType( 8733 Int64Ty, PointerNumAP, nullptr, ArrayType::Normal, 8734 /*IndexTypeQuals=*/0); 8735 Info.SizesArray = 8736 CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer(); 8737 } else { 8738 // We expect all the sizes to be constant, so we collect them to create 8739 // a constant array. 8740 SmallVector<llvm::Constant *, 16> ConstSizes; 8741 for (llvm::Value *S : Sizes) 8742 ConstSizes.push_back(cast<llvm::Constant>(S)); 8743 8744 auto *SizesArrayInit = llvm::ConstantArray::get( 8745 llvm::ArrayType::get(CGM.Int64Ty, ConstSizes.size()), ConstSizes); 8746 std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"}); 8747 auto *SizesArrayGbl = new llvm::GlobalVariable( 8748 CGM.getModule(), SizesArrayInit->getType(), 8749 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 8750 SizesArrayInit, Name); 8751 SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 8752 Info.SizesArray = SizesArrayGbl; 8753 } 8754 8755 // The map types are always constant so we don't need to generate code to 8756 // fill arrays. Instead, we create an array constant. 8757 SmallVector<uint64_t, 4> Mapping(MapTypes.size(), 0); 8758 llvm::copy(MapTypes, Mapping.begin()); 8759 llvm::Constant *MapTypesArrayInit = 8760 llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping); 8761 std::string MaptypesName = 8762 CGM.getOpenMPRuntime().getName({"offload_maptypes"}); 8763 auto *MapTypesArrayGbl = new llvm::GlobalVariable( 8764 CGM.getModule(), MapTypesArrayInit->getType(), 8765 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 8766 MapTypesArrayInit, MaptypesName); 8767 MapTypesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 8768 Info.MapTypesArray = MapTypesArrayGbl; 8769 8770 for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) { 8771 llvm::Value *BPVal = *BasePointers[I]; 8772 llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32( 8773 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8774 Info.BasePointersArray, 0, I); 8775 BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8776 BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0)); 8777 Address BPAddr(BP, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 8778 CGF.Builder.CreateStore(BPVal, BPAddr); 8779 8780 if (Info.requiresDevicePointerInfo()) 8781 if (const ValueDecl *DevVD = BasePointers[I].getDevicePtrDecl()) 8782 Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr); 8783 8784 llvm::Value *PVal = Pointers[I]; 8785 llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32( 8786 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8787 Info.PointersArray, 0, I); 8788 P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8789 P, PVal->getType()->getPointerTo(/*AddrSpace=*/0)); 8790 Address PAddr(P, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 8791 CGF.Builder.CreateStore(PVal, PAddr); 8792 8793 if (hasRuntimeEvaluationCaptureSize) { 8794 llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32( 8795 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 8796 Info.SizesArray, 8797 /*Idx0=*/0, 8798 /*Idx1=*/I); 8799 Address SAddr(S, Ctx.getTypeAlignInChars(Int64Ty)); 8800 CGF.Builder.CreateStore( 8801 CGF.Builder.CreateIntCast(Sizes[I], CGM.Int64Ty, /*isSigned=*/true), 8802 SAddr); 8803 } 8804 } 8805 } 8806 } 8807 8808 /// Emit the arguments to be passed to the runtime library based on the 8809 /// arrays of pointers, sizes and map types. 8810 static void emitOffloadingArraysArgument( 8811 CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg, 8812 llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg, 8813 llvm::Value *&MapTypesArrayArg, CGOpenMPRuntime::TargetDataInfo &Info) { 8814 CodeGenModule &CGM = CGF.CGM; 8815 if (Info.NumberOfPtrs) { 8816 BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8817 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8818 Info.BasePointersArray, 8819 /*Idx0=*/0, /*Idx1=*/0); 8820 PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8821 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8822 Info.PointersArray, 8823 /*Idx0=*/0, 8824 /*Idx1=*/0); 8825 SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8826 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), Info.SizesArray, 8827 /*Idx0=*/0, /*Idx1=*/0); 8828 MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8829 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 8830 Info.MapTypesArray, 8831 /*Idx0=*/0, 8832 /*Idx1=*/0); 8833 } else { 8834 BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 8835 PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 8836 SizesArrayArg = llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 8837 MapTypesArrayArg = 8838 llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 8839 } 8840 } 8841 8842 /// Check for inner distribute directive. 8843 static const OMPExecutableDirective * 8844 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) { 8845 const auto *CS = D.getInnermostCapturedStmt(); 8846 const auto *Body = 8847 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 8848 const Stmt *ChildStmt = 8849 CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 8850 8851 if (const auto *NestedDir = 8852 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 8853 OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind(); 8854 switch (D.getDirectiveKind()) { 8855 case OMPD_target: 8856 if (isOpenMPDistributeDirective(DKind)) 8857 return NestedDir; 8858 if (DKind == OMPD_teams) { 8859 Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers( 8860 /*IgnoreCaptured=*/true); 8861 if (!Body) 8862 return nullptr; 8863 ChildStmt = CGOpenMPSIMDRuntime::getSingleCompoundChild(Ctx, Body); 8864 if (const auto *NND = 8865 dyn_cast_or_null<OMPExecutableDirective>(ChildStmt)) { 8866 DKind = NND->getDirectiveKind(); 8867 if (isOpenMPDistributeDirective(DKind)) 8868 return NND; 8869 } 8870 } 8871 return nullptr; 8872 case OMPD_target_teams: 8873 if (isOpenMPDistributeDirective(DKind)) 8874 return NestedDir; 8875 return nullptr; 8876 case OMPD_target_parallel: 8877 case OMPD_target_simd: 8878 case OMPD_target_parallel_for: 8879 case OMPD_target_parallel_for_simd: 8880 return nullptr; 8881 case OMPD_target_teams_distribute: 8882 case OMPD_target_teams_distribute_simd: 8883 case OMPD_target_teams_distribute_parallel_for: 8884 case OMPD_target_teams_distribute_parallel_for_simd: 8885 case OMPD_parallel: 8886 case OMPD_for: 8887 case OMPD_parallel_for: 8888 case OMPD_parallel_sections: 8889 case OMPD_for_simd: 8890 case OMPD_parallel_for_simd: 8891 case OMPD_cancel: 8892 case OMPD_cancellation_point: 8893 case OMPD_ordered: 8894 case OMPD_threadprivate: 8895 case OMPD_allocate: 8896 case OMPD_task: 8897 case OMPD_simd: 8898 case OMPD_sections: 8899 case OMPD_section: 8900 case OMPD_single: 8901 case OMPD_master: 8902 case OMPD_critical: 8903 case OMPD_taskyield: 8904 case OMPD_barrier: 8905 case OMPD_taskwait: 8906 case OMPD_taskgroup: 8907 case OMPD_atomic: 8908 case OMPD_flush: 8909 case OMPD_teams: 8910 case OMPD_target_data: 8911 case OMPD_target_exit_data: 8912 case OMPD_target_enter_data: 8913 case OMPD_distribute: 8914 case OMPD_distribute_simd: 8915 case OMPD_distribute_parallel_for: 8916 case OMPD_distribute_parallel_for_simd: 8917 case OMPD_teams_distribute: 8918 case OMPD_teams_distribute_simd: 8919 case OMPD_teams_distribute_parallel_for: 8920 case OMPD_teams_distribute_parallel_for_simd: 8921 case OMPD_target_update: 8922 case OMPD_declare_simd: 8923 case OMPD_declare_variant: 8924 case OMPD_declare_target: 8925 case OMPD_end_declare_target: 8926 case OMPD_declare_reduction: 8927 case OMPD_declare_mapper: 8928 case OMPD_taskloop: 8929 case OMPD_taskloop_simd: 8930 case OMPD_master_taskloop: 8931 case OMPD_requires: 8932 case OMPD_unknown: 8933 llvm_unreachable("Unexpected directive."); 8934 } 8935 } 8936 8937 return nullptr; 8938 } 8939 8940 /// Emit the user-defined mapper function. The code generation follows the 8941 /// pattern in the example below. 8942 /// \code 8943 /// void .omp_mapper.<type_name>.<mapper_id>.(void *rt_mapper_handle, 8944 /// void *base, void *begin, 8945 /// int64_t size, int64_t type) { 8946 /// // Allocate space for an array section first. 8947 /// if (size > 1 && !maptype.IsDelete) 8948 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 8949 /// size*sizeof(Ty), clearToFrom(type)); 8950 /// // Map members. 8951 /// for (unsigned i = 0; i < size; i++) { 8952 /// // For each component specified by this mapper: 8953 /// for (auto c : all_components) { 8954 /// if (c.hasMapper()) 8955 /// (*c.Mapper())(rt_mapper_handle, c.arg_base, c.arg_begin, c.arg_size, 8956 /// c.arg_type); 8957 /// else 8958 /// __tgt_push_mapper_component(rt_mapper_handle, c.arg_base, 8959 /// c.arg_begin, c.arg_size, c.arg_type); 8960 /// } 8961 /// } 8962 /// // Delete the array section. 8963 /// if (size > 1 && maptype.IsDelete) 8964 /// __tgt_push_mapper_component(rt_mapper_handle, base, begin, 8965 /// size*sizeof(Ty), clearToFrom(type)); 8966 /// } 8967 /// \endcode 8968 void CGOpenMPRuntime::emitUserDefinedMapper(const OMPDeclareMapperDecl *D, 8969 CodeGenFunction *CGF) { 8970 if (UDMMap.count(D) > 0) 8971 return; 8972 ASTContext &C = CGM.getContext(); 8973 QualType Ty = D->getType(); 8974 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 8975 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 8976 auto *MapperVarDecl = 8977 cast<VarDecl>(cast<DeclRefExpr>(D->getMapperVarRef())->getDecl()); 8978 SourceLocation Loc = D->getLocation(); 8979 CharUnits ElementSize = C.getTypeSizeInChars(Ty); 8980 8981 // Prepare mapper function arguments and attributes. 8982 ImplicitParamDecl HandleArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 8983 C.VoidPtrTy, ImplicitParamDecl::Other); 8984 ImplicitParamDecl BaseArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 8985 ImplicitParamDecl::Other); 8986 ImplicitParamDecl BeginArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 8987 C.VoidPtrTy, ImplicitParamDecl::Other); 8988 ImplicitParamDecl SizeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 8989 ImplicitParamDecl::Other); 8990 ImplicitParamDecl TypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, Int64Ty, 8991 ImplicitParamDecl::Other); 8992 FunctionArgList Args; 8993 Args.push_back(&HandleArg); 8994 Args.push_back(&BaseArg); 8995 Args.push_back(&BeginArg); 8996 Args.push_back(&SizeArg); 8997 Args.push_back(&TypeArg); 8998 const CGFunctionInfo &FnInfo = 8999 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 9000 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 9001 SmallString<64> TyStr; 9002 llvm::raw_svector_ostream Out(TyStr); 9003 CGM.getCXXABI().getMangleContext().mangleTypeName(Ty, Out); 9004 std::string Name = getName({"omp_mapper", TyStr, D->getName()}); 9005 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 9006 Name, &CGM.getModule()); 9007 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 9008 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 9009 // Start the mapper function code generation. 9010 CodeGenFunction MapperCGF(CGM); 9011 MapperCGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 9012 // Compute the starting and end addreses of array elements. 9013 llvm::Value *Size = MapperCGF.EmitLoadOfScalar( 9014 MapperCGF.GetAddrOfLocalVar(&SizeArg), /*Volatile=*/false, 9015 C.getPointerType(Int64Ty), Loc); 9016 llvm::Value *PtrBegin = MapperCGF.Builder.CreateBitCast( 9017 MapperCGF.GetAddrOfLocalVar(&BeginArg).getPointer(), 9018 CGM.getTypes().ConvertTypeForMem(C.getPointerType(PtrTy))); 9019 llvm::Value *PtrEnd = MapperCGF.Builder.CreateGEP(PtrBegin, Size); 9020 llvm::Value *MapType = MapperCGF.EmitLoadOfScalar( 9021 MapperCGF.GetAddrOfLocalVar(&TypeArg), /*Volatile=*/false, 9022 C.getPointerType(Int64Ty), Loc); 9023 // Prepare common arguments for array initiation and deletion. 9024 llvm::Value *Handle = MapperCGF.EmitLoadOfScalar( 9025 MapperCGF.GetAddrOfLocalVar(&HandleArg), 9026 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9027 llvm::Value *BaseIn = MapperCGF.EmitLoadOfScalar( 9028 MapperCGF.GetAddrOfLocalVar(&BaseArg), 9029 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9030 llvm::Value *BeginIn = MapperCGF.EmitLoadOfScalar( 9031 MapperCGF.GetAddrOfLocalVar(&BeginArg), 9032 /*Volatile=*/false, C.getPointerType(C.VoidPtrTy), Loc); 9033 9034 // Emit array initiation if this is an array section and \p MapType indicates 9035 // that memory allocation is required. 9036 llvm::BasicBlock *HeadBB = MapperCGF.createBasicBlock("omp.arraymap.head"); 9037 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 9038 ElementSize, HeadBB, /*IsInit=*/true); 9039 9040 // Emit a for loop to iterate through SizeArg of elements and map all of them. 9041 9042 // Emit the loop header block. 9043 MapperCGF.EmitBlock(HeadBB); 9044 llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.arraymap.body"); 9045 llvm::BasicBlock *DoneBB = MapperCGF.createBasicBlock("omp.done"); 9046 // Evaluate whether the initial condition is satisfied. 9047 llvm::Value *IsEmpty = 9048 MapperCGF.Builder.CreateICmpEQ(PtrBegin, PtrEnd, "omp.arraymap.isempty"); 9049 MapperCGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 9050 llvm::BasicBlock *EntryBB = MapperCGF.Builder.GetInsertBlock(); 9051 9052 // Emit the loop body block. 9053 MapperCGF.EmitBlock(BodyBB); 9054 llvm::PHINode *PtrPHI = MapperCGF.Builder.CreatePHI( 9055 PtrBegin->getType(), 2, "omp.arraymap.ptrcurrent"); 9056 PtrPHI->addIncoming(PtrBegin, EntryBB); 9057 Address PtrCurrent = 9058 Address(PtrPHI, MapperCGF.GetAddrOfLocalVar(&BeginArg) 9059 .getAlignment() 9060 .alignmentOfArrayElement(ElementSize)); 9061 // Privatize the declared variable of mapper to be the current array element. 9062 CodeGenFunction::OMPPrivateScope Scope(MapperCGF); 9063 Scope.addPrivate(MapperVarDecl, [&MapperCGF, PtrCurrent, PtrTy]() { 9064 return MapperCGF 9065 .EmitLoadOfPointerLValue(PtrCurrent, PtrTy->castAs<PointerType>()) 9066 .getAddress(); 9067 }); 9068 (void)Scope.Privatize(); 9069 9070 // Get map clause information. Fill up the arrays with all mapped variables. 9071 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 9072 MappableExprsHandler::MapValuesArrayTy Pointers; 9073 MappableExprsHandler::MapValuesArrayTy Sizes; 9074 MappableExprsHandler::MapFlagsArrayTy MapTypes; 9075 MappableExprsHandler MEHandler(*D, MapperCGF); 9076 MEHandler.generateAllInfoForMapper(BasePointers, Pointers, Sizes, MapTypes); 9077 9078 // Call the runtime API __tgt_mapper_num_components to get the number of 9079 // pre-existing components. 9080 llvm::Value *OffloadingArgs[] = {Handle}; 9081 llvm::Value *PreviousSize = MapperCGF.EmitRuntimeCall( 9082 createRuntimeFunction(OMPRTL__tgt_mapper_num_components), OffloadingArgs); 9083 llvm::Value *ShiftedPreviousSize = MapperCGF.Builder.CreateShl( 9084 PreviousSize, 9085 MapperCGF.Builder.getInt64(MappableExprsHandler::getFlagMemberOffset())); 9086 9087 // Fill up the runtime mapper handle for all components. 9088 for (unsigned I = 0; I < BasePointers.size(); ++I) { 9089 llvm::Value *CurBaseArg = MapperCGF.Builder.CreateBitCast( 9090 *BasePointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 9091 llvm::Value *CurBeginArg = MapperCGF.Builder.CreateBitCast( 9092 Pointers[I], CGM.getTypes().ConvertTypeForMem(C.VoidPtrTy)); 9093 llvm::Value *CurSizeArg = Sizes[I]; 9094 9095 // Extract the MEMBER_OF field from the map type. 9096 llvm::BasicBlock *MemberBB = MapperCGF.createBasicBlock("omp.member"); 9097 MapperCGF.EmitBlock(MemberBB); 9098 llvm::Value *OriMapType = MapperCGF.Builder.getInt64(MapTypes[I]); 9099 llvm::Value *Member = MapperCGF.Builder.CreateAnd( 9100 OriMapType, 9101 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_MEMBER_OF)); 9102 llvm::BasicBlock *MemberCombineBB = 9103 MapperCGF.createBasicBlock("omp.member.combine"); 9104 llvm::BasicBlock *TypeBB = MapperCGF.createBasicBlock("omp.type"); 9105 llvm::Value *IsMember = MapperCGF.Builder.CreateIsNull(Member); 9106 MapperCGF.Builder.CreateCondBr(IsMember, TypeBB, MemberCombineBB); 9107 // Add the number of pre-existing components to the MEMBER_OF field if it 9108 // is valid. 9109 MapperCGF.EmitBlock(MemberCombineBB); 9110 llvm::Value *CombinedMember = 9111 MapperCGF.Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize); 9112 // Do nothing if it is not a member of previous components. 9113 MapperCGF.EmitBlock(TypeBB); 9114 llvm::PHINode *MemberMapType = 9115 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.membermaptype"); 9116 MemberMapType->addIncoming(OriMapType, MemberBB); 9117 MemberMapType->addIncoming(CombinedMember, MemberCombineBB); 9118 9119 // Combine the map type inherited from user-defined mapper with that 9120 // specified in the program. According to the OMP_MAP_TO and OMP_MAP_FROM 9121 // bits of the \a MapType, which is the input argument of the mapper 9122 // function, the following code will set the OMP_MAP_TO and OMP_MAP_FROM 9123 // bits of MemberMapType. 9124 // [OpenMP 5.0], 1.2.6. map-type decay. 9125 // | alloc | to | from | tofrom | release | delete 9126 // ---------------------------------------------------------- 9127 // alloc | alloc | alloc | alloc | alloc | release | delete 9128 // to | alloc | to | alloc | to | release | delete 9129 // from | alloc | alloc | from | from | release | delete 9130 // tofrom | alloc | to | from | tofrom | release | delete 9131 llvm::Value *LeftToFrom = MapperCGF.Builder.CreateAnd( 9132 MapType, 9133 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO | 9134 MappableExprsHandler::OMP_MAP_FROM)); 9135 llvm::BasicBlock *AllocBB = MapperCGF.createBasicBlock("omp.type.alloc"); 9136 llvm::BasicBlock *AllocElseBB = 9137 MapperCGF.createBasicBlock("omp.type.alloc.else"); 9138 llvm::BasicBlock *ToBB = MapperCGF.createBasicBlock("omp.type.to"); 9139 llvm::BasicBlock *ToElseBB = MapperCGF.createBasicBlock("omp.type.to.else"); 9140 llvm::BasicBlock *FromBB = MapperCGF.createBasicBlock("omp.type.from"); 9141 llvm::BasicBlock *EndBB = MapperCGF.createBasicBlock("omp.type.end"); 9142 llvm::Value *IsAlloc = MapperCGF.Builder.CreateIsNull(LeftToFrom); 9143 MapperCGF.Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB); 9144 // In case of alloc, clear OMP_MAP_TO and OMP_MAP_FROM. 9145 MapperCGF.EmitBlock(AllocBB); 9146 llvm::Value *AllocMapType = MapperCGF.Builder.CreateAnd( 9147 MemberMapType, 9148 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 9149 MappableExprsHandler::OMP_MAP_FROM))); 9150 MapperCGF.Builder.CreateBr(EndBB); 9151 MapperCGF.EmitBlock(AllocElseBB); 9152 llvm::Value *IsTo = MapperCGF.Builder.CreateICmpEQ( 9153 LeftToFrom, 9154 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_TO)); 9155 MapperCGF.Builder.CreateCondBr(IsTo, ToBB, ToElseBB); 9156 // In case of to, clear OMP_MAP_FROM. 9157 MapperCGF.EmitBlock(ToBB); 9158 llvm::Value *ToMapType = MapperCGF.Builder.CreateAnd( 9159 MemberMapType, 9160 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_FROM)); 9161 MapperCGF.Builder.CreateBr(EndBB); 9162 MapperCGF.EmitBlock(ToElseBB); 9163 llvm::Value *IsFrom = MapperCGF.Builder.CreateICmpEQ( 9164 LeftToFrom, 9165 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_FROM)); 9166 MapperCGF.Builder.CreateCondBr(IsFrom, FromBB, EndBB); 9167 // In case of from, clear OMP_MAP_TO. 9168 MapperCGF.EmitBlock(FromBB); 9169 llvm::Value *FromMapType = MapperCGF.Builder.CreateAnd( 9170 MemberMapType, 9171 MapperCGF.Builder.getInt64(~MappableExprsHandler::OMP_MAP_TO)); 9172 // In case of tofrom, do nothing. 9173 MapperCGF.EmitBlock(EndBB); 9174 llvm::PHINode *CurMapType = 9175 MapperCGF.Builder.CreatePHI(CGM.Int64Ty, 4, "omp.maptype"); 9176 CurMapType->addIncoming(AllocMapType, AllocBB); 9177 CurMapType->addIncoming(ToMapType, ToBB); 9178 CurMapType->addIncoming(FromMapType, FromBB); 9179 CurMapType->addIncoming(MemberMapType, ToElseBB); 9180 9181 // TODO: call the corresponding mapper function if a user-defined mapper is 9182 // associated with this map clause. 9183 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 9184 // data structure. 9185 llvm::Value *OffloadingArgs[] = {Handle, CurBaseArg, CurBeginArg, 9186 CurSizeArg, CurMapType}; 9187 MapperCGF.EmitRuntimeCall( 9188 createRuntimeFunction(OMPRTL__tgt_push_mapper_component), 9189 OffloadingArgs); 9190 } 9191 9192 // Update the pointer to point to the next element that needs to be mapped, 9193 // and check whether we have mapped all elements. 9194 llvm::Value *PtrNext = MapperCGF.Builder.CreateConstGEP1_32( 9195 PtrPHI, /*Idx0=*/1, "omp.arraymap.next"); 9196 PtrPHI->addIncoming(PtrNext, BodyBB); 9197 llvm::Value *IsDone = 9198 MapperCGF.Builder.CreateICmpEQ(PtrNext, PtrEnd, "omp.arraymap.isdone"); 9199 llvm::BasicBlock *ExitBB = MapperCGF.createBasicBlock("omp.arraymap.exit"); 9200 MapperCGF.Builder.CreateCondBr(IsDone, ExitBB, BodyBB); 9201 9202 MapperCGF.EmitBlock(ExitBB); 9203 // Emit array deletion if this is an array section and \p MapType indicates 9204 // that deletion is required. 9205 emitUDMapperArrayInitOrDel(MapperCGF, Handle, BaseIn, BeginIn, Size, MapType, 9206 ElementSize, DoneBB, /*IsInit=*/false); 9207 9208 // Emit the function exit block. 9209 MapperCGF.EmitBlock(DoneBB, /*IsFinished=*/true); 9210 MapperCGF.FinishFunction(); 9211 UDMMap.try_emplace(D, Fn); 9212 if (CGF) { 9213 auto &Decls = FunctionUDMMap.FindAndConstruct(CGF->CurFn); 9214 Decls.second.push_back(D); 9215 } 9216 } 9217 9218 /// Emit the array initialization or deletion portion for user-defined mapper 9219 /// code generation. First, it evaluates whether an array section is mapped and 9220 /// whether the \a MapType instructs to delete this section. If \a IsInit is 9221 /// true, and \a MapType indicates to not delete this array, array 9222 /// initialization code is generated. If \a IsInit is false, and \a MapType 9223 /// indicates to not this array, array deletion code is generated. 9224 void CGOpenMPRuntime::emitUDMapperArrayInitOrDel( 9225 CodeGenFunction &MapperCGF, llvm::Value *Handle, llvm::Value *Base, 9226 llvm::Value *Begin, llvm::Value *Size, llvm::Value *MapType, 9227 CharUnits ElementSize, llvm::BasicBlock *ExitBB, bool IsInit) { 9228 StringRef Prefix = IsInit ? ".init" : ".del"; 9229 9230 // Evaluate if this is an array section. 9231 llvm::BasicBlock *IsDeleteBB = 9232 MapperCGF.createBasicBlock("omp.array" + Prefix + ".evaldelete"); 9233 llvm::BasicBlock *BodyBB = MapperCGF.createBasicBlock("omp.array" + Prefix); 9234 llvm::Value *IsArray = MapperCGF.Builder.CreateICmpSGE( 9235 Size, MapperCGF.Builder.getInt64(1), "omp.arrayinit.isarray"); 9236 MapperCGF.Builder.CreateCondBr(IsArray, IsDeleteBB, ExitBB); 9237 9238 // Evaluate if we are going to delete this section. 9239 MapperCGF.EmitBlock(IsDeleteBB); 9240 llvm::Value *DeleteBit = MapperCGF.Builder.CreateAnd( 9241 MapType, 9242 MapperCGF.Builder.getInt64(MappableExprsHandler::OMP_MAP_DELETE)); 9243 llvm::Value *DeleteCond; 9244 if (IsInit) { 9245 DeleteCond = MapperCGF.Builder.CreateIsNull( 9246 DeleteBit, "omp.array" + Prefix + ".delete"); 9247 } else { 9248 DeleteCond = MapperCGF.Builder.CreateIsNotNull( 9249 DeleteBit, "omp.array" + Prefix + ".delete"); 9250 } 9251 MapperCGF.Builder.CreateCondBr(DeleteCond, BodyBB, ExitBB); 9252 9253 MapperCGF.EmitBlock(BodyBB); 9254 // Get the array size by multiplying element size and element number (i.e., \p 9255 // Size). 9256 llvm::Value *ArraySize = MapperCGF.Builder.CreateNUWMul( 9257 Size, MapperCGF.Builder.getInt64(ElementSize.getQuantity())); 9258 // Remove OMP_MAP_TO and OMP_MAP_FROM from the map type, so that it achieves 9259 // memory allocation/deletion purpose only. 9260 llvm::Value *MapTypeArg = MapperCGF.Builder.CreateAnd( 9261 MapType, 9262 MapperCGF.Builder.getInt64(~(MappableExprsHandler::OMP_MAP_TO | 9263 MappableExprsHandler::OMP_MAP_FROM))); 9264 // Call the runtime API __tgt_push_mapper_component to fill up the runtime 9265 // data structure. 9266 llvm::Value *OffloadingArgs[] = {Handle, Base, Begin, ArraySize, MapTypeArg}; 9267 MapperCGF.EmitRuntimeCall( 9268 createRuntimeFunction(OMPRTL__tgt_push_mapper_component), OffloadingArgs); 9269 } 9270 9271 void CGOpenMPRuntime::emitTargetNumIterationsCall( 9272 CodeGenFunction &CGF, const OMPExecutableDirective &D, 9273 llvm::Value *DeviceID, 9274 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 9275 const OMPLoopDirective &D)> 9276 SizeEmitter) { 9277 OpenMPDirectiveKind Kind = D.getDirectiveKind(); 9278 const OMPExecutableDirective *TD = &D; 9279 // Get nested teams distribute kind directive, if any. 9280 if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind)) 9281 TD = getNestedDistributeDirective(CGM.getContext(), D); 9282 if (!TD) 9283 return; 9284 const auto *LD = cast<OMPLoopDirective>(TD); 9285 auto &&CodeGen = [LD, DeviceID, SizeEmitter, this](CodeGenFunction &CGF, 9286 PrePostActionTy &) { 9287 if (llvm::Value *NumIterations = SizeEmitter(CGF, *LD)) { 9288 llvm::Value *Args[] = {DeviceID, NumIterations}; 9289 CGF.EmitRuntimeCall( 9290 createRuntimeFunction(OMPRTL__kmpc_push_target_tripcount), Args); 9291 } 9292 }; 9293 emitInlinedDirective(CGF, OMPD_unknown, CodeGen); 9294 } 9295 9296 void CGOpenMPRuntime::emitTargetCall( 9297 CodeGenFunction &CGF, const OMPExecutableDirective &D, 9298 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 9299 const Expr *Device, 9300 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 9301 const OMPLoopDirective &D)> 9302 SizeEmitter) { 9303 if (!CGF.HaveInsertPoint()) 9304 return; 9305 9306 assert(OutlinedFn && "Invalid outlined function!"); 9307 9308 const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>(); 9309 llvm::SmallVector<llvm::Value *, 16> CapturedVars; 9310 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 9311 auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF, 9312 PrePostActionTy &) { 9313 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9314 }; 9315 emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen); 9316 9317 CodeGenFunction::OMPTargetDataInfo InputInfo; 9318 llvm::Value *MapTypesArray = nullptr; 9319 // Fill up the pointer arrays and transfer execution to the device. 9320 auto &&ThenGen = [this, Device, OutlinedFn, OutlinedFnID, &D, &InputInfo, 9321 &MapTypesArray, &CS, RequiresOuterTask, &CapturedVars, 9322 SizeEmitter](CodeGenFunction &CGF, PrePostActionTy &) { 9323 // On top of the arrays that were filled up, the target offloading call 9324 // takes as arguments the device id as well as the host pointer. The host 9325 // pointer is used by the runtime library to identify the current target 9326 // region, so it only has to be unique and not necessarily point to 9327 // anything. It could be the pointer to the outlined function that 9328 // implements the target region, but we aren't using that so that the 9329 // compiler doesn't need to keep that, and could therefore inline the host 9330 // function if proven worthwhile during optimization. 9331 9332 // From this point on, we need to have an ID of the target region defined. 9333 assert(OutlinedFnID && "Invalid outlined function ID!"); 9334 9335 // Emit device ID if any. 9336 llvm::Value *DeviceID; 9337 if (Device) { 9338 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 9339 CGF.Int64Ty, /*isSigned=*/true); 9340 } else { 9341 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 9342 } 9343 9344 // Emit the number of elements in the offloading arrays. 9345 llvm::Value *PointerNum = 9346 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 9347 9348 // Return value of the runtime offloading call. 9349 llvm::Value *Return; 9350 9351 llvm::Value *NumTeams = emitNumTeamsForTargetDirective(CGF, D); 9352 llvm::Value *NumThreads = emitNumThreadsForTargetDirective(CGF, D); 9353 9354 // Emit tripcount for the target loop-based directive. 9355 emitTargetNumIterationsCall(CGF, D, DeviceID, SizeEmitter); 9356 9357 bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 9358 // The target region is an outlined function launched by the runtime 9359 // via calls __tgt_target() or __tgt_target_teams(). 9360 // 9361 // __tgt_target() launches a target region with one team and one thread, 9362 // executing a serial region. This master thread may in turn launch 9363 // more threads within its team upon encountering a parallel region, 9364 // however, no additional teams can be launched on the device. 9365 // 9366 // __tgt_target_teams() launches a target region with one or more teams, 9367 // each with one or more threads. This call is required for target 9368 // constructs such as: 9369 // 'target teams' 9370 // 'target' / 'teams' 9371 // 'target teams distribute parallel for' 9372 // 'target parallel' 9373 // and so on. 9374 // 9375 // Note that on the host and CPU targets, the runtime implementation of 9376 // these calls simply call the outlined function without forking threads. 9377 // The outlined functions themselves have runtime calls to 9378 // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by 9379 // the compiler in emitTeamsCall() and emitParallelCall(). 9380 // 9381 // In contrast, on the NVPTX target, the implementation of 9382 // __tgt_target_teams() launches a GPU kernel with the requested number 9383 // of teams and threads so no additional calls to the runtime are required. 9384 if (NumTeams) { 9385 // If we have NumTeams defined this means that we have an enclosed teams 9386 // region. Therefore we also expect to have NumThreads defined. These two 9387 // values should be defined in the presence of a teams directive, 9388 // regardless of having any clauses associated. If the user is using teams 9389 // but no clauses, these two values will be the default that should be 9390 // passed to the runtime library - a 32-bit integer with the value zero. 9391 assert(NumThreads && "Thread limit expression should be available along " 9392 "with number of teams."); 9393 llvm::Value *OffloadingArgs[] = {DeviceID, 9394 OutlinedFnID, 9395 PointerNum, 9396 InputInfo.BasePointersArray.getPointer(), 9397 InputInfo.PointersArray.getPointer(), 9398 InputInfo.SizesArray.getPointer(), 9399 MapTypesArray, 9400 NumTeams, 9401 NumThreads}; 9402 Return = CGF.EmitRuntimeCall( 9403 createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_teams_nowait 9404 : OMPRTL__tgt_target_teams), 9405 OffloadingArgs); 9406 } else { 9407 llvm::Value *OffloadingArgs[] = {DeviceID, 9408 OutlinedFnID, 9409 PointerNum, 9410 InputInfo.BasePointersArray.getPointer(), 9411 InputInfo.PointersArray.getPointer(), 9412 InputInfo.SizesArray.getPointer(), 9413 MapTypesArray}; 9414 Return = CGF.EmitRuntimeCall( 9415 createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_nowait 9416 : OMPRTL__tgt_target), 9417 OffloadingArgs); 9418 } 9419 9420 // Check the error code and execute the host version if required. 9421 llvm::BasicBlock *OffloadFailedBlock = 9422 CGF.createBasicBlock("omp_offload.failed"); 9423 llvm::BasicBlock *OffloadContBlock = 9424 CGF.createBasicBlock("omp_offload.cont"); 9425 llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return); 9426 CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock); 9427 9428 CGF.EmitBlock(OffloadFailedBlock); 9429 if (RequiresOuterTask) { 9430 CapturedVars.clear(); 9431 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9432 } 9433 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9434 CGF.EmitBranch(OffloadContBlock); 9435 9436 CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true); 9437 }; 9438 9439 // Notify that the host version must be executed. 9440 auto &&ElseGen = [this, &D, OutlinedFn, &CS, &CapturedVars, 9441 RequiresOuterTask](CodeGenFunction &CGF, 9442 PrePostActionTy &) { 9443 if (RequiresOuterTask) { 9444 CapturedVars.clear(); 9445 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 9446 } 9447 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 9448 }; 9449 9450 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray, 9451 &CapturedVars, RequiresOuterTask, 9452 &CS](CodeGenFunction &CGF, PrePostActionTy &) { 9453 // Fill up the arrays with all the captured variables. 9454 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 9455 MappableExprsHandler::MapValuesArrayTy Pointers; 9456 MappableExprsHandler::MapValuesArrayTy Sizes; 9457 MappableExprsHandler::MapFlagsArrayTy MapTypes; 9458 9459 // Get mappable expression information. 9460 MappableExprsHandler MEHandler(D, CGF); 9461 llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers; 9462 9463 auto RI = CS.getCapturedRecordDecl()->field_begin(); 9464 auto CV = CapturedVars.begin(); 9465 for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(), 9466 CE = CS.capture_end(); 9467 CI != CE; ++CI, ++RI, ++CV) { 9468 MappableExprsHandler::MapBaseValuesArrayTy CurBasePointers; 9469 MappableExprsHandler::MapValuesArrayTy CurPointers; 9470 MappableExprsHandler::MapValuesArrayTy CurSizes; 9471 MappableExprsHandler::MapFlagsArrayTy CurMapTypes; 9472 MappableExprsHandler::StructRangeInfoTy PartialStruct; 9473 9474 // VLA sizes are passed to the outlined region by copy and do not have map 9475 // information associated. 9476 if (CI->capturesVariableArrayType()) { 9477 CurBasePointers.push_back(*CV); 9478 CurPointers.push_back(*CV); 9479 CurSizes.push_back(CGF.Builder.CreateIntCast( 9480 CGF.getTypeSize(RI->getType()), CGF.Int64Ty, /*isSigned=*/true)); 9481 // Copy to the device as an argument. No need to retrieve it. 9482 CurMapTypes.push_back(MappableExprsHandler::OMP_MAP_LITERAL | 9483 MappableExprsHandler::OMP_MAP_TARGET_PARAM | 9484 MappableExprsHandler::OMP_MAP_IMPLICIT); 9485 } else { 9486 // If we have any information in the map clause, we use it, otherwise we 9487 // just do a default mapping. 9488 MEHandler.generateInfoForCapture(CI, *CV, CurBasePointers, CurPointers, 9489 CurSizes, CurMapTypes, PartialStruct); 9490 if (CurBasePointers.empty()) 9491 MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurBasePointers, 9492 CurPointers, CurSizes, CurMapTypes); 9493 // Generate correct mapping for variables captured by reference in 9494 // lambdas. 9495 if (CI->capturesVariable()) 9496 MEHandler.generateInfoForLambdaCaptures( 9497 CI->getCapturedVar(), *CV, CurBasePointers, CurPointers, CurSizes, 9498 CurMapTypes, LambdaPointers); 9499 } 9500 // We expect to have at least an element of information for this capture. 9501 assert(!CurBasePointers.empty() && 9502 "Non-existing map pointer for capture!"); 9503 assert(CurBasePointers.size() == CurPointers.size() && 9504 CurBasePointers.size() == CurSizes.size() && 9505 CurBasePointers.size() == CurMapTypes.size() && 9506 "Inconsistent map information sizes!"); 9507 9508 // If there is an entry in PartialStruct it means we have a struct with 9509 // individual members mapped. Emit an extra combined entry. 9510 if (PartialStruct.Base.isValid()) 9511 MEHandler.emitCombinedEntry(BasePointers, Pointers, Sizes, MapTypes, 9512 CurMapTypes, PartialStruct); 9513 9514 // We need to append the results of this capture to what we already have. 9515 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 9516 Pointers.append(CurPointers.begin(), CurPointers.end()); 9517 Sizes.append(CurSizes.begin(), CurSizes.end()); 9518 MapTypes.append(CurMapTypes.begin(), CurMapTypes.end()); 9519 } 9520 // Adjust MEMBER_OF flags for the lambdas captures. 9521 MEHandler.adjustMemberOfForLambdaCaptures(LambdaPointers, BasePointers, 9522 Pointers, MapTypes); 9523 // Map other list items in the map clause which are not captured variables 9524 // but "declare target link" global variables. 9525 MEHandler.generateInfoForDeclareTargetLink(BasePointers, Pointers, Sizes, 9526 MapTypes); 9527 9528 TargetDataInfo Info; 9529 // Fill up the arrays and create the arguments. 9530 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 9531 emitOffloadingArraysArgument(CGF, Info.BasePointersArray, 9532 Info.PointersArray, Info.SizesArray, 9533 Info.MapTypesArray, Info); 9534 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 9535 InputInfo.BasePointersArray = 9536 Address(Info.BasePointersArray, CGM.getPointerAlign()); 9537 InputInfo.PointersArray = 9538 Address(Info.PointersArray, CGM.getPointerAlign()); 9539 InputInfo.SizesArray = Address(Info.SizesArray, CGM.getPointerAlign()); 9540 MapTypesArray = Info.MapTypesArray; 9541 if (RequiresOuterTask) 9542 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 9543 else 9544 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 9545 }; 9546 9547 auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask]( 9548 CodeGenFunction &CGF, PrePostActionTy &) { 9549 if (RequiresOuterTask) { 9550 CodeGenFunction::OMPTargetDataInfo InputInfo; 9551 CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo); 9552 } else { 9553 emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen); 9554 } 9555 }; 9556 9557 // If we have a target function ID it means that we need to support 9558 // offloading, otherwise, just execute on the host. We need to execute on host 9559 // regardless of the conditional in the if clause if, e.g., the user do not 9560 // specify target triples. 9561 if (OutlinedFnID) { 9562 if (IfCond) { 9563 emitOMPIfClause(CGF, IfCond, TargetThenGen, TargetElseGen); 9564 } else { 9565 RegionCodeGenTy ThenRCG(TargetThenGen); 9566 ThenRCG(CGF); 9567 } 9568 } else { 9569 RegionCodeGenTy ElseRCG(TargetElseGen); 9570 ElseRCG(CGF); 9571 } 9572 } 9573 9574 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S, 9575 StringRef ParentName) { 9576 if (!S) 9577 return; 9578 9579 // Codegen OMP target directives that offload compute to the device. 9580 bool RequiresDeviceCodegen = 9581 isa<OMPExecutableDirective>(S) && 9582 isOpenMPTargetExecutionDirective( 9583 cast<OMPExecutableDirective>(S)->getDirectiveKind()); 9584 9585 if (RequiresDeviceCodegen) { 9586 const auto &E = *cast<OMPExecutableDirective>(S); 9587 unsigned DeviceID; 9588 unsigned FileID; 9589 unsigned Line; 9590 getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID, 9591 FileID, Line); 9592 9593 // Is this a target region that should not be emitted as an entry point? If 9594 // so just signal we are done with this target region. 9595 if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID, 9596 ParentName, Line)) 9597 return; 9598 9599 switch (E.getDirectiveKind()) { 9600 case OMPD_target: 9601 CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName, 9602 cast<OMPTargetDirective>(E)); 9603 break; 9604 case OMPD_target_parallel: 9605 CodeGenFunction::EmitOMPTargetParallelDeviceFunction( 9606 CGM, ParentName, cast<OMPTargetParallelDirective>(E)); 9607 break; 9608 case OMPD_target_teams: 9609 CodeGenFunction::EmitOMPTargetTeamsDeviceFunction( 9610 CGM, ParentName, cast<OMPTargetTeamsDirective>(E)); 9611 break; 9612 case OMPD_target_teams_distribute: 9613 CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction( 9614 CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E)); 9615 break; 9616 case OMPD_target_teams_distribute_simd: 9617 CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction( 9618 CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E)); 9619 break; 9620 case OMPD_target_parallel_for: 9621 CodeGenFunction::EmitOMPTargetParallelForDeviceFunction( 9622 CGM, ParentName, cast<OMPTargetParallelForDirective>(E)); 9623 break; 9624 case OMPD_target_parallel_for_simd: 9625 CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction( 9626 CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E)); 9627 break; 9628 case OMPD_target_simd: 9629 CodeGenFunction::EmitOMPTargetSimdDeviceFunction( 9630 CGM, ParentName, cast<OMPTargetSimdDirective>(E)); 9631 break; 9632 case OMPD_target_teams_distribute_parallel_for: 9633 CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction( 9634 CGM, ParentName, 9635 cast<OMPTargetTeamsDistributeParallelForDirective>(E)); 9636 break; 9637 case OMPD_target_teams_distribute_parallel_for_simd: 9638 CodeGenFunction:: 9639 EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction( 9640 CGM, ParentName, 9641 cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E)); 9642 break; 9643 case OMPD_parallel: 9644 case OMPD_for: 9645 case OMPD_parallel_for: 9646 case OMPD_parallel_sections: 9647 case OMPD_for_simd: 9648 case OMPD_parallel_for_simd: 9649 case OMPD_cancel: 9650 case OMPD_cancellation_point: 9651 case OMPD_ordered: 9652 case OMPD_threadprivate: 9653 case OMPD_allocate: 9654 case OMPD_task: 9655 case OMPD_simd: 9656 case OMPD_sections: 9657 case OMPD_section: 9658 case OMPD_single: 9659 case OMPD_master: 9660 case OMPD_critical: 9661 case OMPD_taskyield: 9662 case OMPD_barrier: 9663 case OMPD_taskwait: 9664 case OMPD_taskgroup: 9665 case OMPD_atomic: 9666 case OMPD_flush: 9667 case OMPD_teams: 9668 case OMPD_target_data: 9669 case OMPD_target_exit_data: 9670 case OMPD_target_enter_data: 9671 case OMPD_distribute: 9672 case OMPD_distribute_simd: 9673 case OMPD_distribute_parallel_for: 9674 case OMPD_distribute_parallel_for_simd: 9675 case OMPD_teams_distribute: 9676 case OMPD_teams_distribute_simd: 9677 case OMPD_teams_distribute_parallel_for: 9678 case OMPD_teams_distribute_parallel_for_simd: 9679 case OMPD_target_update: 9680 case OMPD_declare_simd: 9681 case OMPD_declare_variant: 9682 case OMPD_declare_target: 9683 case OMPD_end_declare_target: 9684 case OMPD_declare_reduction: 9685 case OMPD_declare_mapper: 9686 case OMPD_taskloop: 9687 case OMPD_taskloop_simd: 9688 case OMPD_master_taskloop: 9689 case OMPD_requires: 9690 case OMPD_unknown: 9691 llvm_unreachable("Unknown target directive for OpenMP device codegen."); 9692 } 9693 return; 9694 } 9695 9696 if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) { 9697 if (!E->hasAssociatedStmt() || !E->getAssociatedStmt()) 9698 return; 9699 9700 scanForTargetRegionsFunctions( 9701 E->getInnermostCapturedStmt()->getCapturedStmt(), ParentName); 9702 return; 9703 } 9704 9705 // If this is a lambda function, look into its body. 9706 if (const auto *L = dyn_cast<LambdaExpr>(S)) 9707 S = L->getBody(); 9708 9709 // Keep looking for target regions recursively. 9710 for (const Stmt *II : S->children()) 9711 scanForTargetRegionsFunctions(II, ParentName); 9712 } 9713 9714 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) { 9715 // If emitting code for the host, we do not process FD here. Instead we do 9716 // the normal code generation. 9717 if (!CGM.getLangOpts().OpenMPIsDevice) { 9718 if (const auto *FD = dyn_cast<FunctionDecl>(GD.getDecl())) { 9719 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 9720 OMPDeclareTargetDeclAttr::getDeviceType(FD); 9721 // Do not emit device_type(nohost) functions for the host. 9722 if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_NoHost) 9723 return true; 9724 } 9725 return false; 9726 } 9727 9728 const ValueDecl *VD = cast<ValueDecl>(GD.getDecl()); 9729 StringRef Name = CGM.getMangledName(GD); 9730 // Try to detect target regions in the function. 9731 if (const auto *FD = dyn_cast<FunctionDecl>(VD)) { 9732 scanForTargetRegionsFunctions(FD->getBody(), Name); 9733 Optional<OMPDeclareTargetDeclAttr::DevTypeTy> DevTy = 9734 OMPDeclareTargetDeclAttr::getDeviceType(FD); 9735 // Do not emit device_type(nohost) functions for the host. 9736 if (DevTy && *DevTy == OMPDeclareTargetDeclAttr::DT_Host) 9737 return true; 9738 } 9739 9740 // Do not to emit function if it is not marked as declare target. 9741 return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) && 9742 AlreadyEmittedTargetFunctions.count(Name) == 0; 9743 } 9744 9745 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 9746 if (!CGM.getLangOpts().OpenMPIsDevice) 9747 return false; 9748 9749 // Check if there are Ctors/Dtors in this declaration and look for target 9750 // regions in it. We use the complete variant to produce the kernel name 9751 // mangling. 9752 QualType RDTy = cast<VarDecl>(GD.getDecl())->getType(); 9753 if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) { 9754 for (const CXXConstructorDecl *Ctor : RD->ctors()) { 9755 StringRef ParentName = 9756 CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete)); 9757 scanForTargetRegionsFunctions(Ctor->getBody(), ParentName); 9758 } 9759 if (const CXXDestructorDecl *Dtor = RD->getDestructor()) { 9760 StringRef ParentName = 9761 CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete)); 9762 scanForTargetRegionsFunctions(Dtor->getBody(), ParentName); 9763 } 9764 } 9765 9766 // Do not to emit variable if it is not marked as declare target. 9767 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 9768 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration( 9769 cast<VarDecl>(GD.getDecl())); 9770 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link || 9771 (*Res == OMPDeclareTargetDeclAttr::MT_To && 9772 HasRequiresUnifiedSharedMemory)) { 9773 DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl())); 9774 return true; 9775 } 9776 return false; 9777 } 9778 9779 llvm::Constant * 9780 CGOpenMPRuntime::registerTargetFirstprivateCopy(CodeGenFunction &CGF, 9781 const VarDecl *VD) { 9782 assert(VD->getType().isConstant(CGM.getContext()) && 9783 "Expected constant variable."); 9784 StringRef VarName; 9785 llvm::Constant *Addr; 9786 llvm::GlobalValue::LinkageTypes Linkage; 9787 QualType Ty = VD->getType(); 9788 SmallString<128> Buffer; 9789 { 9790 unsigned DeviceID; 9791 unsigned FileID; 9792 unsigned Line; 9793 getTargetEntryUniqueInfo(CGM.getContext(), VD->getLocation(), DeviceID, 9794 FileID, Line); 9795 llvm::raw_svector_ostream OS(Buffer); 9796 OS << "__omp_offloading_firstprivate_" << llvm::format("_%x", DeviceID) 9797 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 9798 VarName = OS.str(); 9799 } 9800 Linkage = llvm::GlobalValue::InternalLinkage; 9801 Addr = 9802 getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(Ty), VarName, 9803 getDefaultFirstprivateAddressSpace()); 9804 cast<llvm::GlobalValue>(Addr)->setLinkage(Linkage); 9805 CharUnits VarSize = CGM.getContext().getTypeSizeInChars(Ty); 9806 CGM.addCompilerUsedGlobal(cast<llvm::GlobalValue>(Addr)); 9807 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 9808 VarName, Addr, VarSize, 9809 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo, Linkage); 9810 return Addr; 9811 } 9812 9813 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD, 9814 llvm::Constant *Addr) { 9815 if (CGM.getLangOpts().OMPTargetTriples.empty() && 9816 !CGM.getLangOpts().OpenMPIsDevice) 9817 return; 9818 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 9819 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 9820 if (!Res) { 9821 if (CGM.getLangOpts().OpenMPIsDevice) { 9822 // Register non-target variables being emitted in device code (debug info 9823 // may cause this). 9824 StringRef VarName = CGM.getMangledName(VD); 9825 EmittedNonTargetVariables.try_emplace(VarName, Addr); 9826 } 9827 return; 9828 } 9829 // Register declare target variables. 9830 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags; 9831 StringRef VarName; 9832 CharUnits VarSize; 9833 llvm::GlobalValue::LinkageTypes Linkage; 9834 9835 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 9836 !HasRequiresUnifiedSharedMemory) { 9837 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 9838 VarName = CGM.getMangledName(VD); 9839 if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) { 9840 VarSize = CGM.getContext().getTypeSizeInChars(VD->getType()); 9841 assert(!VarSize.isZero() && "Expected non-zero size of the variable"); 9842 } else { 9843 VarSize = CharUnits::Zero(); 9844 } 9845 Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false); 9846 // Temp solution to prevent optimizations of the internal variables. 9847 if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) { 9848 std::string RefName = getName({VarName, "ref"}); 9849 if (!CGM.GetGlobalValue(RefName)) { 9850 llvm::Constant *AddrRef = 9851 getOrCreateInternalVariable(Addr->getType(), RefName); 9852 auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef); 9853 GVAddrRef->setConstant(/*Val=*/true); 9854 GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage); 9855 GVAddrRef->setInitializer(Addr); 9856 CGM.addCompilerUsedGlobal(GVAddrRef); 9857 } 9858 } 9859 } else { 9860 assert(((*Res == OMPDeclareTargetDeclAttr::MT_Link) || 9861 (*Res == OMPDeclareTargetDeclAttr::MT_To && 9862 HasRequiresUnifiedSharedMemory)) && 9863 "Declare target attribute must link or to with unified memory."); 9864 if (*Res == OMPDeclareTargetDeclAttr::MT_Link) 9865 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink; 9866 else 9867 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 9868 9869 if (CGM.getLangOpts().OpenMPIsDevice) { 9870 VarName = Addr->getName(); 9871 Addr = nullptr; 9872 } else { 9873 VarName = getAddrOfDeclareTargetVar(VD).getName(); 9874 Addr = cast<llvm::Constant>(getAddrOfDeclareTargetVar(VD).getPointer()); 9875 } 9876 VarSize = CGM.getPointerSize(); 9877 Linkage = llvm::GlobalValue::WeakAnyLinkage; 9878 } 9879 9880 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 9881 VarName, Addr, VarSize, Flags, Linkage); 9882 } 9883 9884 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) { 9885 if (isa<FunctionDecl>(GD.getDecl()) || 9886 isa<OMPDeclareReductionDecl>(GD.getDecl())) 9887 return emitTargetFunctions(GD); 9888 9889 return emitTargetGlobalVariable(GD); 9890 } 9891 9892 void CGOpenMPRuntime::emitDeferredTargetDecls() const { 9893 for (const VarDecl *VD : DeferredGlobalVariables) { 9894 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 9895 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 9896 if (!Res) 9897 continue; 9898 if (*Res == OMPDeclareTargetDeclAttr::MT_To && 9899 !HasRequiresUnifiedSharedMemory) { 9900 CGM.EmitGlobal(VD); 9901 } else { 9902 assert((*Res == OMPDeclareTargetDeclAttr::MT_Link || 9903 (*Res == OMPDeclareTargetDeclAttr::MT_To && 9904 HasRequiresUnifiedSharedMemory)) && 9905 "Expected link clause or to clause with unified memory."); 9906 (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetVar(VD); 9907 } 9908 } 9909 } 9910 9911 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas( 9912 CodeGenFunction &CGF, const OMPExecutableDirective &D) const { 9913 assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) && 9914 " Expected target-based directive."); 9915 } 9916 9917 void CGOpenMPRuntime::checkArchForUnifiedAddressing( 9918 const OMPRequiresDecl *D) { 9919 for (const OMPClause *Clause : D->clauselists()) { 9920 if (Clause->getClauseKind() == OMPC_unified_shared_memory) { 9921 HasRequiresUnifiedSharedMemory = true; 9922 break; 9923 } 9924 } 9925 } 9926 9927 bool CGOpenMPRuntime::hasAllocateAttributeForGlobalVar(const VarDecl *VD, 9928 LangAS &AS) { 9929 if (!VD || !VD->hasAttr<OMPAllocateDeclAttr>()) 9930 return false; 9931 const auto *A = VD->getAttr<OMPAllocateDeclAttr>(); 9932 switch(A->getAllocatorType()) { 9933 case OMPAllocateDeclAttr::OMPDefaultMemAlloc: 9934 // Not supported, fallback to the default mem space. 9935 case OMPAllocateDeclAttr::OMPLargeCapMemAlloc: 9936 case OMPAllocateDeclAttr::OMPCGroupMemAlloc: 9937 case OMPAllocateDeclAttr::OMPHighBWMemAlloc: 9938 case OMPAllocateDeclAttr::OMPLowLatMemAlloc: 9939 case OMPAllocateDeclAttr::OMPThreadMemAlloc: 9940 case OMPAllocateDeclAttr::OMPConstMemAlloc: 9941 case OMPAllocateDeclAttr::OMPPTeamMemAlloc: 9942 AS = LangAS::Default; 9943 return true; 9944 case OMPAllocateDeclAttr::OMPUserDefinedMemAlloc: 9945 llvm_unreachable("Expected predefined allocator for the variables with the " 9946 "static storage."); 9947 } 9948 return false; 9949 } 9950 9951 bool CGOpenMPRuntime::hasRequiresUnifiedSharedMemory() const { 9952 return HasRequiresUnifiedSharedMemory; 9953 } 9954 9955 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII( 9956 CodeGenModule &CGM) 9957 : CGM(CGM) { 9958 if (CGM.getLangOpts().OpenMPIsDevice) { 9959 SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal; 9960 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false; 9961 } 9962 } 9963 9964 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() { 9965 if (CGM.getLangOpts().OpenMPIsDevice) 9966 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal; 9967 } 9968 9969 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) { 9970 if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal) 9971 return true; 9972 9973 StringRef Name = CGM.getMangledName(GD); 9974 const auto *D = cast<FunctionDecl>(GD.getDecl()); 9975 // Do not to emit function if it is marked as declare target as it was already 9976 // emitted. 9977 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) { 9978 if (D->hasBody() && AlreadyEmittedTargetFunctions.count(Name) == 0) { 9979 if (auto *F = dyn_cast_or_null<llvm::Function>(CGM.GetGlobalValue(Name))) 9980 return !F->isDeclaration(); 9981 return false; 9982 } 9983 return true; 9984 } 9985 9986 return !AlreadyEmittedTargetFunctions.insert(Name).second; 9987 } 9988 9989 llvm::Function *CGOpenMPRuntime::emitRequiresDirectiveRegFun() { 9990 // If we don't have entries or if we are emitting code for the device, we 9991 // don't need to do anything. 9992 if (CGM.getLangOpts().OMPTargetTriples.empty() || 9993 CGM.getLangOpts().OpenMPSimd || CGM.getLangOpts().OpenMPIsDevice || 9994 (OffloadEntriesInfoManager.empty() && 9995 !HasEmittedDeclareTargetRegion && 9996 !HasEmittedTargetRegion)) 9997 return nullptr; 9998 9999 // Create and register the function that handles the requires directives. 10000 ASTContext &C = CGM.getContext(); 10001 10002 llvm::Function *RequiresRegFn; 10003 { 10004 CodeGenFunction CGF(CGM); 10005 const auto &FI = CGM.getTypes().arrangeNullaryFunction(); 10006 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 10007 std::string ReqName = getName({"omp_offloading", "requires_reg"}); 10008 RequiresRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, ReqName, FI); 10009 CGF.StartFunction(GlobalDecl(), C.VoidTy, RequiresRegFn, FI, {}); 10010 OpenMPOffloadingRequiresDirFlags Flags = OMP_REQ_NONE; 10011 // TODO: check for other requires clauses. 10012 // The requires directive takes effect only when a target region is 10013 // present in the compilation unit. Otherwise it is ignored and not 10014 // passed to the runtime. This avoids the runtime from throwing an error 10015 // for mismatching requires clauses across compilation units that don't 10016 // contain at least 1 target region. 10017 assert((HasEmittedTargetRegion || 10018 HasEmittedDeclareTargetRegion || 10019 !OffloadEntriesInfoManager.empty()) && 10020 "Target or declare target region expected."); 10021 if (HasRequiresUnifiedSharedMemory) 10022 Flags = OMP_REQ_UNIFIED_SHARED_MEMORY; 10023 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_requires), 10024 llvm::ConstantInt::get(CGM.Int64Ty, Flags)); 10025 CGF.FinishFunction(); 10026 } 10027 return RequiresRegFn; 10028 } 10029 10030 llvm::Function *CGOpenMPRuntime::emitRegistrationFunction() { 10031 // If we have offloading in the current module, we need to emit the entries 10032 // now and register the offloading descriptor. 10033 createOffloadEntriesAndInfoMetadata(); 10034 10035 // Create and register the offloading binary descriptors. This is the main 10036 // entity that captures all the information about offloading in the current 10037 // compilation unit. 10038 return createOffloadingBinaryDescriptorRegistration(); 10039 } 10040 10041 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF, 10042 const OMPExecutableDirective &D, 10043 SourceLocation Loc, 10044 llvm::Function *OutlinedFn, 10045 ArrayRef<llvm::Value *> CapturedVars) { 10046 if (!CGF.HaveInsertPoint()) 10047 return; 10048 10049 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 10050 CodeGenFunction::RunCleanupsScope Scope(CGF); 10051 10052 // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn); 10053 llvm::Value *Args[] = { 10054 RTLoc, 10055 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 10056 CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())}; 10057 llvm::SmallVector<llvm::Value *, 16> RealArgs; 10058 RealArgs.append(std::begin(Args), std::end(Args)); 10059 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 10060 10061 llvm::FunctionCallee RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_teams); 10062 CGF.EmitRuntimeCall(RTLFn, RealArgs); 10063 } 10064 10065 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 10066 const Expr *NumTeams, 10067 const Expr *ThreadLimit, 10068 SourceLocation Loc) { 10069 if (!CGF.HaveInsertPoint()) 10070 return; 10071 10072 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 10073 10074 llvm::Value *NumTeamsVal = 10075 NumTeams 10076 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams), 10077 CGF.CGM.Int32Ty, /* isSigned = */ true) 10078 : CGF.Builder.getInt32(0); 10079 10080 llvm::Value *ThreadLimitVal = 10081 ThreadLimit 10082 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit), 10083 CGF.CGM.Int32Ty, /* isSigned = */ true) 10084 : CGF.Builder.getInt32(0); 10085 10086 // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit) 10087 llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal, 10088 ThreadLimitVal}; 10089 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_teams), 10090 PushNumTeamsArgs); 10091 } 10092 10093 void CGOpenMPRuntime::emitTargetDataCalls( 10094 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 10095 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 10096 if (!CGF.HaveInsertPoint()) 10097 return; 10098 10099 // Action used to replace the default codegen action and turn privatization 10100 // off. 10101 PrePostActionTy NoPrivAction; 10102 10103 // Generate the code for the opening of the data environment. Capture all the 10104 // arguments of the runtime call by reference because they are used in the 10105 // closing of the region. 10106 auto &&BeginThenGen = [this, &D, Device, &Info, 10107 &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) { 10108 // Fill up the arrays with all the mapped variables. 10109 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 10110 MappableExprsHandler::MapValuesArrayTy Pointers; 10111 MappableExprsHandler::MapValuesArrayTy Sizes; 10112 MappableExprsHandler::MapFlagsArrayTy MapTypes; 10113 10114 // Get map clause information. 10115 MappableExprsHandler MCHandler(D, CGF); 10116 MCHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes); 10117 10118 // Fill up the arrays and create the arguments. 10119 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 10120 10121 llvm::Value *BasePointersArrayArg = nullptr; 10122 llvm::Value *PointersArrayArg = nullptr; 10123 llvm::Value *SizesArrayArg = nullptr; 10124 llvm::Value *MapTypesArrayArg = nullptr; 10125 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 10126 SizesArrayArg, MapTypesArrayArg, Info); 10127 10128 // Emit device ID if any. 10129 llvm::Value *DeviceID = nullptr; 10130 if (Device) { 10131 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10132 CGF.Int64Ty, /*isSigned=*/true); 10133 } else { 10134 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10135 } 10136 10137 // Emit the number of elements in the offloading arrays. 10138 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 10139 10140 llvm::Value *OffloadingArgs[] = { 10141 DeviceID, PointerNum, BasePointersArrayArg, 10142 PointersArrayArg, SizesArrayArg, MapTypesArrayArg}; 10143 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_begin), 10144 OffloadingArgs); 10145 10146 // If device pointer privatization is required, emit the body of the region 10147 // here. It will have to be duplicated: with and without privatization. 10148 if (!Info.CaptureDeviceAddrMap.empty()) 10149 CodeGen(CGF); 10150 }; 10151 10152 // Generate code for the closing of the data region. 10153 auto &&EndThenGen = [this, Device, &Info](CodeGenFunction &CGF, 10154 PrePostActionTy &) { 10155 assert(Info.isValid() && "Invalid data environment closing arguments."); 10156 10157 llvm::Value *BasePointersArrayArg = nullptr; 10158 llvm::Value *PointersArrayArg = nullptr; 10159 llvm::Value *SizesArrayArg = nullptr; 10160 llvm::Value *MapTypesArrayArg = nullptr; 10161 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 10162 SizesArrayArg, MapTypesArrayArg, Info); 10163 10164 // Emit device ID if any. 10165 llvm::Value *DeviceID = nullptr; 10166 if (Device) { 10167 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10168 CGF.Int64Ty, /*isSigned=*/true); 10169 } else { 10170 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10171 } 10172 10173 // Emit the number of elements in the offloading arrays. 10174 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 10175 10176 llvm::Value *OffloadingArgs[] = { 10177 DeviceID, PointerNum, BasePointersArrayArg, 10178 PointersArrayArg, SizesArrayArg, MapTypesArrayArg}; 10179 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_end), 10180 OffloadingArgs); 10181 }; 10182 10183 // If we need device pointer privatization, we need to emit the body of the 10184 // region with no privatization in the 'else' branch of the conditional. 10185 // Otherwise, we don't have to do anything. 10186 auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF, 10187 PrePostActionTy &) { 10188 if (!Info.CaptureDeviceAddrMap.empty()) { 10189 CodeGen.setAction(NoPrivAction); 10190 CodeGen(CGF); 10191 } 10192 }; 10193 10194 // We don't have to do anything to close the region if the if clause evaluates 10195 // to false. 10196 auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {}; 10197 10198 if (IfCond) { 10199 emitOMPIfClause(CGF, IfCond, BeginThenGen, BeginElseGen); 10200 } else { 10201 RegionCodeGenTy RCG(BeginThenGen); 10202 RCG(CGF); 10203 } 10204 10205 // If we don't require privatization of device pointers, we emit the body in 10206 // between the runtime calls. This avoids duplicating the body code. 10207 if (Info.CaptureDeviceAddrMap.empty()) { 10208 CodeGen.setAction(NoPrivAction); 10209 CodeGen(CGF); 10210 } 10211 10212 if (IfCond) { 10213 emitOMPIfClause(CGF, IfCond, EndThenGen, EndElseGen); 10214 } else { 10215 RegionCodeGenTy RCG(EndThenGen); 10216 RCG(CGF); 10217 } 10218 } 10219 10220 void CGOpenMPRuntime::emitTargetDataStandAloneCall( 10221 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 10222 const Expr *Device) { 10223 if (!CGF.HaveInsertPoint()) 10224 return; 10225 10226 assert((isa<OMPTargetEnterDataDirective>(D) || 10227 isa<OMPTargetExitDataDirective>(D) || 10228 isa<OMPTargetUpdateDirective>(D)) && 10229 "Expecting either target enter, exit data, or update directives."); 10230 10231 CodeGenFunction::OMPTargetDataInfo InputInfo; 10232 llvm::Value *MapTypesArray = nullptr; 10233 // Generate the code for the opening of the data environment. 10234 auto &&ThenGen = [this, &D, Device, &InputInfo, 10235 &MapTypesArray](CodeGenFunction &CGF, PrePostActionTy &) { 10236 // Emit device ID if any. 10237 llvm::Value *DeviceID = nullptr; 10238 if (Device) { 10239 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 10240 CGF.Int64Ty, /*isSigned=*/true); 10241 } else { 10242 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 10243 } 10244 10245 // Emit the number of elements in the offloading arrays. 10246 llvm::Constant *PointerNum = 10247 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 10248 10249 llvm::Value *OffloadingArgs[] = {DeviceID, 10250 PointerNum, 10251 InputInfo.BasePointersArray.getPointer(), 10252 InputInfo.PointersArray.getPointer(), 10253 InputInfo.SizesArray.getPointer(), 10254 MapTypesArray}; 10255 10256 // Select the right runtime function call for each expected standalone 10257 // directive. 10258 const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 10259 OpenMPRTLFunction RTLFn; 10260 switch (D.getDirectiveKind()) { 10261 case OMPD_target_enter_data: 10262 RTLFn = HasNowait ? OMPRTL__tgt_target_data_begin_nowait 10263 : OMPRTL__tgt_target_data_begin; 10264 break; 10265 case OMPD_target_exit_data: 10266 RTLFn = HasNowait ? OMPRTL__tgt_target_data_end_nowait 10267 : OMPRTL__tgt_target_data_end; 10268 break; 10269 case OMPD_target_update: 10270 RTLFn = HasNowait ? OMPRTL__tgt_target_data_update_nowait 10271 : OMPRTL__tgt_target_data_update; 10272 break; 10273 case OMPD_parallel: 10274 case OMPD_for: 10275 case OMPD_parallel_for: 10276 case OMPD_parallel_sections: 10277 case OMPD_for_simd: 10278 case OMPD_parallel_for_simd: 10279 case OMPD_cancel: 10280 case OMPD_cancellation_point: 10281 case OMPD_ordered: 10282 case OMPD_threadprivate: 10283 case OMPD_allocate: 10284 case OMPD_task: 10285 case OMPD_simd: 10286 case OMPD_sections: 10287 case OMPD_section: 10288 case OMPD_single: 10289 case OMPD_master: 10290 case OMPD_critical: 10291 case OMPD_taskyield: 10292 case OMPD_barrier: 10293 case OMPD_taskwait: 10294 case OMPD_taskgroup: 10295 case OMPD_atomic: 10296 case OMPD_flush: 10297 case OMPD_teams: 10298 case OMPD_target_data: 10299 case OMPD_distribute: 10300 case OMPD_distribute_simd: 10301 case OMPD_distribute_parallel_for: 10302 case OMPD_distribute_parallel_for_simd: 10303 case OMPD_teams_distribute: 10304 case OMPD_teams_distribute_simd: 10305 case OMPD_teams_distribute_parallel_for: 10306 case OMPD_teams_distribute_parallel_for_simd: 10307 case OMPD_declare_simd: 10308 case OMPD_declare_variant: 10309 case OMPD_declare_target: 10310 case OMPD_end_declare_target: 10311 case OMPD_declare_reduction: 10312 case OMPD_declare_mapper: 10313 case OMPD_taskloop: 10314 case OMPD_taskloop_simd: 10315 case OMPD_master_taskloop: 10316 case OMPD_target: 10317 case OMPD_target_simd: 10318 case OMPD_target_teams_distribute: 10319 case OMPD_target_teams_distribute_simd: 10320 case OMPD_target_teams_distribute_parallel_for: 10321 case OMPD_target_teams_distribute_parallel_for_simd: 10322 case OMPD_target_teams: 10323 case OMPD_target_parallel: 10324 case OMPD_target_parallel_for: 10325 case OMPD_target_parallel_for_simd: 10326 case OMPD_requires: 10327 case OMPD_unknown: 10328 llvm_unreachable("Unexpected standalone target data directive."); 10329 break; 10330 } 10331 CGF.EmitRuntimeCall(createRuntimeFunction(RTLFn), OffloadingArgs); 10332 }; 10333 10334 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray]( 10335 CodeGenFunction &CGF, PrePostActionTy &) { 10336 // Fill up the arrays with all the mapped variables. 10337 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 10338 MappableExprsHandler::MapValuesArrayTy Pointers; 10339 MappableExprsHandler::MapValuesArrayTy Sizes; 10340 MappableExprsHandler::MapFlagsArrayTy MapTypes; 10341 10342 // Get map clause information. 10343 MappableExprsHandler MEHandler(D, CGF); 10344 MEHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes); 10345 10346 TargetDataInfo Info; 10347 // Fill up the arrays and create the arguments. 10348 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 10349 emitOffloadingArraysArgument(CGF, Info.BasePointersArray, 10350 Info.PointersArray, Info.SizesArray, 10351 Info.MapTypesArray, Info); 10352 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 10353 InputInfo.BasePointersArray = 10354 Address(Info.BasePointersArray, CGM.getPointerAlign()); 10355 InputInfo.PointersArray = 10356 Address(Info.PointersArray, CGM.getPointerAlign()); 10357 InputInfo.SizesArray = 10358 Address(Info.SizesArray, CGM.getPointerAlign()); 10359 MapTypesArray = Info.MapTypesArray; 10360 if (D.hasClausesOfKind<OMPDependClause>()) 10361 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 10362 else 10363 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 10364 }; 10365 10366 if (IfCond) { 10367 emitOMPIfClause(CGF, IfCond, TargetThenGen, 10368 [](CodeGenFunction &CGF, PrePostActionTy &) {}); 10369 } else { 10370 RegionCodeGenTy ThenRCG(TargetThenGen); 10371 ThenRCG(CGF); 10372 } 10373 } 10374 10375 namespace { 10376 /// Kind of parameter in a function with 'declare simd' directive. 10377 enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector }; 10378 /// Attribute set of the parameter. 10379 struct ParamAttrTy { 10380 ParamKindTy Kind = Vector; 10381 llvm::APSInt StrideOrArg; 10382 llvm::APSInt Alignment; 10383 }; 10384 } // namespace 10385 10386 static unsigned evaluateCDTSize(const FunctionDecl *FD, 10387 ArrayRef<ParamAttrTy> ParamAttrs) { 10388 // Every vector variant of a SIMD-enabled function has a vector length (VLEN). 10389 // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument 10390 // of that clause. The VLEN value must be power of 2. 10391 // In other case the notion of the function`s "characteristic data type" (CDT) 10392 // is used to compute the vector length. 10393 // CDT is defined in the following order: 10394 // a) For non-void function, the CDT is the return type. 10395 // b) If the function has any non-uniform, non-linear parameters, then the 10396 // CDT is the type of the first such parameter. 10397 // c) If the CDT determined by a) or b) above is struct, union, or class 10398 // type which is pass-by-value (except for the type that maps to the 10399 // built-in complex data type), the characteristic data type is int. 10400 // d) If none of the above three cases is applicable, the CDT is int. 10401 // The VLEN is then determined based on the CDT and the size of vector 10402 // register of that ISA for which current vector version is generated. The 10403 // VLEN is computed using the formula below: 10404 // VLEN = sizeof(vector_register) / sizeof(CDT), 10405 // where vector register size specified in section 3.2.1 Registers and the 10406 // Stack Frame of original AMD64 ABI document. 10407 QualType RetType = FD->getReturnType(); 10408 if (RetType.isNull()) 10409 return 0; 10410 ASTContext &C = FD->getASTContext(); 10411 QualType CDT; 10412 if (!RetType.isNull() && !RetType->isVoidType()) { 10413 CDT = RetType; 10414 } else { 10415 unsigned Offset = 0; 10416 if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) { 10417 if (ParamAttrs[Offset].Kind == Vector) 10418 CDT = C.getPointerType(C.getRecordType(MD->getParent())); 10419 ++Offset; 10420 } 10421 if (CDT.isNull()) { 10422 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 10423 if (ParamAttrs[I + Offset].Kind == Vector) { 10424 CDT = FD->getParamDecl(I)->getType(); 10425 break; 10426 } 10427 } 10428 } 10429 } 10430 if (CDT.isNull()) 10431 CDT = C.IntTy; 10432 CDT = CDT->getCanonicalTypeUnqualified(); 10433 if (CDT->isRecordType() || CDT->isUnionType()) 10434 CDT = C.IntTy; 10435 return C.getTypeSize(CDT); 10436 } 10437 10438 static void 10439 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn, 10440 const llvm::APSInt &VLENVal, 10441 ArrayRef<ParamAttrTy> ParamAttrs, 10442 OMPDeclareSimdDeclAttr::BranchStateTy State) { 10443 struct ISADataTy { 10444 char ISA; 10445 unsigned VecRegSize; 10446 }; 10447 ISADataTy ISAData[] = { 10448 { 10449 'b', 128 10450 }, // SSE 10451 { 10452 'c', 256 10453 }, // AVX 10454 { 10455 'd', 256 10456 }, // AVX2 10457 { 10458 'e', 512 10459 }, // AVX512 10460 }; 10461 llvm::SmallVector<char, 2> Masked; 10462 switch (State) { 10463 case OMPDeclareSimdDeclAttr::BS_Undefined: 10464 Masked.push_back('N'); 10465 Masked.push_back('M'); 10466 break; 10467 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10468 Masked.push_back('N'); 10469 break; 10470 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10471 Masked.push_back('M'); 10472 break; 10473 } 10474 for (char Mask : Masked) { 10475 for (const ISADataTy &Data : ISAData) { 10476 SmallString<256> Buffer; 10477 llvm::raw_svector_ostream Out(Buffer); 10478 Out << "_ZGV" << Data.ISA << Mask; 10479 if (!VLENVal) { 10480 unsigned NumElts = evaluateCDTSize(FD, ParamAttrs); 10481 assert(NumElts && "Non-zero simdlen/cdtsize expected"); 10482 Out << llvm::APSInt::getUnsigned(Data.VecRegSize / NumElts); 10483 } else { 10484 Out << VLENVal; 10485 } 10486 for (const ParamAttrTy &ParamAttr : ParamAttrs) { 10487 switch (ParamAttr.Kind){ 10488 case LinearWithVarStride: 10489 Out << 's' << ParamAttr.StrideOrArg; 10490 break; 10491 case Linear: 10492 Out << 'l'; 10493 if (!!ParamAttr.StrideOrArg) 10494 Out << ParamAttr.StrideOrArg; 10495 break; 10496 case Uniform: 10497 Out << 'u'; 10498 break; 10499 case Vector: 10500 Out << 'v'; 10501 break; 10502 } 10503 if (!!ParamAttr.Alignment) 10504 Out << 'a' << ParamAttr.Alignment; 10505 } 10506 Out << '_' << Fn->getName(); 10507 Fn->addFnAttr(Out.str()); 10508 } 10509 } 10510 } 10511 10512 // This are the Functions that are needed to mangle the name of the 10513 // vector functions generated by the compiler, according to the rules 10514 // defined in the "Vector Function ABI specifications for AArch64", 10515 // available at 10516 // https://developer.arm.com/products/software-development-tools/hpc/arm-compiler-for-hpc/vector-function-abi. 10517 10518 /// Maps To Vector (MTV), as defined in 3.1.1 of the AAVFABI. 10519 /// 10520 /// TODO: Need to implement the behavior for reference marked with a 10521 /// var or no linear modifiers (1.b in the section). For this, we 10522 /// need to extend ParamKindTy to support the linear modifiers. 10523 static bool getAArch64MTV(QualType QT, ParamKindTy Kind) { 10524 QT = QT.getCanonicalType(); 10525 10526 if (QT->isVoidType()) 10527 return false; 10528 10529 if (Kind == ParamKindTy::Uniform) 10530 return false; 10531 10532 if (Kind == ParamKindTy::Linear) 10533 return false; 10534 10535 // TODO: Handle linear references with modifiers 10536 10537 if (Kind == ParamKindTy::LinearWithVarStride) 10538 return false; 10539 10540 return true; 10541 } 10542 10543 /// Pass By Value (PBV), as defined in 3.1.2 of the AAVFABI. 10544 static bool getAArch64PBV(QualType QT, ASTContext &C) { 10545 QT = QT.getCanonicalType(); 10546 unsigned Size = C.getTypeSize(QT); 10547 10548 // Only scalars and complex within 16 bytes wide set PVB to true. 10549 if (Size != 8 && Size != 16 && Size != 32 && Size != 64 && Size != 128) 10550 return false; 10551 10552 if (QT->isFloatingType()) 10553 return true; 10554 10555 if (QT->isIntegerType()) 10556 return true; 10557 10558 if (QT->isPointerType()) 10559 return true; 10560 10561 // TODO: Add support for complex types (section 3.1.2, item 2). 10562 10563 return false; 10564 } 10565 10566 /// Computes the lane size (LS) of a return type or of an input parameter, 10567 /// as defined by `LS(P)` in 3.2.1 of the AAVFABI. 10568 /// TODO: Add support for references, section 3.2.1, item 1. 10569 static unsigned getAArch64LS(QualType QT, ParamKindTy Kind, ASTContext &C) { 10570 if (getAArch64MTV(QT, Kind) && QT.getCanonicalType()->isPointerType()) { 10571 QualType PTy = QT.getCanonicalType()->getPointeeType(); 10572 if (getAArch64PBV(PTy, C)) 10573 return C.getTypeSize(PTy); 10574 } 10575 if (getAArch64PBV(QT, C)) 10576 return C.getTypeSize(QT); 10577 10578 return C.getTypeSize(C.getUIntPtrType()); 10579 } 10580 10581 // Get Narrowest Data Size (NDS) and Widest Data Size (WDS) from the 10582 // signature of the scalar function, as defined in 3.2.2 of the 10583 // AAVFABI. 10584 static std::tuple<unsigned, unsigned, bool> 10585 getNDSWDS(const FunctionDecl *FD, ArrayRef<ParamAttrTy> ParamAttrs) { 10586 QualType RetType = FD->getReturnType().getCanonicalType(); 10587 10588 ASTContext &C = FD->getASTContext(); 10589 10590 bool OutputBecomesInput = false; 10591 10592 llvm::SmallVector<unsigned, 8> Sizes; 10593 if (!RetType->isVoidType()) { 10594 Sizes.push_back(getAArch64LS(RetType, ParamKindTy::Vector, C)); 10595 if (!getAArch64PBV(RetType, C) && getAArch64MTV(RetType, {})) 10596 OutputBecomesInput = true; 10597 } 10598 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 10599 QualType QT = FD->getParamDecl(I)->getType().getCanonicalType(); 10600 Sizes.push_back(getAArch64LS(QT, ParamAttrs[I].Kind, C)); 10601 } 10602 10603 assert(!Sizes.empty() && "Unable to determine NDS and WDS."); 10604 // The LS of a function parameter / return value can only be a power 10605 // of 2, starting from 8 bits, up to 128. 10606 assert(std::all_of(Sizes.begin(), Sizes.end(), 10607 [](unsigned Size) { 10608 return Size == 8 || Size == 16 || Size == 32 || 10609 Size == 64 || Size == 128; 10610 }) && 10611 "Invalid size"); 10612 10613 return std::make_tuple(*std::min_element(std::begin(Sizes), std::end(Sizes)), 10614 *std::max_element(std::begin(Sizes), std::end(Sizes)), 10615 OutputBecomesInput); 10616 } 10617 10618 /// Mangle the parameter part of the vector function name according to 10619 /// their OpenMP classification. The mangling function is defined in 10620 /// section 3.5 of the AAVFABI. 10621 static std::string mangleVectorParameters(ArrayRef<ParamAttrTy> ParamAttrs) { 10622 SmallString<256> Buffer; 10623 llvm::raw_svector_ostream Out(Buffer); 10624 for (const auto &ParamAttr : ParamAttrs) { 10625 switch (ParamAttr.Kind) { 10626 case LinearWithVarStride: 10627 Out << "ls" << ParamAttr.StrideOrArg; 10628 break; 10629 case Linear: 10630 Out << 'l'; 10631 // Don't print the step value if it is not present or if it is 10632 // equal to 1. 10633 if (!!ParamAttr.StrideOrArg && ParamAttr.StrideOrArg != 1) 10634 Out << ParamAttr.StrideOrArg; 10635 break; 10636 case Uniform: 10637 Out << 'u'; 10638 break; 10639 case Vector: 10640 Out << 'v'; 10641 break; 10642 } 10643 10644 if (!!ParamAttr.Alignment) 10645 Out << 'a' << ParamAttr.Alignment; 10646 } 10647 10648 return Out.str(); 10649 } 10650 10651 // Function used to add the attribute. The parameter `VLEN` is 10652 // templated to allow the use of "x" when targeting scalable functions 10653 // for SVE. 10654 template <typename T> 10655 static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix, 10656 char ISA, StringRef ParSeq, 10657 StringRef MangledName, bool OutputBecomesInput, 10658 llvm::Function *Fn) { 10659 SmallString<256> Buffer; 10660 llvm::raw_svector_ostream Out(Buffer); 10661 Out << Prefix << ISA << LMask << VLEN; 10662 if (OutputBecomesInput) 10663 Out << "v"; 10664 Out << ParSeq << "_" << MangledName; 10665 Fn->addFnAttr(Out.str()); 10666 } 10667 10668 // Helper function to generate the Advanced SIMD names depending on 10669 // the value of the NDS when simdlen is not present. 10670 static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask, 10671 StringRef Prefix, char ISA, 10672 StringRef ParSeq, StringRef MangledName, 10673 bool OutputBecomesInput, 10674 llvm::Function *Fn) { 10675 switch (NDS) { 10676 case 8: 10677 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 10678 OutputBecomesInput, Fn); 10679 addAArch64VectorName(16, Mask, Prefix, ISA, ParSeq, MangledName, 10680 OutputBecomesInput, Fn); 10681 break; 10682 case 16: 10683 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 10684 OutputBecomesInput, Fn); 10685 addAArch64VectorName(8, Mask, Prefix, ISA, ParSeq, MangledName, 10686 OutputBecomesInput, Fn); 10687 break; 10688 case 32: 10689 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 10690 OutputBecomesInput, Fn); 10691 addAArch64VectorName(4, Mask, Prefix, ISA, ParSeq, MangledName, 10692 OutputBecomesInput, Fn); 10693 break; 10694 case 64: 10695 case 128: 10696 addAArch64VectorName(2, Mask, Prefix, ISA, ParSeq, MangledName, 10697 OutputBecomesInput, Fn); 10698 break; 10699 default: 10700 llvm_unreachable("Scalar type is too wide."); 10701 } 10702 } 10703 10704 /// Emit vector function attributes for AArch64, as defined in the AAVFABI. 10705 static void emitAArch64DeclareSimdFunction( 10706 CodeGenModule &CGM, const FunctionDecl *FD, unsigned UserVLEN, 10707 ArrayRef<ParamAttrTy> ParamAttrs, 10708 OMPDeclareSimdDeclAttr::BranchStateTy State, StringRef MangledName, 10709 char ISA, unsigned VecRegSize, llvm::Function *Fn, SourceLocation SLoc) { 10710 10711 // Get basic data for building the vector signature. 10712 const auto Data = getNDSWDS(FD, ParamAttrs); 10713 const unsigned NDS = std::get<0>(Data); 10714 const unsigned WDS = std::get<1>(Data); 10715 const bool OutputBecomesInput = std::get<2>(Data); 10716 10717 // Check the values provided via `simdlen` by the user. 10718 // 1. A `simdlen(1)` doesn't produce vector signatures, 10719 if (UserVLEN == 1) { 10720 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10721 DiagnosticsEngine::Warning, 10722 "The clause simdlen(1) has no effect when targeting aarch64."); 10723 CGM.getDiags().Report(SLoc, DiagID); 10724 return; 10725 } 10726 10727 // 2. Section 3.3.1, item 1: user input must be a power of 2 for 10728 // Advanced SIMD output. 10729 if (ISA == 'n' && UserVLEN && !llvm::isPowerOf2_32(UserVLEN)) { 10730 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10731 DiagnosticsEngine::Warning, "The value specified in simdlen must be a " 10732 "power of 2 when targeting Advanced SIMD."); 10733 CGM.getDiags().Report(SLoc, DiagID); 10734 return; 10735 } 10736 10737 // 3. Section 3.4.1. SVE fixed lengh must obey the architectural 10738 // limits. 10739 if (ISA == 's' && UserVLEN != 0) { 10740 if ((UserVLEN * WDS > 2048) || (UserVLEN * WDS % 128 != 0)) { 10741 unsigned DiagID = CGM.getDiags().getCustomDiagID( 10742 DiagnosticsEngine::Warning, "The clause simdlen must fit the %0-bit " 10743 "lanes in the architectural constraints " 10744 "for SVE (min is 128-bit, max is " 10745 "2048-bit, by steps of 128-bit)"); 10746 CGM.getDiags().Report(SLoc, DiagID) << WDS; 10747 return; 10748 } 10749 } 10750 10751 // Sort out parameter sequence. 10752 const std::string ParSeq = mangleVectorParameters(ParamAttrs); 10753 StringRef Prefix = "_ZGV"; 10754 // Generate simdlen from user input (if any). 10755 if (UserVLEN) { 10756 if (ISA == 's') { 10757 // SVE generates only a masked function. 10758 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10759 OutputBecomesInput, Fn); 10760 } else { 10761 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 10762 // Advanced SIMD generates one or two functions, depending on 10763 // the `[not]inbranch` clause. 10764 switch (State) { 10765 case OMPDeclareSimdDeclAttr::BS_Undefined: 10766 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 10767 OutputBecomesInput, Fn); 10768 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10769 OutputBecomesInput, Fn); 10770 break; 10771 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10772 addAArch64VectorName(UserVLEN, "N", Prefix, ISA, ParSeq, MangledName, 10773 OutputBecomesInput, Fn); 10774 break; 10775 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10776 addAArch64VectorName(UserVLEN, "M", Prefix, ISA, ParSeq, MangledName, 10777 OutputBecomesInput, Fn); 10778 break; 10779 } 10780 } 10781 } else { 10782 // If no user simdlen is provided, follow the AAVFABI rules for 10783 // generating the vector length. 10784 if (ISA == 's') { 10785 // SVE, section 3.4.1, item 1. 10786 addAArch64VectorName("x", "M", Prefix, ISA, ParSeq, MangledName, 10787 OutputBecomesInput, Fn); 10788 } else { 10789 assert(ISA == 'n' && "Expected ISA either 's' or 'n'."); 10790 // Advanced SIMD, Section 3.3.1 of the AAVFABI, generates one or 10791 // two vector names depending on the use of the clause 10792 // `[not]inbranch`. 10793 switch (State) { 10794 case OMPDeclareSimdDeclAttr::BS_Undefined: 10795 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 10796 OutputBecomesInput, Fn); 10797 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 10798 OutputBecomesInput, Fn); 10799 break; 10800 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 10801 addAArch64AdvSIMDNDSNames(NDS, "N", Prefix, ISA, ParSeq, MangledName, 10802 OutputBecomesInput, Fn); 10803 break; 10804 case OMPDeclareSimdDeclAttr::BS_Inbranch: 10805 addAArch64AdvSIMDNDSNames(NDS, "M", Prefix, ISA, ParSeq, MangledName, 10806 OutputBecomesInput, Fn); 10807 break; 10808 } 10809 } 10810 } 10811 } 10812 10813 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD, 10814 llvm::Function *Fn) { 10815 ASTContext &C = CGM.getContext(); 10816 FD = FD->getMostRecentDecl(); 10817 // Map params to their positions in function decl. 10818 llvm::DenseMap<const Decl *, unsigned> ParamPositions; 10819 if (isa<CXXMethodDecl>(FD)) 10820 ParamPositions.try_emplace(FD, 0); 10821 unsigned ParamPos = ParamPositions.size(); 10822 for (const ParmVarDecl *P : FD->parameters()) { 10823 ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos); 10824 ++ParamPos; 10825 } 10826 while (FD) { 10827 for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) { 10828 llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size()); 10829 // Mark uniform parameters. 10830 for (const Expr *E : Attr->uniforms()) { 10831 E = E->IgnoreParenImpCasts(); 10832 unsigned Pos; 10833 if (isa<CXXThisExpr>(E)) { 10834 Pos = ParamPositions[FD]; 10835 } else { 10836 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 10837 ->getCanonicalDecl(); 10838 Pos = ParamPositions[PVD]; 10839 } 10840 ParamAttrs[Pos].Kind = Uniform; 10841 } 10842 // Get alignment info. 10843 auto NI = Attr->alignments_begin(); 10844 for (const Expr *E : Attr->aligneds()) { 10845 E = E->IgnoreParenImpCasts(); 10846 unsigned Pos; 10847 QualType ParmTy; 10848 if (isa<CXXThisExpr>(E)) { 10849 Pos = ParamPositions[FD]; 10850 ParmTy = E->getType(); 10851 } else { 10852 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 10853 ->getCanonicalDecl(); 10854 Pos = ParamPositions[PVD]; 10855 ParmTy = PVD->getType(); 10856 } 10857 ParamAttrs[Pos].Alignment = 10858 (*NI) 10859 ? (*NI)->EvaluateKnownConstInt(C) 10860 : llvm::APSInt::getUnsigned( 10861 C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy)) 10862 .getQuantity()); 10863 ++NI; 10864 } 10865 // Mark linear parameters. 10866 auto SI = Attr->steps_begin(); 10867 auto MI = Attr->modifiers_begin(); 10868 for (const Expr *E : Attr->linears()) { 10869 E = E->IgnoreParenImpCasts(); 10870 unsigned Pos; 10871 if (isa<CXXThisExpr>(E)) { 10872 Pos = ParamPositions[FD]; 10873 } else { 10874 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 10875 ->getCanonicalDecl(); 10876 Pos = ParamPositions[PVD]; 10877 } 10878 ParamAttrTy &ParamAttr = ParamAttrs[Pos]; 10879 ParamAttr.Kind = Linear; 10880 if (*SI) { 10881 Expr::EvalResult Result; 10882 if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) { 10883 if (const auto *DRE = 10884 cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) { 10885 if (const auto *StridePVD = cast<ParmVarDecl>(DRE->getDecl())) { 10886 ParamAttr.Kind = LinearWithVarStride; 10887 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned( 10888 ParamPositions[StridePVD->getCanonicalDecl()]); 10889 } 10890 } 10891 } else { 10892 ParamAttr.StrideOrArg = Result.Val.getInt(); 10893 } 10894 } 10895 ++SI; 10896 ++MI; 10897 } 10898 llvm::APSInt VLENVal; 10899 SourceLocation ExprLoc; 10900 const Expr *VLENExpr = Attr->getSimdlen(); 10901 if (VLENExpr) { 10902 VLENVal = VLENExpr->EvaluateKnownConstInt(C); 10903 ExprLoc = VLENExpr->getExprLoc(); 10904 } 10905 OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState(); 10906 if (CGM.getTriple().getArch() == llvm::Triple::x86 || 10907 CGM.getTriple().getArch() == llvm::Triple::x86_64) { 10908 emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State); 10909 } else if (CGM.getTriple().getArch() == llvm::Triple::aarch64) { 10910 unsigned VLEN = VLENVal.getExtValue(); 10911 StringRef MangledName = Fn->getName(); 10912 if (CGM.getTarget().hasFeature("sve")) 10913 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 10914 MangledName, 's', 128, Fn, ExprLoc); 10915 if (CGM.getTarget().hasFeature("neon")) 10916 emitAArch64DeclareSimdFunction(CGM, FD, VLEN, ParamAttrs, State, 10917 MangledName, 'n', 128, Fn, ExprLoc); 10918 } 10919 } 10920 FD = FD->getPreviousDecl(); 10921 } 10922 } 10923 10924 namespace { 10925 /// Cleanup action for doacross support. 10926 class DoacrossCleanupTy final : public EHScopeStack::Cleanup { 10927 public: 10928 static const int DoacrossFinArgs = 2; 10929 10930 private: 10931 llvm::FunctionCallee RTLFn; 10932 llvm::Value *Args[DoacrossFinArgs]; 10933 10934 public: 10935 DoacrossCleanupTy(llvm::FunctionCallee RTLFn, 10936 ArrayRef<llvm::Value *> CallArgs) 10937 : RTLFn(RTLFn) { 10938 assert(CallArgs.size() == DoacrossFinArgs); 10939 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 10940 } 10941 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 10942 if (!CGF.HaveInsertPoint()) 10943 return; 10944 CGF.EmitRuntimeCall(RTLFn, Args); 10945 } 10946 }; 10947 } // namespace 10948 10949 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF, 10950 const OMPLoopDirective &D, 10951 ArrayRef<Expr *> NumIterations) { 10952 if (!CGF.HaveInsertPoint()) 10953 return; 10954 10955 ASTContext &C = CGM.getContext(); 10956 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 10957 RecordDecl *RD; 10958 if (KmpDimTy.isNull()) { 10959 // Build struct kmp_dim { // loop bounds info casted to kmp_int64 10960 // kmp_int64 lo; // lower 10961 // kmp_int64 up; // upper 10962 // kmp_int64 st; // stride 10963 // }; 10964 RD = C.buildImplicitRecord("kmp_dim"); 10965 RD->startDefinition(); 10966 addFieldToRecordDecl(C, RD, Int64Ty); 10967 addFieldToRecordDecl(C, RD, Int64Ty); 10968 addFieldToRecordDecl(C, RD, Int64Ty); 10969 RD->completeDefinition(); 10970 KmpDimTy = C.getRecordType(RD); 10971 } else { 10972 RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl()); 10973 } 10974 llvm::APInt Size(/*numBits=*/32, NumIterations.size()); 10975 QualType ArrayTy = 10976 C.getConstantArrayType(KmpDimTy, Size, nullptr, ArrayType::Normal, 0); 10977 10978 Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims"); 10979 CGF.EmitNullInitialization(DimsAddr, ArrayTy); 10980 enum { LowerFD = 0, UpperFD, StrideFD }; 10981 // Fill dims with data. 10982 for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) { 10983 LValue DimsLVal = CGF.MakeAddrLValue( 10984 CGF.Builder.CreateConstArrayGEP(DimsAddr, I), KmpDimTy); 10985 // dims.upper = num_iterations; 10986 LValue UpperLVal = CGF.EmitLValueForField( 10987 DimsLVal, *std::next(RD->field_begin(), UpperFD)); 10988 llvm::Value *NumIterVal = 10989 CGF.EmitScalarConversion(CGF.EmitScalarExpr(NumIterations[I]), 10990 D.getNumIterations()->getType(), Int64Ty, 10991 D.getNumIterations()->getExprLoc()); 10992 CGF.EmitStoreOfScalar(NumIterVal, UpperLVal); 10993 // dims.stride = 1; 10994 LValue StrideLVal = CGF.EmitLValueForField( 10995 DimsLVal, *std::next(RD->field_begin(), StrideFD)); 10996 CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1), 10997 StrideLVal); 10998 } 10999 11000 // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, 11001 // kmp_int32 num_dims, struct kmp_dim * dims); 11002 llvm::Value *Args[] = { 11003 emitUpdateLocation(CGF, D.getBeginLoc()), 11004 getThreadID(CGF, D.getBeginLoc()), 11005 llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()), 11006 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11007 CGF.Builder.CreateConstArrayGEP(DimsAddr, 0).getPointer(), 11008 CGM.VoidPtrTy)}; 11009 11010 llvm::FunctionCallee RTLFn = 11011 createRuntimeFunction(OMPRTL__kmpc_doacross_init); 11012 CGF.EmitRuntimeCall(RTLFn, Args); 11013 llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = { 11014 emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())}; 11015 llvm::FunctionCallee FiniRTLFn = 11016 createRuntimeFunction(OMPRTL__kmpc_doacross_fini); 11017 CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 11018 llvm::makeArrayRef(FiniArgs)); 11019 } 11020 11021 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 11022 const OMPDependClause *C) { 11023 QualType Int64Ty = 11024 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 11025 llvm::APInt Size(/*numBits=*/32, C->getNumLoops()); 11026 QualType ArrayTy = CGM.getContext().getConstantArrayType( 11027 Int64Ty, Size, nullptr, ArrayType::Normal, 0); 11028 Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr"); 11029 for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) { 11030 const Expr *CounterVal = C->getLoopData(I); 11031 assert(CounterVal); 11032 llvm::Value *CntVal = CGF.EmitScalarConversion( 11033 CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty, 11034 CounterVal->getExprLoc()); 11035 CGF.EmitStoreOfScalar(CntVal, CGF.Builder.CreateConstArrayGEP(CntAddr, I), 11036 /*Volatile=*/false, Int64Ty); 11037 } 11038 llvm::Value *Args[] = { 11039 emitUpdateLocation(CGF, C->getBeginLoc()), 11040 getThreadID(CGF, C->getBeginLoc()), 11041 CGF.Builder.CreateConstArrayGEP(CntAddr, 0).getPointer()}; 11042 llvm::FunctionCallee RTLFn; 11043 if (C->getDependencyKind() == OMPC_DEPEND_source) { 11044 RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_post); 11045 } else { 11046 assert(C->getDependencyKind() == OMPC_DEPEND_sink); 11047 RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_wait); 11048 } 11049 CGF.EmitRuntimeCall(RTLFn, Args); 11050 } 11051 11052 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc, 11053 llvm::FunctionCallee Callee, 11054 ArrayRef<llvm::Value *> Args) const { 11055 assert(Loc.isValid() && "Outlined function call location must be valid."); 11056 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 11057 11058 if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) { 11059 if (Fn->doesNotThrow()) { 11060 CGF.EmitNounwindRuntimeCall(Fn, Args); 11061 return; 11062 } 11063 } 11064 CGF.EmitRuntimeCall(Callee, Args); 11065 } 11066 11067 void CGOpenMPRuntime::emitOutlinedFunctionCall( 11068 CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn, 11069 ArrayRef<llvm::Value *> Args) const { 11070 emitCall(CGF, Loc, OutlinedFn, Args); 11071 } 11072 11073 void CGOpenMPRuntime::emitFunctionProlog(CodeGenFunction &CGF, const Decl *D) { 11074 if (const auto *FD = dyn_cast<FunctionDecl>(D)) 11075 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(FD)) 11076 HasEmittedDeclareTargetRegion = true; 11077 } 11078 11079 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF, 11080 const VarDecl *NativeParam, 11081 const VarDecl *TargetParam) const { 11082 return CGF.GetAddrOfLocalVar(NativeParam); 11083 } 11084 11085 namespace { 11086 /// Cleanup action for allocate support. 11087 class OMPAllocateCleanupTy final : public EHScopeStack::Cleanup { 11088 public: 11089 static const int CleanupArgs = 3; 11090 11091 private: 11092 llvm::FunctionCallee RTLFn; 11093 llvm::Value *Args[CleanupArgs]; 11094 11095 public: 11096 OMPAllocateCleanupTy(llvm::FunctionCallee RTLFn, 11097 ArrayRef<llvm::Value *> CallArgs) 11098 : RTLFn(RTLFn) { 11099 assert(CallArgs.size() == CleanupArgs && 11100 "Size of arguments does not match."); 11101 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 11102 } 11103 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 11104 if (!CGF.HaveInsertPoint()) 11105 return; 11106 CGF.EmitRuntimeCall(RTLFn, Args); 11107 } 11108 }; 11109 } // namespace 11110 11111 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF, 11112 const VarDecl *VD) { 11113 if (!VD) 11114 return Address::invalid(); 11115 const VarDecl *CVD = VD->getCanonicalDecl(); 11116 if (!CVD->hasAttr<OMPAllocateDeclAttr>()) 11117 return Address::invalid(); 11118 const auto *AA = CVD->getAttr<OMPAllocateDeclAttr>(); 11119 // Use the default allocation. 11120 if (AA->getAllocatorType() == OMPAllocateDeclAttr::OMPDefaultMemAlloc && 11121 !AA->getAllocator()) 11122 return Address::invalid(); 11123 llvm::Value *Size; 11124 CharUnits Align = CGM.getContext().getDeclAlign(CVD); 11125 if (CVD->getType()->isVariablyModifiedType()) { 11126 Size = CGF.getTypeSize(CVD->getType()); 11127 // Align the size: ((size + align - 1) / align) * align 11128 Size = CGF.Builder.CreateNUWAdd( 11129 Size, CGM.getSize(Align - CharUnits::fromQuantity(1))); 11130 Size = CGF.Builder.CreateUDiv(Size, CGM.getSize(Align)); 11131 Size = CGF.Builder.CreateNUWMul(Size, CGM.getSize(Align)); 11132 } else { 11133 CharUnits Sz = CGM.getContext().getTypeSizeInChars(CVD->getType()); 11134 Size = CGM.getSize(Sz.alignTo(Align)); 11135 } 11136 llvm::Value *ThreadID = getThreadID(CGF, CVD->getBeginLoc()); 11137 assert(AA->getAllocator() && 11138 "Expected allocator expression for non-default allocator."); 11139 llvm::Value *Allocator = CGF.EmitScalarExpr(AA->getAllocator()); 11140 // According to the standard, the original allocator type is a enum (integer). 11141 // Convert to pointer type, if required. 11142 if (Allocator->getType()->isIntegerTy()) 11143 Allocator = CGF.Builder.CreateIntToPtr(Allocator, CGM.VoidPtrTy); 11144 else if (Allocator->getType()->isPointerTy()) 11145 Allocator = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Allocator, 11146 CGM.VoidPtrTy); 11147 llvm::Value *Args[] = {ThreadID, Size, Allocator}; 11148 11149 llvm::Value *Addr = 11150 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_alloc), Args, 11151 CVD->getName() + ".void.addr"); 11152 llvm::Value *FiniArgs[OMPAllocateCleanupTy::CleanupArgs] = {ThreadID, Addr, 11153 Allocator}; 11154 llvm::FunctionCallee FiniRTLFn = createRuntimeFunction(OMPRTL__kmpc_free); 11155 11156 CGF.EHStack.pushCleanup<OMPAllocateCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 11157 llvm::makeArrayRef(FiniArgs)); 11158 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 11159 Addr, 11160 CGF.ConvertTypeForMem(CGM.getContext().getPointerType(CVD->getType())), 11161 CVD->getName() + ".addr"); 11162 return Address(Addr, Align); 11163 } 11164 11165 /// Checks current context and returns true if it matches the context selector. 11166 template <OMPDeclareVariantAttr::CtxSelectorSetType CtxSet, 11167 OMPDeclareVariantAttr::CtxSelectorType Ctx> 11168 static bool checkContext(const OMPDeclareVariantAttr *A) { 11169 assert(CtxSet != OMPDeclareVariantAttr::CtxSetUnknown && 11170 Ctx != OMPDeclareVariantAttr::CtxUnknown && 11171 "Unknown context selector or context selector set."); 11172 return false; 11173 } 11174 11175 /// Checks for implementation={vendor(<vendor>)} context selector. 11176 /// \returns true iff <vendor>="llvm", false otherwise. 11177 template <> 11178 bool checkContext<OMPDeclareVariantAttr::CtxSetImplementation, 11179 OMPDeclareVariantAttr::CtxVendor>( 11180 const OMPDeclareVariantAttr *A) { 11181 return llvm::all_of(A->implVendors(), 11182 [](StringRef S) { return !S.compare_lower("llvm"); }); 11183 } 11184 11185 static bool greaterCtxScore(ASTContext &Ctx, const Expr *LHS, const Expr *RHS) { 11186 // If both scores are unknown, choose the very first one. 11187 if (!LHS && !RHS) 11188 return true; 11189 // If only one is known, return this one. 11190 if (LHS && !RHS) 11191 return true; 11192 if (!LHS && RHS) 11193 return false; 11194 llvm::APSInt LHSVal = LHS->EvaluateKnownConstInt(Ctx); 11195 llvm::APSInt RHSVal = RHS->EvaluateKnownConstInt(Ctx); 11196 return llvm::APSInt::compareValues(LHSVal, RHSVal) >= 0; 11197 } 11198 11199 namespace { 11200 /// Comparator for the priority queue for context selector. 11201 class OMPDeclareVariantAttrComparer 11202 : public std::greater<const OMPDeclareVariantAttr *> { 11203 private: 11204 ASTContext &Ctx; 11205 11206 public: 11207 OMPDeclareVariantAttrComparer(ASTContext &Ctx) : Ctx(Ctx) {} 11208 bool operator()(const OMPDeclareVariantAttr *LHS, 11209 const OMPDeclareVariantAttr *RHS) const { 11210 const Expr *LHSExpr = nullptr; 11211 const Expr *RHSExpr = nullptr; 11212 if (LHS->getCtxScore() == OMPDeclareVariantAttr::ScoreSpecified) 11213 LHSExpr = LHS->getScore(); 11214 if (RHS->getCtxScore() == OMPDeclareVariantAttr::ScoreSpecified) 11215 RHSExpr = RHS->getScore(); 11216 return greaterCtxScore(Ctx, LHSExpr, RHSExpr); 11217 } 11218 }; 11219 } // anonymous namespace 11220 11221 /// Finds the variant function that matches current context with its context 11222 /// selector. 11223 static const FunctionDecl *getDeclareVariantFunction(ASTContext &Ctx, 11224 const FunctionDecl *FD) { 11225 if (!FD->hasAttrs() || !FD->hasAttr<OMPDeclareVariantAttr>()) 11226 return FD; 11227 // Iterate through all DeclareVariant attributes and check context selectors. 11228 auto &&Comparer = [&Ctx](const OMPDeclareVariantAttr *LHS, 11229 const OMPDeclareVariantAttr *RHS) { 11230 const Expr *LHSExpr = nullptr; 11231 const Expr *RHSExpr = nullptr; 11232 if (LHS->getCtxScore() == OMPDeclareVariantAttr::ScoreSpecified) 11233 LHSExpr = LHS->getScore(); 11234 if (RHS->getCtxScore() == OMPDeclareVariantAttr::ScoreSpecified) 11235 RHSExpr = RHS->getScore(); 11236 return greaterCtxScore(Ctx, LHSExpr, RHSExpr); 11237 }; 11238 const OMPDeclareVariantAttr *TopMostAttr = nullptr; 11239 for (const auto *A : FD->specific_attrs<OMPDeclareVariantAttr>()) { 11240 const OMPDeclareVariantAttr *SelectedAttr = nullptr; 11241 switch (A->getCtxSelectorSet()) { 11242 case OMPDeclareVariantAttr::CtxSetImplementation: 11243 switch (A->getCtxSelector()) { 11244 case OMPDeclareVariantAttr::CtxVendor: 11245 if (checkContext<OMPDeclareVariantAttr::CtxSetImplementation, 11246 OMPDeclareVariantAttr::CtxVendor>(A)) 11247 SelectedAttr = A; 11248 break; 11249 case OMPDeclareVariantAttr::CtxUnknown: 11250 llvm_unreachable( 11251 "Unknown context selector in implementation selector set."); 11252 } 11253 break; 11254 case OMPDeclareVariantAttr::CtxSetUnknown: 11255 llvm_unreachable("Unknown context selector set."); 11256 } 11257 // If the attribute matches the context, find the attribute with the highest 11258 // score. 11259 if (SelectedAttr && (!TopMostAttr || !Comparer(TopMostAttr, SelectedAttr))) 11260 TopMostAttr = SelectedAttr; 11261 } 11262 if (!TopMostAttr) 11263 return FD; 11264 return cast<FunctionDecl>( 11265 cast<DeclRefExpr>(TopMostAttr->getVariantFuncRef()->IgnoreParenImpCasts()) 11266 ->getDecl()); 11267 } 11268 11269 bool CGOpenMPRuntime::emitDeclareVariant(GlobalDecl GD, bool IsForDefinition) { 11270 const auto *D = cast<FunctionDecl>(GD.getDecl()); 11271 // If the original function is defined already, use its definition. 11272 StringRef MangledName = CGM.getMangledName(GD); 11273 llvm::GlobalValue *Orig = CGM.GetGlobalValue(MangledName); 11274 if (Orig && !Orig->isDeclaration()) 11275 return false; 11276 const FunctionDecl *NewFD = getDeclareVariantFunction(CGM.getContext(), D); 11277 // Emit original function if it does not have declare variant attribute or the 11278 // context does not match. 11279 if (NewFD == D) 11280 return false; 11281 GlobalDecl NewGD = GD.getWithDecl(NewFD); 11282 if (tryEmitDeclareVariant(NewGD, GD, Orig, IsForDefinition)) { 11283 DeferredVariantFunction.erase(D); 11284 return true; 11285 } 11286 DeferredVariantFunction.insert(std::make_pair(D, std::make_pair(NewGD, GD))); 11287 return true; 11288 } 11289 11290 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction( 11291 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11292 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 11293 llvm_unreachable("Not supported in SIMD-only mode"); 11294 } 11295 11296 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction( 11297 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11298 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 11299 llvm_unreachable("Not supported in SIMD-only mode"); 11300 } 11301 11302 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction( 11303 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 11304 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 11305 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 11306 bool Tied, unsigned &NumberOfParts) { 11307 llvm_unreachable("Not supported in SIMD-only mode"); 11308 } 11309 11310 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF, 11311 SourceLocation Loc, 11312 llvm::Function *OutlinedFn, 11313 ArrayRef<llvm::Value *> CapturedVars, 11314 const Expr *IfCond) { 11315 llvm_unreachable("Not supported in SIMD-only mode"); 11316 } 11317 11318 void CGOpenMPSIMDRuntime::emitCriticalRegion( 11319 CodeGenFunction &CGF, StringRef CriticalName, 11320 const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc, 11321 const Expr *Hint) { 11322 llvm_unreachable("Not supported in SIMD-only mode"); 11323 } 11324 11325 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF, 11326 const RegionCodeGenTy &MasterOpGen, 11327 SourceLocation Loc) { 11328 llvm_unreachable("Not supported in SIMD-only mode"); 11329 } 11330 11331 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 11332 SourceLocation Loc) { 11333 llvm_unreachable("Not supported in SIMD-only mode"); 11334 } 11335 11336 void CGOpenMPSIMDRuntime::emitTaskgroupRegion( 11337 CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen, 11338 SourceLocation Loc) { 11339 llvm_unreachable("Not supported in SIMD-only mode"); 11340 } 11341 11342 void CGOpenMPSIMDRuntime::emitSingleRegion( 11343 CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen, 11344 SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars, 11345 ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs, 11346 ArrayRef<const Expr *> AssignmentOps) { 11347 llvm_unreachable("Not supported in SIMD-only mode"); 11348 } 11349 11350 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF, 11351 const RegionCodeGenTy &OrderedOpGen, 11352 SourceLocation Loc, 11353 bool IsThreads) { 11354 llvm_unreachable("Not supported in SIMD-only mode"); 11355 } 11356 11357 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF, 11358 SourceLocation Loc, 11359 OpenMPDirectiveKind Kind, 11360 bool EmitChecks, 11361 bool ForceSimpleCall) { 11362 llvm_unreachable("Not supported in SIMD-only mode"); 11363 } 11364 11365 void CGOpenMPSIMDRuntime::emitForDispatchInit( 11366 CodeGenFunction &CGF, SourceLocation Loc, 11367 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 11368 bool Ordered, const DispatchRTInput &DispatchValues) { 11369 llvm_unreachable("Not supported in SIMD-only mode"); 11370 } 11371 11372 void CGOpenMPSIMDRuntime::emitForStaticInit( 11373 CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind, 11374 const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) { 11375 llvm_unreachable("Not supported in SIMD-only mode"); 11376 } 11377 11378 void CGOpenMPSIMDRuntime::emitDistributeStaticInit( 11379 CodeGenFunction &CGF, SourceLocation Loc, 11380 OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) { 11381 llvm_unreachable("Not supported in SIMD-only mode"); 11382 } 11383 11384 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 11385 SourceLocation Loc, 11386 unsigned IVSize, 11387 bool IVSigned) { 11388 llvm_unreachable("Not supported in SIMD-only mode"); 11389 } 11390 11391 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF, 11392 SourceLocation Loc, 11393 OpenMPDirectiveKind DKind) { 11394 llvm_unreachable("Not supported in SIMD-only mode"); 11395 } 11396 11397 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF, 11398 SourceLocation Loc, 11399 unsigned IVSize, bool IVSigned, 11400 Address IL, Address LB, 11401 Address UB, Address ST) { 11402 llvm_unreachable("Not supported in SIMD-only mode"); 11403 } 11404 11405 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 11406 llvm::Value *NumThreads, 11407 SourceLocation Loc) { 11408 llvm_unreachable("Not supported in SIMD-only mode"); 11409 } 11410 11411 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF, 11412 OpenMPProcBindClauseKind ProcBind, 11413 SourceLocation Loc) { 11414 llvm_unreachable("Not supported in SIMD-only mode"); 11415 } 11416 11417 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 11418 const VarDecl *VD, 11419 Address VDAddr, 11420 SourceLocation Loc) { 11421 llvm_unreachable("Not supported in SIMD-only mode"); 11422 } 11423 11424 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition( 11425 const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit, 11426 CodeGenFunction *CGF) { 11427 llvm_unreachable("Not supported in SIMD-only mode"); 11428 } 11429 11430 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate( 11431 CodeGenFunction &CGF, QualType VarType, StringRef Name) { 11432 llvm_unreachable("Not supported in SIMD-only mode"); 11433 } 11434 11435 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF, 11436 ArrayRef<const Expr *> Vars, 11437 SourceLocation Loc) { 11438 llvm_unreachable("Not supported in SIMD-only mode"); 11439 } 11440 11441 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 11442 const OMPExecutableDirective &D, 11443 llvm::Function *TaskFunction, 11444 QualType SharedsTy, Address Shareds, 11445 const Expr *IfCond, 11446 const OMPTaskDataTy &Data) { 11447 llvm_unreachable("Not supported in SIMD-only mode"); 11448 } 11449 11450 void CGOpenMPSIMDRuntime::emitTaskLoopCall( 11451 CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D, 11452 llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, 11453 const Expr *IfCond, const OMPTaskDataTy &Data) { 11454 llvm_unreachable("Not supported in SIMD-only mode"); 11455 } 11456 11457 void CGOpenMPSIMDRuntime::emitReduction( 11458 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates, 11459 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 11460 ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) { 11461 assert(Options.SimpleReduction && "Only simple reduction is expected."); 11462 CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs, 11463 ReductionOps, Options); 11464 } 11465 11466 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit( 11467 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 11468 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 11469 llvm_unreachable("Not supported in SIMD-only mode"); 11470 } 11471 11472 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 11473 SourceLocation Loc, 11474 ReductionCodeGen &RCG, 11475 unsigned N) { 11476 llvm_unreachable("Not supported in SIMD-only mode"); 11477 } 11478 11479 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF, 11480 SourceLocation Loc, 11481 llvm::Value *ReductionsPtr, 11482 LValue SharedLVal) { 11483 llvm_unreachable("Not supported in SIMD-only mode"); 11484 } 11485 11486 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 11487 SourceLocation Loc) { 11488 llvm_unreachable("Not supported in SIMD-only mode"); 11489 } 11490 11491 void CGOpenMPSIMDRuntime::emitCancellationPointCall( 11492 CodeGenFunction &CGF, SourceLocation Loc, 11493 OpenMPDirectiveKind CancelRegion) { 11494 llvm_unreachable("Not supported in SIMD-only mode"); 11495 } 11496 11497 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF, 11498 SourceLocation Loc, const Expr *IfCond, 11499 OpenMPDirectiveKind CancelRegion) { 11500 llvm_unreachable("Not supported in SIMD-only mode"); 11501 } 11502 11503 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction( 11504 const OMPExecutableDirective &D, StringRef ParentName, 11505 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 11506 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 11507 llvm_unreachable("Not supported in SIMD-only mode"); 11508 } 11509 11510 void CGOpenMPSIMDRuntime::emitTargetCall( 11511 CodeGenFunction &CGF, const OMPExecutableDirective &D, 11512 llvm::Function *OutlinedFn, llvm::Value *OutlinedFnID, const Expr *IfCond, 11513 const Expr *Device, 11514 llvm::function_ref<llvm::Value *(CodeGenFunction &CGF, 11515 const OMPLoopDirective &D)> 11516 SizeEmitter) { 11517 llvm_unreachable("Not supported in SIMD-only mode"); 11518 } 11519 11520 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) { 11521 llvm_unreachable("Not supported in SIMD-only mode"); 11522 } 11523 11524 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 11525 llvm_unreachable("Not supported in SIMD-only mode"); 11526 } 11527 11528 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) { 11529 return false; 11530 } 11531 11532 llvm::Function *CGOpenMPSIMDRuntime::emitRegistrationFunction() { 11533 return nullptr; 11534 } 11535 11536 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF, 11537 const OMPExecutableDirective &D, 11538 SourceLocation Loc, 11539 llvm::Function *OutlinedFn, 11540 ArrayRef<llvm::Value *> CapturedVars) { 11541 llvm_unreachable("Not supported in SIMD-only mode"); 11542 } 11543 11544 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 11545 const Expr *NumTeams, 11546 const Expr *ThreadLimit, 11547 SourceLocation Loc) { 11548 llvm_unreachable("Not supported in SIMD-only mode"); 11549 } 11550 11551 void CGOpenMPSIMDRuntime::emitTargetDataCalls( 11552 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 11553 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 11554 llvm_unreachable("Not supported in SIMD-only mode"); 11555 } 11556 11557 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall( 11558 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 11559 const Expr *Device) { 11560 llvm_unreachable("Not supported in SIMD-only mode"); 11561 } 11562 11563 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF, 11564 const OMPLoopDirective &D, 11565 ArrayRef<Expr *> NumIterations) { 11566 llvm_unreachable("Not supported in SIMD-only mode"); 11567 } 11568 11569 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 11570 const OMPDependClause *C) { 11571 llvm_unreachable("Not supported in SIMD-only mode"); 11572 } 11573 11574 const VarDecl * 11575 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD, 11576 const VarDecl *NativeParam) const { 11577 llvm_unreachable("Not supported in SIMD-only mode"); 11578 } 11579 11580 Address 11581 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF, 11582 const VarDecl *NativeParam, 11583 const VarDecl *TargetParam) const { 11584 llvm_unreachable("Not supported in SIMD-only mode"); 11585 } 11586