1 //===----- CGOpenMPRuntime.cpp - Interface to OpenMP Runtimes -------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This provides a class for OpenMP runtime code generation. 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "CGCXXABI.h" 14 #include "CGCleanup.h" 15 #include "CGOpenMPRuntime.h" 16 #include "CGRecordLayout.h" 17 #include "CodeGenFunction.h" 18 #include "clang/CodeGen/ConstantInitBuilder.h" 19 #include "clang/AST/Decl.h" 20 #include "clang/AST/StmtOpenMP.h" 21 #include "clang/Basic/BitmaskEnum.h" 22 #include "llvm/ADT/ArrayRef.h" 23 #include "llvm/Bitcode/BitcodeReader.h" 24 #include "llvm/IR/DerivedTypes.h" 25 #include "llvm/IR/GlobalValue.h" 26 #include "llvm/IR/Value.h" 27 #include "llvm/Support/Format.h" 28 #include "llvm/Support/raw_ostream.h" 29 #include <cassert> 30 31 using namespace clang; 32 using namespace CodeGen; 33 34 namespace { 35 /// Base class for handling code generation inside OpenMP regions. 36 class CGOpenMPRegionInfo : public CodeGenFunction::CGCapturedStmtInfo { 37 public: 38 /// Kinds of OpenMP regions used in codegen. 39 enum CGOpenMPRegionKind { 40 /// Region with outlined function for standalone 'parallel' 41 /// directive. 42 ParallelOutlinedRegion, 43 /// Region with outlined function for standalone 'task' directive. 44 TaskOutlinedRegion, 45 /// Region for constructs that do not require function outlining, 46 /// like 'for', 'sections', 'atomic' etc. directives. 47 InlinedRegion, 48 /// Region with outlined function for standalone 'target' directive. 49 TargetRegion, 50 }; 51 52 CGOpenMPRegionInfo(const CapturedStmt &CS, 53 const CGOpenMPRegionKind RegionKind, 54 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 55 bool HasCancel) 56 : CGCapturedStmtInfo(CS, CR_OpenMP), RegionKind(RegionKind), 57 CodeGen(CodeGen), Kind(Kind), HasCancel(HasCancel) {} 58 59 CGOpenMPRegionInfo(const CGOpenMPRegionKind RegionKind, 60 const RegionCodeGenTy &CodeGen, OpenMPDirectiveKind Kind, 61 bool HasCancel) 62 : CGCapturedStmtInfo(CR_OpenMP), RegionKind(RegionKind), CodeGen(CodeGen), 63 Kind(Kind), HasCancel(HasCancel) {} 64 65 /// Get a variable or parameter for storing global thread id 66 /// inside OpenMP construct. 67 virtual const VarDecl *getThreadIDVariable() const = 0; 68 69 /// Emit the captured statement body. 70 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override; 71 72 /// Get an LValue for the current ThreadID variable. 73 /// \return LValue for thread id variable. This LValue always has type int32*. 74 virtual LValue getThreadIDVariableLValue(CodeGenFunction &CGF); 75 76 virtual void emitUntiedSwitch(CodeGenFunction & /*CGF*/) {} 77 78 CGOpenMPRegionKind getRegionKind() const { return RegionKind; } 79 80 OpenMPDirectiveKind getDirectiveKind() const { return Kind; } 81 82 bool hasCancel() const { return HasCancel; } 83 84 static bool classof(const CGCapturedStmtInfo *Info) { 85 return Info->getKind() == CR_OpenMP; 86 } 87 88 ~CGOpenMPRegionInfo() override = default; 89 90 protected: 91 CGOpenMPRegionKind RegionKind; 92 RegionCodeGenTy CodeGen; 93 OpenMPDirectiveKind Kind; 94 bool HasCancel; 95 }; 96 97 /// API for captured statement code generation in OpenMP constructs. 98 class CGOpenMPOutlinedRegionInfo final : public CGOpenMPRegionInfo { 99 public: 100 CGOpenMPOutlinedRegionInfo(const CapturedStmt &CS, const VarDecl *ThreadIDVar, 101 const RegionCodeGenTy &CodeGen, 102 OpenMPDirectiveKind Kind, bool HasCancel, 103 StringRef HelperName) 104 : CGOpenMPRegionInfo(CS, ParallelOutlinedRegion, CodeGen, Kind, 105 HasCancel), 106 ThreadIDVar(ThreadIDVar), HelperName(HelperName) { 107 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 108 } 109 110 /// Get a variable or parameter for storing global thread id 111 /// inside OpenMP construct. 112 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 113 114 /// Get the name of the capture helper. 115 StringRef getHelperName() const override { return HelperName; } 116 117 static bool classof(const CGCapturedStmtInfo *Info) { 118 return CGOpenMPRegionInfo::classof(Info) && 119 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 120 ParallelOutlinedRegion; 121 } 122 123 private: 124 /// A variable or parameter storing global thread id for OpenMP 125 /// constructs. 126 const VarDecl *ThreadIDVar; 127 StringRef HelperName; 128 }; 129 130 /// API for captured statement code generation in OpenMP constructs. 131 class CGOpenMPTaskOutlinedRegionInfo final : public CGOpenMPRegionInfo { 132 public: 133 class UntiedTaskActionTy final : public PrePostActionTy { 134 bool Untied; 135 const VarDecl *PartIDVar; 136 const RegionCodeGenTy UntiedCodeGen; 137 llvm::SwitchInst *UntiedSwitch = nullptr; 138 139 public: 140 UntiedTaskActionTy(bool Tied, const VarDecl *PartIDVar, 141 const RegionCodeGenTy &UntiedCodeGen) 142 : Untied(!Tied), PartIDVar(PartIDVar), UntiedCodeGen(UntiedCodeGen) {} 143 void Enter(CodeGenFunction &CGF) override { 144 if (Untied) { 145 // Emit task switching point. 146 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 147 CGF.GetAddrOfLocalVar(PartIDVar), 148 PartIDVar->getType()->castAs<PointerType>()); 149 llvm::Value *Res = 150 CGF.EmitLoadOfScalar(PartIdLVal, PartIDVar->getLocation()); 151 llvm::BasicBlock *DoneBB = CGF.createBasicBlock(".untied.done."); 152 UntiedSwitch = CGF.Builder.CreateSwitch(Res, DoneBB); 153 CGF.EmitBlock(DoneBB); 154 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 155 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 156 UntiedSwitch->addCase(CGF.Builder.getInt32(0), 157 CGF.Builder.GetInsertBlock()); 158 emitUntiedSwitch(CGF); 159 } 160 } 161 void emitUntiedSwitch(CodeGenFunction &CGF) const { 162 if (Untied) { 163 LValue PartIdLVal = CGF.EmitLoadOfPointerLValue( 164 CGF.GetAddrOfLocalVar(PartIDVar), 165 PartIDVar->getType()->castAs<PointerType>()); 166 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 167 PartIdLVal); 168 UntiedCodeGen(CGF); 169 CodeGenFunction::JumpDest CurPoint = 170 CGF.getJumpDestInCurrentScope(".untied.next."); 171 CGF.EmitBranchThroughCleanup(CGF.ReturnBlock); 172 CGF.EmitBlock(CGF.createBasicBlock(".untied.jmp.")); 173 UntiedSwitch->addCase(CGF.Builder.getInt32(UntiedSwitch->getNumCases()), 174 CGF.Builder.GetInsertBlock()); 175 CGF.EmitBranchThroughCleanup(CurPoint); 176 CGF.EmitBlock(CurPoint.getBlock()); 177 } 178 } 179 unsigned getNumberOfParts() const { return UntiedSwitch->getNumCases(); } 180 }; 181 CGOpenMPTaskOutlinedRegionInfo(const CapturedStmt &CS, 182 const VarDecl *ThreadIDVar, 183 const RegionCodeGenTy &CodeGen, 184 OpenMPDirectiveKind Kind, bool HasCancel, 185 const UntiedTaskActionTy &Action) 186 : CGOpenMPRegionInfo(CS, TaskOutlinedRegion, CodeGen, Kind, HasCancel), 187 ThreadIDVar(ThreadIDVar), Action(Action) { 188 assert(ThreadIDVar != nullptr && "No ThreadID in OpenMP region."); 189 } 190 191 /// Get a variable or parameter for storing global thread id 192 /// inside OpenMP construct. 193 const VarDecl *getThreadIDVariable() const override { return ThreadIDVar; } 194 195 /// Get an LValue for the current ThreadID variable. 196 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override; 197 198 /// Get the name of the capture helper. 199 StringRef getHelperName() const override { return ".omp_outlined."; } 200 201 void emitUntiedSwitch(CodeGenFunction &CGF) override { 202 Action.emitUntiedSwitch(CGF); 203 } 204 205 static bool classof(const CGCapturedStmtInfo *Info) { 206 return CGOpenMPRegionInfo::classof(Info) && 207 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == 208 TaskOutlinedRegion; 209 } 210 211 private: 212 /// A variable or parameter storing global thread id for OpenMP 213 /// constructs. 214 const VarDecl *ThreadIDVar; 215 /// Action for emitting code for untied tasks. 216 const UntiedTaskActionTy &Action; 217 }; 218 219 /// API for inlined captured statement code generation in OpenMP 220 /// constructs. 221 class CGOpenMPInlinedRegionInfo : public CGOpenMPRegionInfo { 222 public: 223 CGOpenMPInlinedRegionInfo(CodeGenFunction::CGCapturedStmtInfo *OldCSI, 224 const RegionCodeGenTy &CodeGen, 225 OpenMPDirectiveKind Kind, bool HasCancel) 226 : CGOpenMPRegionInfo(InlinedRegion, CodeGen, Kind, HasCancel), 227 OldCSI(OldCSI), 228 OuterRegionInfo(dyn_cast_or_null<CGOpenMPRegionInfo>(OldCSI)) {} 229 230 // Retrieve the value of the context parameter. 231 llvm::Value *getContextValue() const override { 232 if (OuterRegionInfo) 233 return OuterRegionInfo->getContextValue(); 234 llvm_unreachable("No context value for inlined OpenMP region"); 235 } 236 237 void setContextValue(llvm::Value *V) override { 238 if (OuterRegionInfo) { 239 OuterRegionInfo->setContextValue(V); 240 return; 241 } 242 llvm_unreachable("No context value for inlined OpenMP region"); 243 } 244 245 /// Lookup the captured field decl for a variable. 246 const FieldDecl *lookup(const VarDecl *VD) const override { 247 if (OuterRegionInfo) 248 return OuterRegionInfo->lookup(VD); 249 // If there is no outer outlined region,no need to lookup in a list of 250 // captured variables, we can use the original one. 251 return nullptr; 252 } 253 254 FieldDecl *getThisFieldDecl() const override { 255 if (OuterRegionInfo) 256 return OuterRegionInfo->getThisFieldDecl(); 257 return nullptr; 258 } 259 260 /// Get a variable or parameter for storing global thread id 261 /// inside OpenMP construct. 262 const VarDecl *getThreadIDVariable() const override { 263 if (OuterRegionInfo) 264 return OuterRegionInfo->getThreadIDVariable(); 265 return nullptr; 266 } 267 268 /// Get an LValue for the current ThreadID variable. 269 LValue getThreadIDVariableLValue(CodeGenFunction &CGF) override { 270 if (OuterRegionInfo) 271 return OuterRegionInfo->getThreadIDVariableLValue(CGF); 272 llvm_unreachable("No LValue for inlined OpenMP construct"); 273 } 274 275 /// Get the name of the capture helper. 276 StringRef getHelperName() const override { 277 if (auto *OuterRegionInfo = getOldCSI()) 278 return OuterRegionInfo->getHelperName(); 279 llvm_unreachable("No helper name for inlined OpenMP construct"); 280 } 281 282 void emitUntiedSwitch(CodeGenFunction &CGF) override { 283 if (OuterRegionInfo) 284 OuterRegionInfo->emitUntiedSwitch(CGF); 285 } 286 287 CodeGenFunction::CGCapturedStmtInfo *getOldCSI() const { return OldCSI; } 288 289 static bool classof(const CGCapturedStmtInfo *Info) { 290 return CGOpenMPRegionInfo::classof(Info) && 291 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == InlinedRegion; 292 } 293 294 ~CGOpenMPInlinedRegionInfo() override = default; 295 296 private: 297 /// CodeGen info about outer OpenMP region. 298 CodeGenFunction::CGCapturedStmtInfo *OldCSI; 299 CGOpenMPRegionInfo *OuterRegionInfo; 300 }; 301 302 /// API for captured statement code generation in OpenMP target 303 /// constructs. For this captures, implicit parameters are used instead of the 304 /// captured fields. The name of the target region has to be unique in a given 305 /// application so it is provided by the client, because only the client has 306 /// the information to generate that. 307 class CGOpenMPTargetRegionInfo final : public CGOpenMPRegionInfo { 308 public: 309 CGOpenMPTargetRegionInfo(const CapturedStmt &CS, 310 const RegionCodeGenTy &CodeGen, StringRef HelperName) 311 : CGOpenMPRegionInfo(CS, TargetRegion, CodeGen, OMPD_target, 312 /*HasCancel=*/false), 313 HelperName(HelperName) {} 314 315 /// This is unused for target regions because each starts executing 316 /// with a single thread. 317 const VarDecl *getThreadIDVariable() const override { return nullptr; } 318 319 /// Get the name of the capture helper. 320 StringRef getHelperName() const override { return HelperName; } 321 322 static bool classof(const CGCapturedStmtInfo *Info) { 323 return CGOpenMPRegionInfo::classof(Info) && 324 cast<CGOpenMPRegionInfo>(Info)->getRegionKind() == TargetRegion; 325 } 326 327 private: 328 StringRef HelperName; 329 }; 330 331 static void EmptyCodeGen(CodeGenFunction &, PrePostActionTy &) { 332 llvm_unreachable("No codegen for expressions"); 333 } 334 /// API for generation of expressions captured in a innermost OpenMP 335 /// region. 336 class CGOpenMPInnerExprInfo final : public CGOpenMPInlinedRegionInfo { 337 public: 338 CGOpenMPInnerExprInfo(CodeGenFunction &CGF, const CapturedStmt &CS) 339 : CGOpenMPInlinedRegionInfo(CGF.CapturedStmtInfo, EmptyCodeGen, 340 OMPD_unknown, 341 /*HasCancel=*/false), 342 PrivScope(CGF) { 343 // Make sure the globals captured in the provided statement are local by 344 // using the privatization logic. We assume the same variable is not 345 // captured more than once. 346 for (const auto &C : CS.captures()) { 347 if (!C.capturesVariable() && !C.capturesVariableByCopy()) 348 continue; 349 350 const VarDecl *VD = C.getCapturedVar(); 351 if (VD->isLocalVarDeclOrParm()) 352 continue; 353 354 DeclRefExpr DRE(CGF.getContext(), const_cast<VarDecl *>(VD), 355 /*RefersToEnclosingVariableOrCapture=*/false, 356 VD->getType().getNonReferenceType(), VK_LValue, 357 C.getLocation()); 358 PrivScope.addPrivate( 359 VD, [&CGF, &DRE]() { return CGF.EmitLValue(&DRE).getAddress(); }); 360 } 361 (void)PrivScope.Privatize(); 362 } 363 364 /// Lookup the captured field decl for a variable. 365 const FieldDecl *lookup(const VarDecl *VD) const override { 366 if (const FieldDecl *FD = CGOpenMPInlinedRegionInfo::lookup(VD)) 367 return FD; 368 return nullptr; 369 } 370 371 /// Emit the captured statement body. 372 void EmitBody(CodeGenFunction &CGF, const Stmt *S) override { 373 llvm_unreachable("No body for expressions"); 374 } 375 376 /// Get a variable or parameter for storing global thread id 377 /// inside OpenMP construct. 378 const VarDecl *getThreadIDVariable() const override { 379 llvm_unreachable("No thread id for expressions"); 380 } 381 382 /// Get the name of the capture helper. 383 StringRef getHelperName() const override { 384 llvm_unreachable("No helper name for expressions"); 385 } 386 387 static bool classof(const CGCapturedStmtInfo *Info) { return false; } 388 389 private: 390 /// Private scope to capture global variables. 391 CodeGenFunction::OMPPrivateScope PrivScope; 392 }; 393 394 /// RAII for emitting code of OpenMP constructs. 395 class InlinedOpenMPRegionRAII { 396 CodeGenFunction &CGF; 397 llvm::DenseMap<const VarDecl *, FieldDecl *> LambdaCaptureFields; 398 FieldDecl *LambdaThisCaptureField = nullptr; 399 const CodeGen::CGBlockInfo *BlockInfo = nullptr; 400 401 public: 402 /// Constructs region for combined constructs. 403 /// \param CodeGen Code generation sequence for combined directives. Includes 404 /// a list of functions used for code generation of implicitly inlined 405 /// regions. 406 InlinedOpenMPRegionRAII(CodeGenFunction &CGF, const RegionCodeGenTy &CodeGen, 407 OpenMPDirectiveKind Kind, bool HasCancel) 408 : CGF(CGF) { 409 // Start emission for the construct. 410 CGF.CapturedStmtInfo = new CGOpenMPInlinedRegionInfo( 411 CGF.CapturedStmtInfo, CodeGen, Kind, HasCancel); 412 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 413 LambdaThisCaptureField = CGF.LambdaThisCaptureField; 414 CGF.LambdaThisCaptureField = nullptr; 415 BlockInfo = CGF.BlockInfo; 416 CGF.BlockInfo = nullptr; 417 } 418 419 ~InlinedOpenMPRegionRAII() { 420 // Restore original CapturedStmtInfo only if we're done with code emission. 421 auto *OldCSI = 422 cast<CGOpenMPInlinedRegionInfo>(CGF.CapturedStmtInfo)->getOldCSI(); 423 delete CGF.CapturedStmtInfo; 424 CGF.CapturedStmtInfo = OldCSI; 425 std::swap(CGF.LambdaCaptureFields, LambdaCaptureFields); 426 CGF.LambdaThisCaptureField = LambdaThisCaptureField; 427 CGF.BlockInfo = BlockInfo; 428 } 429 }; 430 431 /// Values for bit flags used in the ident_t to describe the fields. 432 /// All enumeric elements are named and described in accordance with the code 433 /// from https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 434 enum OpenMPLocationFlags : unsigned { 435 /// Use trampoline for internal microtask. 436 OMP_IDENT_IMD = 0x01, 437 /// Use c-style ident structure. 438 OMP_IDENT_KMPC = 0x02, 439 /// Atomic reduction option for kmpc_reduce. 440 OMP_ATOMIC_REDUCE = 0x10, 441 /// Explicit 'barrier' directive. 442 OMP_IDENT_BARRIER_EXPL = 0x20, 443 /// Implicit barrier in code. 444 OMP_IDENT_BARRIER_IMPL = 0x40, 445 /// Implicit barrier in 'for' directive. 446 OMP_IDENT_BARRIER_IMPL_FOR = 0x40, 447 /// Implicit barrier in 'sections' directive. 448 OMP_IDENT_BARRIER_IMPL_SECTIONS = 0xC0, 449 /// Implicit barrier in 'single' directive. 450 OMP_IDENT_BARRIER_IMPL_SINGLE = 0x140, 451 /// Call of __kmp_for_static_init for static loop. 452 OMP_IDENT_WORK_LOOP = 0x200, 453 /// Call of __kmp_for_static_init for sections. 454 OMP_IDENT_WORK_SECTIONS = 0x400, 455 /// Call of __kmp_for_static_init for distribute. 456 OMP_IDENT_WORK_DISTRIBUTE = 0x800, 457 LLVM_MARK_AS_BITMASK_ENUM(/*LargestValue=*/OMP_IDENT_WORK_DISTRIBUTE) 458 }; 459 460 /// Describes ident structure that describes a source location. 461 /// All descriptions are taken from 462 /// https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h 463 /// Original structure: 464 /// typedef struct ident { 465 /// kmp_int32 reserved_1; /**< might be used in Fortran; 466 /// see above */ 467 /// kmp_int32 flags; /**< also f.flags; KMP_IDENT_xxx flags; 468 /// KMP_IDENT_KMPC identifies this union 469 /// member */ 470 /// kmp_int32 reserved_2; /**< not really used in Fortran any more; 471 /// see above */ 472 ///#if USE_ITT_BUILD 473 /// /* but currently used for storing 474 /// region-specific ITT */ 475 /// /* contextual information. */ 476 ///#endif /* USE_ITT_BUILD */ 477 /// kmp_int32 reserved_3; /**< source[4] in Fortran, do not use for 478 /// C++ */ 479 /// char const *psource; /**< String describing the source location. 480 /// The string is composed of semi-colon separated 481 // fields which describe the source file, 482 /// the function and a pair of line numbers that 483 /// delimit the construct. 484 /// */ 485 /// } ident_t; 486 enum IdentFieldIndex { 487 /// might be used in Fortran 488 IdentField_Reserved_1, 489 /// OMP_IDENT_xxx flags; OMP_IDENT_KMPC identifies this union member. 490 IdentField_Flags, 491 /// Not really used in Fortran any more 492 IdentField_Reserved_2, 493 /// Source[4] in Fortran, do not use for C++ 494 IdentField_Reserved_3, 495 /// String describing the source location. The string is composed of 496 /// semi-colon separated fields which describe the source file, the function 497 /// and a pair of line numbers that delimit the construct. 498 IdentField_PSource 499 }; 500 501 /// Schedule types for 'omp for' loops (these enumerators are taken from 502 /// the enum sched_type in kmp.h). 503 enum OpenMPSchedType { 504 /// Lower bound for default (unordered) versions. 505 OMP_sch_lower = 32, 506 OMP_sch_static_chunked = 33, 507 OMP_sch_static = 34, 508 OMP_sch_dynamic_chunked = 35, 509 OMP_sch_guided_chunked = 36, 510 OMP_sch_runtime = 37, 511 OMP_sch_auto = 38, 512 /// static with chunk adjustment (e.g., simd) 513 OMP_sch_static_balanced_chunked = 45, 514 /// Lower bound for 'ordered' versions. 515 OMP_ord_lower = 64, 516 OMP_ord_static_chunked = 65, 517 OMP_ord_static = 66, 518 OMP_ord_dynamic_chunked = 67, 519 OMP_ord_guided_chunked = 68, 520 OMP_ord_runtime = 69, 521 OMP_ord_auto = 70, 522 OMP_sch_default = OMP_sch_static, 523 /// dist_schedule types 524 OMP_dist_sch_static_chunked = 91, 525 OMP_dist_sch_static = 92, 526 /// Support for OpenMP 4.5 monotonic and nonmonotonic schedule modifiers. 527 /// Set if the monotonic schedule modifier was present. 528 OMP_sch_modifier_monotonic = (1 << 29), 529 /// Set if the nonmonotonic schedule modifier was present. 530 OMP_sch_modifier_nonmonotonic = (1 << 30), 531 }; 532 533 enum OpenMPRTLFunction { 534 /// Call to void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, 535 /// kmpc_micro microtask, ...); 536 OMPRTL__kmpc_fork_call, 537 /// Call to void *__kmpc_threadprivate_cached(ident_t *loc, 538 /// kmp_int32 global_tid, void *data, size_t size, void ***cache); 539 OMPRTL__kmpc_threadprivate_cached, 540 /// Call to void __kmpc_threadprivate_register( ident_t *, 541 /// void *data, kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor); 542 OMPRTL__kmpc_threadprivate_register, 543 // Call to __kmpc_int32 kmpc_global_thread_num(ident_t *loc); 544 OMPRTL__kmpc_global_thread_num, 545 // Call to void __kmpc_critical(ident_t *loc, kmp_int32 global_tid, 546 // kmp_critical_name *crit); 547 OMPRTL__kmpc_critical, 548 // Call to void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 549 // global_tid, kmp_critical_name *crit, uintptr_t hint); 550 OMPRTL__kmpc_critical_with_hint, 551 // Call to void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid, 552 // kmp_critical_name *crit); 553 OMPRTL__kmpc_end_critical, 554 // Call to kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32 555 // global_tid); 556 OMPRTL__kmpc_cancel_barrier, 557 // Call to void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid); 558 OMPRTL__kmpc_barrier, 559 // Call to void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid); 560 OMPRTL__kmpc_for_static_fini, 561 // Call to void __kmpc_serialized_parallel(ident_t *loc, kmp_int32 562 // global_tid); 563 OMPRTL__kmpc_serialized_parallel, 564 // Call to void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32 565 // global_tid); 566 OMPRTL__kmpc_end_serialized_parallel, 567 // Call to void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid, 568 // kmp_int32 num_threads); 569 OMPRTL__kmpc_push_num_threads, 570 // Call to void __kmpc_flush(ident_t *loc); 571 OMPRTL__kmpc_flush, 572 // Call to kmp_int32 __kmpc_master(ident_t *, kmp_int32 global_tid); 573 OMPRTL__kmpc_master, 574 // Call to void __kmpc_end_master(ident_t *, kmp_int32 global_tid); 575 OMPRTL__kmpc_end_master, 576 // Call to kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid, 577 // int end_part); 578 OMPRTL__kmpc_omp_taskyield, 579 // Call to kmp_int32 __kmpc_single(ident_t *, kmp_int32 global_tid); 580 OMPRTL__kmpc_single, 581 // Call to void __kmpc_end_single(ident_t *, kmp_int32 global_tid); 582 OMPRTL__kmpc_end_single, 583 // Call to kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 584 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 585 // kmp_routine_entry_t *task_entry); 586 OMPRTL__kmpc_omp_task_alloc, 587 // Call to kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t * 588 // new_task); 589 OMPRTL__kmpc_omp_task, 590 // Call to void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid, 591 // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *), 592 // kmp_int32 didit); 593 OMPRTL__kmpc_copyprivate, 594 // Call to kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid, 595 // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void 596 // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck); 597 OMPRTL__kmpc_reduce, 598 // Call to kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32 599 // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, 600 // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name 601 // *lck); 602 OMPRTL__kmpc_reduce_nowait, 603 // Call to void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid, 604 // kmp_critical_name *lck); 605 OMPRTL__kmpc_end_reduce, 606 // Call to void __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid, 607 // kmp_critical_name *lck); 608 OMPRTL__kmpc_end_reduce_nowait, 609 // Call to void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 610 // kmp_task_t * new_task); 611 OMPRTL__kmpc_omp_task_begin_if0, 612 // Call to void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 613 // kmp_task_t * new_task); 614 OMPRTL__kmpc_omp_task_complete_if0, 615 // Call to void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid); 616 OMPRTL__kmpc_ordered, 617 // Call to void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid); 618 OMPRTL__kmpc_end_ordered, 619 // Call to kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 620 // global_tid); 621 OMPRTL__kmpc_omp_taskwait, 622 // Call to void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid); 623 OMPRTL__kmpc_taskgroup, 624 // Call to void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid); 625 OMPRTL__kmpc_end_taskgroup, 626 // Call to void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid, 627 // int proc_bind); 628 OMPRTL__kmpc_push_proc_bind, 629 // Call to kmp_int32 __kmpc_omp_task_with_deps(ident_t *loc_ref, kmp_int32 630 // gtid, kmp_task_t * new_task, kmp_int32 ndeps, kmp_depend_info_t 631 // *dep_list, kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 632 OMPRTL__kmpc_omp_task_with_deps, 633 // Call to void __kmpc_omp_wait_deps(ident_t *loc_ref, kmp_int32 634 // gtid, kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 635 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 636 OMPRTL__kmpc_omp_wait_deps, 637 // Call to kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 638 // global_tid, kmp_int32 cncl_kind); 639 OMPRTL__kmpc_cancellationpoint, 640 // Call to kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 641 // kmp_int32 cncl_kind); 642 OMPRTL__kmpc_cancel, 643 // Call to void __kmpc_push_num_teams(ident_t *loc, kmp_int32 global_tid, 644 // kmp_int32 num_teams, kmp_int32 thread_limit); 645 OMPRTL__kmpc_push_num_teams, 646 // Call to void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro 647 // microtask, ...); 648 OMPRTL__kmpc_fork_teams, 649 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 650 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 651 // sched, kmp_uint64 grainsize, void *task_dup); 652 OMPRTL__kmpc_taskloop, 653 // Call to void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32 654 // num_dims, struct kmp_dim *dims); 655 OMPRTL__kmpc_doacross_init, 656 // Call to void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid); 657 OMPRTL__kmpc_doacross_fini, 658 // Call to void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64 659 // *vec); 660 OMPRTL__kmpc_doacross_post, 661 // Call to void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64 662 // *vec); 663 OMPRTL__kmpc_doacross_wait, 664 // Call to void *__kmpc_task_reduction_init(int gtid, int num_data, void 665 // *data); 666 OMPRTL__kmpc_task_reduction_init, 667 // Call to void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 668 // *d); 669 OMPRTL__kmpc_task_reduction_get_th_data, 670 671 // 672 // Offloading related calls 673 // 674 // Call to void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64 675 // size); 676 OMPRTL__kmpc_push_target_tripcount, 677 // Call to int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t 678 // arg_num, void** args_base, void **args, size_t *arg_sizes, int64_t 679 // *arg_types); 680 OMPRTL__tgt_target, 681 // Call to int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr, 682 // int32_t arg_num, void** args_base, void **args, size_t *arg_sizes, int64_t 683 // *arg_types); 684 OMPRTL__tgt_target_nowait, 685 // Call to int32_t __tgt_target_teams(int64_t device_id, void *host_ptr, 686 // int32_t arg_num, void** args_base, void **args, size_t *arg_sizes, int64_t 687 // *arg_types, int32_t num_teams, int32_t thread_limit); 688 OMPRTL__tgt_target_teams, 689 // Call to int32_t __tgt_target_teams_nowait(int64_t device_id, void 690 // *host_ptr, int32_t arg_num, void** args_base, void **args, size_t 691 // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 692 OMPRTL__tgt_target_teams_nowait, 693 // Call to void __tgt_register_lib(__tgt_bin_desc *desc); 694 OMPRTL__tgt_register_lib, 695 // Call to void __tgt_unregister_lib(__tgt_bin_desc *desc); 696 OMPRTL__tgt_unregister_lib, 697 // Call to void __tgt_target_data_begin(int64_t device_id, int32_t arg_num, 698 // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types); 699 OMPRTL__tgt_target_data_begin, 700 // Call to void __tgt_target_data_begin_nowait(int64_t device_id, int32_t 701 // arg_num, void** args_base, void **args, size_t *arg_sizes, int64_t 702 // *arg_types); 703 OMPRTL__tgt_target_data_begin_nowait, 704 // Call to void __tgt_target_data_end(int64_t device_id, int32_t arg_num, 705 // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types); 706 OMPRTL__tgt_target_data_end, 707 // Call to void __tgt_target_data_end_nowait(int64_t device_id, int32_t 708 // arg_num, void** args_base, void **args, size_t *arg_sizes, int64_t 709 // *arg_types); 710 OMPRTL__tgt_target_data_end_nowait, 711 // Call to void __tgt_target_data_update(int64_t device_id, int32_t arg_num, 712 // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types); 713 OMPRTL__tgt_target_data_update, 714 // Call to void __tgt_target_data_update_nowait(int64_t device_id, int32_t 715 // arg_num, void** args_base, void **args, size_t *arg_sizes, int64_t 716 // *arg_types); 717 OMPRTL__tgt_target_data_update_nowait, 718 }; 719 720 /// A basic class for pre|post-action for advanced codegen sequence for OpenMP 721 /// region. 722 class CleanupTy final : public EHScopeStack::Cleanup { 723 PrePostActionTy *Action; 724 725 public: 726 explicit CleanupTy(PrePostActionTy *Action) : Action(Action) {} 727 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 728 if (!CGF.HaveInsertPoint()) 729 return; 730 Action->Exit(CGF); 731 } 732 }; 733 734 } // anonymous namespace 735 736 void RegionCodeGenTy::operator()(CodeGenFunction &CGF) const { 737 CodeGenFunction::RunCleanupsScope Scope(CGF); 738 if (PrePostAction) { 739 CGF.EHStack.pushCleanup<CleanupTy>(NormalAndEHCleanup, PrePostAction); 740 Callback(CodeGen, CGF, *PrePostAction); 741 } else { 742 PrePostActionTy Action; 743 Callback(CodeGen, CGF, Action); 744 } 745 } 746 747 /// Check if the combiner is a call to UDR combiner and if it is so return the 748 /// UDR decl used for reduction. 749 static const OMPDeclareReductionDecl * 750 getReductionInit(const Expr *ReductionOp) { 751 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 752 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 753 if (const auto *DRE = 754 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 755 if (const auto *DRD = dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) 756 return DRD; 757 return nullptr; 758 } 759 760 static void emitInitWithReductionInitializer(CodeGenFunction &CGF, 761 const OMPDeclareReductionDecl *DRD, 762 const Expr *InitOp, 763 Address Private, Address Original, 764 QualType Ty) { 765 if (DRD->getInitializer()) { 766 std::pair<llvm::Function *, llvm::Function *> Reduction = 767 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 768 const auto *CE = cast<CallExpr>(InitOp); 769 const auto *OVE = cast<OpaqueValueExpr>(CE->getCallee()); 770 const Expr *LHS = CE->getArg(/*Arg=*/0)->IgnoreParenImpCasts(); 771 const Expr *RHS = CE->getArg(/*Arg=*/1)->IgnoreParenImpCasts(); 772 const auto *LHSDRE = 773 cast<DeclRefExpr>(cast<UnaryOperator>(LHS)->getSubExpr()); 774 const auto *RHSDRE = 775 cast<DeclRefExpr>(cast<UnaryOperator>(RHS)->getSubExpr()); 776 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 777 PrivateScope.addPrivate(cast<VarDecl>(LHSDRE->getDecl()), 778 [=]() { return Private; }); 779 PrivateScope.addPrivate(cast<VarDecl>(RHSDRE->getDecl()), 780 [=]() { return Original; }); 781 (void)PrivateScope.Privatize(); 782 RValue Func = RValue::get(Reduction.second); 783 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 784 CGF.EmitIgnoredExpr(InitOp); 785 } else { 786 llvm::Constant *Init = CGF.CGM.EmitNullConstant(Ty); 787 std::string Name = CGF.CGM.getOpenMPRuntime().getName({"init"}); 788 auto *GV = new llvm::GlobalVariable( 789 CGF.CGM.getModule(), Init->getType(), /*isConstant=*/true, 790 llvm::GlobalValue::PrivateLinkage, Init, Name); 791 LValue LV = CGF.MakeNaturalAlignAddrLValue(GV, Ty); 792 RValue InitRVal; 793 switch (CGF.getEvaluationKind(Ty)) { 794 case TEK_Scalar: 795 InitRVal = CGF.EmitLoadOfLValue(LV, DRD->getLocation()); 796 break; 797 case TEK_Complex: 798 InitRVal = 799 RValue::getComplex(CGF.EmitLoadOfComplex(LV, DRD->getLocation())); 800 break; 801 case TEK_Aggregate: 802 InitRVal = RValue::getAggregate(LV.getAddress()); 803 break; 804 } 805 OpaqueValueExpr OVE(DRD->getLocation(), Ty, VK_RValue); 806 CodeGenFunction::OpaqueValueMapping OpaqueMap(CGF, &OVE, InitRVal); 807 CGF.EmitAnyExprToMem(&OVE, Private, Ty.getQualifiers(), 808 /*IsInitializer=*/false); 809 } 810 } 811 812 /// Emit initialization of arrays of complex types. 813 /// \param DestAddr Address of the array. 814 /// \param Type Type of array. 815 /// \param Init Initial expression of array. 816 /// \param SrcAddr Address of the original array. 817 static void EmitOMPAggregateInit(CodeGenFunction &CGF, Address DestAddr, 818 QualType Type, bool EmitDeclareReductionInit, 819 const Expr *Init, 820 const OMPDeclareReductionDecl *DRD, 821 Address SrcAddr = Address::invalid()) { 822 // Perform element-by-element initialization. 823 QualType ElementTy; 824 825 // Drill down to the base element type on both arrays. 826 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 827 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, DestAddr); 828 DestAddr = 829 CGF.Builder.CreateElementBitCast(DestAddr, DestAddr.getElementType()); 830 if (DRD) 831 SrcAddr = 832 CGF.Builder.CreateElementBitCast(SrcAddr, DestAddr.getElementType()); 833 834 llvm::Value *SrcBegin = nullptr; 835 if (DRD) 836 SrcBegin = SrcAddr.getPointer(); 837 llvm::Value *DestBegin = DestAddr.getPointer(); 838 // Cast from pointer to array type to pointer to single element. 839 llvm::Value *DestEnd = CGF.Builder.CreateGEP(DestBegin, NumElements); 840 // The basic structure here is a while-do loop. 841 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arrayinit.body"); 842 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arrayinit.done"); 843 llvm::Value *IsEmpty = 844 CGF.Builder.CreateICmpEQ(DestBegin, DestEnd, "omp.arrayinit.isempty"); 845 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 846 847 // Enter the loop body, making that address the current address. 848 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 849 CGF.EmitBlock(BodyBB); 850 851 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 852 853 llvm::PHINode *SrcElementPHI = nullptr; 854 Address SrcElementCurrent = Address::invalid(); 855 if (DRD) { 856 SrcElementPHI = CGF.Builder.CreatePHI(SrcBegin->getType(), 2, 857 "omp.arraycpy.srcElementPast"); 858 SrcElementPHI->addIncoming(SrcBegin, EntryBB); 859 SrcElementCurrent = 860 Address(SrcElementPHI, 861 SrcAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 862 } 863 llvm::PHINode *DestElementPHI = CGF.Builder.CreatePHI( 864 DestBegin->getType(), 2, "omp.arraycpy.destElementPast"); 865 DestElementPHI->addIncoming(DestBegin, EntryBB); 866 Address DestElementCurrent = 867 Address(DestElementPHI, 868 DestAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 869 870 // Emit copy. 871 { 872 CodeGenFunction::RunCleanupsScope InitScope(CGF); 873 if (EmitDeclareReductionInit) { 874 emitInitWithReductionInitializer(CGF, DRD, Init, DestElementCurrent, 875 SrcElementCurrent, ElementTy); 876 } else 877 CGF.EmitAnyExprToMem(Init, DestElementCurrent, ElementTy.getQualifiers(), 878 /*IsInitializer=*/false); 879 } 880 881 if (DRD) { 882 // Shift the address forward by one element. 883 llvm::Value *SrcElementNext = CGF.Builder.CreateConstGEP1_32( 884 SrcElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 885 SrcElementPHI->addIncoming(SrcElementNext, CGF.Builder.GetInsertBlock()); 886 } 887 888 // Shift the address forward by one element. 889 llvm::Value *DestElementNext = CGF.Builder.CreateConstGEP1_32( 890 DestElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 891 // Check whether we've reached the end. 892 llvm::Value *Done = 893 CGF.Builder.CreateICmpEQ(DestElementNext, DestEnd, "omp.arraycpy.done"); 894 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 895 DestElementPHI->addIncoming(DestElementNext, CGF.Builder.GetInsertBlock()); 896 897 // Done. 898 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 899 } 900 901 LValue ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, const Expr *E) { 902 return CGF.EmitOMPSharedLValue(E); 903 } 904 905 LValue ReductionCodeGen::emitSharedLValueUB(CodeGenFunction &CGF, 906 const Expr *E) { 907 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(E)) 908 return CGF.EmitOMPArraySectionExpr(OASE, /*IsLowerBound=*/false); 909 return LValue(); 910 } 911 912 void ReductionCodeGen::emitAggregateInitialization( 913 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 914 const OMPDeclareReductionDecl *DRD) { 915 // Emit VarDecl with copy init for arrays. 916 // Get the address of the original variable captured in current 917 // captured region. 918 const auto *PrivateVD = 919 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 920 bool EmitDeclareReductionInit = 921 DRD && (DRD->getInitializer() || !PrivateVD->hasInit()); 922 EmitOMPAggregateInit(CGF, PrivateAddr, PrivateVD->getType(), 923 EmitDeclareReductionInit, 924 EmitDeclareReductionInit ? ClausesData[N].ReductionOp 925 : PrivateVD->getInit(), 926 DRD, SharedLVal.getAddress()); 927 } 928 929 ReductionCodeGen::ReductionCodeGen(ArrayRef<const Expr *> Shareds, 930 ArrayRef<const Expr *> Privates, 931 ArrayRef<const Expr *> ReductionOps) { 932 ClausesData.reserve(Shareds.size()); 933 SharedAddresses.reserve(Shareds.size()); 934 Sizes.reserve(Shareds.size()); 935 BaseDecls.reserve(Shareds.size()); 936 auto IPriv = Privates.begin(); 937 auto IRed = ReductionOps.begin(); 938 for (const Expr *Ref : Shareds) { 939 ClausesData.emplace_back(Ref, *IPriv, *IRed); 940 std::advance(IPriv, 1); 941 std::advance(IRed, 1); 942 } 943 } 944 945 void ReductionCodeGen::emitSharedLValue(CodeGenFunction &CGF, unsigned N) { 946 assert(SharedAddresses.size() == N && 947 "Number of generated lvalues must be exactly N."); 948 LValue First = emitSharedLValue(CGF, ClausesData[N].Ref); 949 LValue Second = emitSharedLValueUB(CGF, ClausesData[N].Ref); 950 SharedAddresses.emplace_back(First, Second); 951 } 952 953 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N) { 954 const auto *PrivateVD = 955 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 956 QualType PrivateType = PrivateVD->getType(); 957 bool AsArraySection = isa<OMPArraySectionExpr>(ClausesData[N].Ref); 958 if (!PrivateType->isVariablyModifiedType()) { 959 Sizes.emplace_back( 960 CGF.getTypeSize( 961 SharedAddresses[N].first.getType().getNonReferenceType()), 962 nullptr); 963 return; 964 } 965 llvm::Value *Size; 966 llvm::Value *SizeInChars; 967 auto *ElemType = 968 cast<llvm::PointerType>(SharedAddresses[N].first.getPointer()->getType()) 969 ->getElementType(); 970 auto *ElemSizeOf = llvm::ConstantExpr::getSizeOf(ElemType); 971 if (AsArraySection) { 972 Size = CGF.Builder.CreatePtrDiff(SharedAddresses[N].second.getPointer(), 973 SharedAddresses[N].first.getPointer()); 974 Size = CGF.Builder.CreateNUWAdd( 975 Size, llvm::ConstantInt::get(Size->getType(), /*V=*/1)); 976 SizeInChars = CGF.Builder.CreateNUWMul(Size, ElemSizeOf); 977 } else { 978 SizeInChars = CGF.getTypeSize( 979 SharedAddresses[N].first.getType().getNonReferenceType()); 980 Size = CGF.Builder.CreateExactUDiv(SizeInChars, ElemSizeOf); 981 } 982 Sizes.emplace_back(SizeInChars, Size); 983 CodeGenFunction::OpaqueValueMapping OpaqueMap( 984 CGF, 985 cast<OpaqueValueExpr>( 986 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 987 RValue::get(Size)); 988 CGF.EmitVariablyModifiedType(PrivateType); 989 } 990 991 void ReductionCodeGen::emitAggregateType(CodeGenFunction &CGF, unsigned N, 992 llvm::Value *Size) { 993 const auto *PrivateVD = 994 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 995 QualType PrivateType = PrivateVD->getType(); 996 if (!PrivateType->isVariablyModifiedType()) { 997 assert(!Size && !Sizes[N].second && 998 "Size should be nullptr for non-variably modified reduction " 999 "items."); 1000 return; 1001 } 1002 CodeGenFunction::OpaqueValueMapping OpaqueMap( 1003 CGF, 1004 cast<OpaqueValueExpr>( 1005 CGF.getContext().getAsVariableArrayType(PrivateType)->getSizeExpr()), 1006 RValue::get(Size)); 1007 CGF.EmitVariablyModifiedType(PrivateType); 1008 } 1009 1010 void ReductionCodeGen::emitInitialization( 1011 CodeGenFunction &CGF, unsigned N, Address PrivateAddr, LValue SharedLVal, 1012 llvm::function_ref<bool(CodeGenFunction &)> DefaultInit) { 1013 assert(SharedAddresses.size() > N && "No variable was generated"); 1014 const auto *PrivateVD = 1015 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1016 const OMPDeclareReductionDecl *DRD = 1017 getReductionInit(ClausesData[N].ReductionOp); 1018 QualType PrivateType = PrivateVD->getType(); 1019 PrivateAddr = CGF.Builder.CreateElementBitCast( 1020 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 1021 QualType SharedType = SharedAddresses[N].first.getType(); 1022 SharedLVal = CGF.MakeAddrLValue( 1023 CGF.Builder.CreateElementBitCast(SharedLVal.getAddress(), 1024 CGF.ConvertTypeForMem(SharedType)), 1025 SharedType, SharedAddresses[N].first.getBaseInfo(), 1026 CGF.CGM.getTBAAInfoForSubobject(SharedAddresses[N].first, SharedType)); 1027 if (CGF.getContext().getAsArrayType(PrivateVD->getType())) { 1028 emitAggregateInitialization(CGF, N, PrivateAddr, SharedLVal, DRD); 1029 } else if (DRD && (DRD->getInitializer() || !PrivateVD->hasInit())) { 1030 emitInitWithReductionInitializer(CGF, DRD, ClausesData[N].ReductionOp, 1031 PrivateAddr, SharedLVal.getAddress(), 1032 SharedLVal.getType()); 1033 } else if (!DefaultInit(CGF) && PrivateVD->hasInit() && 1034 !CGF.isTrivialInitializer(PrivateVD->getInit())) { 1035 CGF.EmitAnyExprToMem(PrivateVD->getInit(), PrivateAddr, 1036 PrivateVD->getType().getQualifiers(), 1037 /*IsInitializer=*/false); 1038 } 1039 } 1040 1041 bool ReductionCodeGen::needCleanups(unsigned N) { 1042 const auto *PrivateVD = 1043 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1044 QualType PrivateType = PrivateVD->getType(); 1045 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 1046 return DTorKind != QualType::DK_none; 1047 } 1048 1049 void ReductionCodeGen::emitCleanups(CodeGenFunction &CGF, unsigned N, 1050 Address PrivateAddr) { 1051 const auto *PrivateVD = 1052 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Private)->getDecl()); 1053 QualType PrivateType = PrivateVD->getType(); 1054 QualType::DestructionKind DTorKind = PrivateType.isDestructedType(); 1055 if (needCleanups(N)) { 1056 PrivateAddr = CGF.Builder.CreateElementBitCast( 1057 PrivateAddr, CGF.ConvertTypeForMem(PrivateType)); 1058 CGF.pushDestroy(DTorKind, PrivateAddr, PrivateType); 1059 } 1060 } 1061 1062 static LValue loadToBegin(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 1063 LValue BaseLV) { 1064 BaseTy = BaseTy.getNonReferenceType(); 1065 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 1066 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 1067 if (const auto *PtrTy = BaseTy->getAs<PointerType>()) { 1068 BaseLV = CGF.EmitLoadOfPointerLValue(BaseLV.getAddress(), PtrTy); 1069 } else { 1070 LValue RefLVal = CGF.MakeAddrLValue(BaseLV.getAddress(), BaseTy); 1071 BaseLV = CGF.EmitLoadOfReferenceLValue(RefLVal); 1072 } 1073 BaseTy = BaseTy->getPointeeType(); 1074 } 1075 return CGF.MakeAddrLValue( 1076 CGF.Builder.CreateElementBitCast(BaseLV.getAddress(), 1077 CGF.ConvertTypeForMem(ElTy)), 1078 BaseLV.getType(), BaseLV.getBaseInfo(), 1079 CGF.CGM.getTBAAInfoForSubobject(BaseLV, BaseLV.getType())); 1080 } 1081 1082 static Address castToBase(CodeGenFunction &CGF, QualType BaseTy, QualType ElTy, 1083 llvm::Type *BaseLVType, CharUnits BaseLVAlignment, 1084 llvm::Value *Addr) { 1085 Address Tmp = Address::invalid(); 1086 Address TopTmp = Address::invalid(); 1087 Address MostTopTmp = Address::invalid(); 1088 BaseTy = BaseTy.getNonReferenceType(); 1089 while ((BaseTy->isPointerType() || BaseTy->isReferenceType()) && 1090 !CGF.getContext().hasSameType(BaseTy, ElTy)) { 1091 Tmp = CGF.CreateMemTemp(BaseTy); 1092 if (TopTmp.isValid()) 1093 CGF.Builder.CreateStore(Tmp.getPointer(), TopTmp); 1094 else 1095 MostTopTmp = Tmp; 1096 TopTmp = Tmp; 1097 BaseTy = BaseTy->getPointeeType(); 1098 } 1099 llvm::Type *Ty = BaseLVType; 1100 if (Tmp.isValid()) 1101 Ty = Tmp.getElementType(); 1102 Addr = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(Addr, Ty); 1103 if (Tmp.isValid()) { 1104 CGF.Builder.CreateStore(Addr, Tmp); 1105 return MostTopTmp; 1106 } 1107 return Address(Addr, BaseLVAlignment); 1108 } 1109 1110 static const VarDecl *getBaseDecl(const Expr *Ref, const DeclRefExpr *&DE) { 1111 const VarDecl *OrigVD = nullptr; 1112 if (const auto *OASE = dyn_cast<OMPArraySectionExpr>(Ref)) { 1113 const Expr *Base = OASE->getBase()->IgnoreParenImpCasts(); 1114 while (const auto *TempOASE = dyn_cast<OMPArraySectionExpr>(Base)) 1115 Base = TempOASE->getBase()->IgnoreParenImpCasts(); 1116 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 1117 Base = TempASE->getBase()->IgnoreParenImpCasts(); 1118 DE = cast<DeclRefExpr>(Base); 1119 OrigVD = cast<VarDecl>(DE->getDecl()); 1120 } else if (const auto *ASE = dyn_cast<ArraySubscriptExpr>(Ref)) { 1121 const Expr *Base = ASE->getBase()->IgnoreParenImpCasts(); 1122 while (const auto *TempASE = dyn_cast<ArraySubscriptExpr>(Base)) 1123 Base = TempASE->getBase()->IgnoreParenImpCasts(); 1124 DE = cast<DeclRefExpr>(Base); 1125 OrigVD = cast<VarDecl>(DE->getDecl()); 1126 } 1127 return OrigVD; 1128 } 1129 1130 Address ReductionCodeGen::adjustPrivateAddress(CodeGenFunction &CGF, unsigned N, 1131 Address PrivateAddr) { 1132 const DeclRefExpr *DE; 1133 if (const VarDecl *OrigVD = ::getBaseDecl(ClausesData[N].Ref, DE)) { 1134 BaseDecls.emplace_back(OrigVD); 1135 LValue OriginalBaseLValue = CGF.EmitLValue(DE); 1136 LValue BaseLValue = 1137 loadToBegin(CGF, OrigVD->getType(), SharedAddresses[N].first.getType(), 1138 OriginalBaseLValue); 1139 llvm::Value *Adjustment = CGF.Builder.CreatePtrDiff( 1140 BaseLValue.getPointer(), SharedAddresses[N].first.getPointer()); 1141 llvm::Value *PrivatePointer = 1142 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 1143 PrivateAddr.getPointer(), 1144 SharedAddresses[N].first.getAddress().getType()); 1145 llvm::Value *Ptr = CGF.Builder.CreateGEP(PrivatePointer, Adjustment); 1146 return castToBase(CGF, OrigVD->getType(), 1147 SharedAddresses[N].first.getType(), 1148 OriginalBaseLValue.getAddress().getType(), 1149 OriginalBaseLValue.getAlignment(), Ptr); 1150 } 1151 BaseDecls.emplace_back( 1152 cast<VarDecl>(cast<DeclRefExpr>(ClausesData[N].Ref)->getDecl())); 1153 return PrivateAddr; 1154 } 1155 1156 bool ReductionCodeGen::usesReductionInitializer(unsigned N) const { 1157 const OMPDeclareReductionDecl *DRD = 1158 getReductionInit(ClausesData[N].ReductionOp); 1159 return DRD && DRD->getInitializer(); 1160 } 1161 1162 LValue CGOpenMPRegionInfo::getThreadIDVariableLValue(CodeGenFunction &CGF) { 1163 return CGF.EmitLoadOfPointerLValue( 1164 CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1165 getThreadIDVariable()->getType()->castAs<PointerType>()); 1166 } 1167 1168 void CGOpenMPRegionInfo::EmitBody(CodeGenFunction &CGF, const Stmt * /*S*/) { 1169 if (!CGF.HaveInsertPoint()) 1170 return; 1171 // 1.2.2 OpenMP Language Terminology 1172 // Structured block - An executable statement with a single entry at the 1173 // top and a single exit at the bottom. 1174 // The point of exit cannot be a branch out of the structured block. 1175 // longjmp() and throw() must not violate the entry/exit criteria. 1176 CGF.EHStack.pushTerminate(); 1177 CodeGen(CGF); 1178 CGF.EHStack.popTerminate(); 1179 } 1180 1181 LValue CGOpenMPTaskOutlinedRegionInfo::getThreadIDVariableLValue( 1182 CodeGenFunction &CGF) { 1183 return CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(getThreadIDVariable()), 1184 getThreadIDVariable()->getType(), 1185 AlignmentSource::Decl); 1186 } 1187 1188 static FieldDecl *addFieldToRecordDecl(ASTContext &C, DeclContext *DC, 1189 QualType FieldTy) { 1190 auto *Field = FieldDecl::Create( 1191 C, DC, SourceLocation(), SourceLocation(), /*Id=*/nullptr, FieldTy, 1192 C.getTrivialTypeSourceInfo(FieldTy, SourceLocation()), 1193 /*BW=*/nullptr, /*Mutable=*/false, /*InitStyle=*/ICIS_NoInit); 1194 Field->setAccess(AS_public); 1195 DC->addDecl(Field); 1196 return Field; 1197 } 1198 1199 CGOpenMPRuntime::CGOpenMPRuntime(CodeGenModule &CGM, StringRef FirstSeparator, 1200 StringRef Separator) 1201 : CGM(CGM), FirstSeparator(FirstSeparator), Separator(Separator), 1202 OffloadEntriesInfoManager(CGM) { 1203 ASTContext &C = CGM.getContext(); 1204 RecordDecl *RD = C.buildImplicitRecord("ident_t"); 1205 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 1206 RD->startDefinition(); 1207 // reserved_1 1208 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1209 // flags 1210 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1211 // reserved_2 1212 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1213 // reserved_3 1214 addFieldToRecordDecl(C, RD, KmpInt32Ty); 1215 // psource 1216 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 1217 RD->completeDefinition(); 1218 IdentQTy = C.getRecordType(RD); 1219 IdentTy = CGM.getTypes().ConvertRecordDeclType(RD); 1220 KmpCriticalNameTy = llvm::ArrayType::get(CGM.Int32Ty, /*NumElements*/ 8); 1221 1222 loadOffloadInfoMetadata(); 1223 } 1224 1225 void CGOpenMPRuntime::clear() { 1226 InternalVars.clear(); 1227 // Clean non-target variable declarations possibly used only in debug info. 1228 for (const auto &Data : EmittedNonTargetVariables) { 1229 if (!Data.getValue().pointsToAliveValue()) 1230 continue; 1231 auto *GV = dyn_cast<llvm::GlobalVariable>(Data.getValue()); 1232 if (!GV) 1233 continue; 1234 if (!GV->isDeclaration() || GV->getNumUses() > 0) 1235 continue; 1236 GV->eraseFromParent(); 1237 } 1238 } 1239 1240 std::string CGOpenMPRuntime::getName(ArrayRef<StringRef> Parts) const { 1241 SmallString<128> Buffer; 1242 llvm::raw_svector_ostream OS(Buffer); 1243 StringRef Sep = FirstSeparator; 1244 for (StringRef Part : Parts) { 1245 OS << Sep << Part; 1246 Sep = Separator; 1247 } 1248 return OS.str(); 1249 } 1250 1251 static llvm::Function * 1252 emitCombinerOrInitializer(CodeGenModule &CGM, QualType Ty, 1253 const Expr *CombinerInitializer, const VarDecl *In, 1254 const VarDecl *Out, bool IsCombiner) { 1255 // void .omp_combiner.(Ty *in, Ty *out); 1256 ASTContext &C = CGM.getContext(); 1257 QualType PtrTy = C.getPointerType(Ty).withRestrict(); 1258 FunctionArgList Args; 1259 ImplicitParamDecl OmpOutParm(C, /*DC=*/nullptr, Out->getLocation(), 1260 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1261 ImplicitParamDecl OmpInParm(C, /*DC=*/nullptr, In->getLocation(), 1262 /*Id=*/nullptr, PtrTy, ImplicitParamDecl::Other); 1263 Args.push_back(&OmpOutParm); 1264 Args.push_back(&OmpInParm); 1265 const CGFunctionInfo &FnInfo = 1266 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 1267 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 1268 std::string Name = CGM.getOpenMPRuntime().getName( 1269 {IsCombiner ? "omp_combiner" : "omp_initializer", ""}); 1270 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 1271 Name, &CGM.getModule()); 1272 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 1273 Fn->removeFnAttr(llvm::Attribute::NoInline); 1274 Fn->removeFnAttr(llvm::Attribute::OptimizeNone); 1275 Fn->addFnAttr(llvm::Attribute::AlwaysInline); 1276 CodeGenFunction CGF(CGM); 1277 // Map "T omp_in;" variable to "*omp_in_parm" value in all expressions. 1278 // Map "T omp_out;" variable to "*omp_out_parm" value in all expressions. 1279 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, In->getLocation(), 1280 Out->getLocation()); 1281 CodeGenFunction::OMPPrivateScope Scope(CGF); 1282 Address AddrIn = CGF.GetAddrOfLocalVar(&OmpInParm); 1283 Scope.addPrivate(In, [&CGF, AddrIn, PtrTy]() { 1284 return CGF.EmitLoadOfPointerLValue(AddrIn, PtrTy->castAs<PointerType>()) 1285 .getAddress(); 1286 }); 1287 Address AddrOut = CGF.GetAddrOfLocalVar(&OmpOutParm); 1288 Scope.addPrivate(Out, [&CGF, AddrOut, PtrTy]() { 1289 return CGF.EmitLoadOfPointerLValue(AddrOut, PtrTy->castAs<PointerType>()) 1290 .getAddress(); 1291 }); 1292 (void)Scope.Privatize(); 1293 if (!IsCombiner && Out->hasInit() && 1294 !CGF.isTrivialInitializer(Out->getInit())) { 1295 CGF.EmitAnyExprToMem(Out->getInit(), CGF.GetAddrOfLocalVar(Out), 1296 Out->getType().getQualifiers(), 1297 /*IsInitializer=*/true); 1298 } 1299 if (CombinerInitializer) 1300 CGF.EmitIgnoredExpr(CombinerInitializer); 1301 Scope.ForceCleanup(); 1302 CGF.FinishFunction(); 1303 return Fn; 1304 } 1305 1306 void CGOpenMPRuntime::emitUserDefinedReduction( 1307 CodeGenFunction *CGF, const OMPDeclareReductionDecl *D) { 1308 if (UDRMap.count(D) > 0) 1309 return; 1310 llvm::Function *Combiner = emitCombinerOrInitializer( 1311 CGM, D->getType(), D->getCombiner(), 1312 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerIn())->getDecl()), 1313 cast<VarDecl>(cast<DeclRefExpr>(D->getCombinerOut())->getDecl()), 1314 /*IsCombiner=*/true); 1315 llvm::Function *Initializer = nullptr; 1316 if (const Expr *Init = D->getInitializer()) { 1317 Initializer = emitCombinerOrInitializer( 1318 CGM, D->getType(), 1319 D->getInitializerKind() == OMPDeclareReductionDecl::CallInit ? Init 1320 : nullptr, 1321 cast<VarDecl>(cast<DeclRefExpr>(D->getInitOrig())->getDecl()), 1322 cast<VarDecl>(cast<DeclRefExpr>(D->getInitPriv())->getDecl()), 1323 /*IsCombiner=*/false); 1324 } 1325 UDRMap.try_emplace(D, Combiner, Initializer); 1326 if (CGF) { 1327 auto &Decls = FunctionUDRMap.FindAndConstruct(CGF->CurFn); 1328 Decls.second.push_back(D); 1329 } 1330 } 1331 1332 std::pair<llvm::Function *, llvm::Function *> 1333 CGOpenMPRuntime::getUserDefinedReduction(const OMPDeclareReductionDecl *D) { 1334 auto I = UDRMap.find(D); 1335 if (I != UDRMap.end()) 1336 return I->second; 1337 emitUserDefinedReduction(/*CGF=*/nullptr, D); 1338 return UDRMap.lookup(D); 1339 } 1340 1341 static llvm::Function *emitParallelOrTeamsOutlinedFunction( 1342 CodeGenModule &CGM, const OMPExecutableDirective &D, const CapturedStmt *CS, 1343 const VarDecl *ThreadIDVar, OpenMPDirectiveKind InnermostKind, 1344 const StringRef OutlinedHelperName, const RegionCodeGenTy &CodeGen) { 1345 assert(ThreadIDVar->getType()->isPointerType() && 1346 "thread id variable must be of type kmp_int32 *"); 1347 CodeGenFunction CGF(CGM, true); 1348 bool HasCancel = false; 1349 if (const auto *OPD = dyn_cast<OMPParallelDirective>(&D)) 1350 HasCancel = OPD->hasCancel(); 1351 else if (const auto *OPSD = dyn_cast<OMPParallelSectionsDirective>(&D)) 1352 HasCancel = OPSD->hasCancel(); 1353 else if (const auto *OPFD = dyn_cast<OMPParallelForDirective>(&D)) 1354 HasCancel = OPFD->hasCancel(); 1355 else if (const auto *OPFD = dyn_cast<OMPTargetParallelForDirective>(&D)) 1356 HasCancel = OPFD->hasCancel(); 1357 else if (const auto *OPFD = dyn_cast<OMPDistributeParallelForDirective>(&D)) 1358 HasCancel = OPFD->hasCancel(); 1359 else if (const auto *OPFD = 1360 dyn_cast<OMPTeamsDistributeParallelForDirective>(&D)) 1361 HasCancel = OPFD->hasCancel(); 1362 else if (const auto *OPFD = 1363 dyn_cast<OMPTargetTeamsDistributeParallelForDirective>(&D)) 1364 HasCancel = OPFD->hasCancel(); 1365 CGOpenMPOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, InnermostKind, 1366 HasCancel, OutlinedHelperName); 1367 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1368 return CGF.GenerateOpenMPCapturedStmtFunction(*CS); 1369 } 1370 1371 llvm::Function *CGOpenMPRuntime::emitParallelOutlinedFunction( 1372 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1373 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1374 const CapturedStmt *CS = D.getCapturedStmt(OMPD_parallel); 1375 return emitParallelOrTeamsOutlinedFunction( 1376 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1377 } 1378 1379 llvm::Function *CGOpenMPRuntime::emitTeamsOutlinedFunction( 1380 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1381 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 1382 const CapturedStmt *CS = D.getCapturedStmt(OMPD_teams); 1383 return emitParallelOrTeamsOutlinedFunction( 1384 CGM, D, CS, ThreadIDVar, InnermostKind, getOutlinedHelperName(), CodeGen); 1385 } 1386 1387 llvm::Function *CGOpenMPRuntime::emitTaskOutlinedFunction( 1388 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 1389 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 1390 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 1391 bool Tied, unsigned &NumberOfParts) { 1392 auto &&UntiedCodeGen = [this, &D, TaskTVar](CodeGenFunction &CGF, 1393 PrePostActionTy &) { 1394 llvm::Value *ThreadID = getThreadID(CGF, D.getBeginLoc()); 1395 llvm::Value *UpLoc = emitUpdateLocation(CGF, D.getBeginLoc()); 1396 llvm::Value *TaskArgs[] = { 1397 UpLoc, ThreadID, 1398 CGF.EmitLoadOfPointerLValue(CGF.GetAddrOfLocalVar(TaskTVar), 1399 TaskTVar->getType()->castAs<PointerType>()) 1400 .getPointer()}; 1401 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), TaskArgs); 1402 }; 1403 CGOpenMPTaskOutlinedRegionInfo::UntiedTaskActionTy Action(Tied, PartIDVar, 1404 UntiedCodeGen); 1405 CodeGen.setAction(Action); 1406 assert(!ThreadIDVar->getType()->isPointerType() && 1407 "thread id variable must be of type kmp_int32 for tasks"); 1408 const OpenMPDirectiveKind Region = 1409 isOpenMPTaskLoopDirective(D.getDirectiveKind()) ? OMPD_taskloop 1410 : OMPD_task; 1411 const CapturedStmt *CS = D.getCapturedStmt(Region); 1412 const auto *TD = dyn_cast<OMPTaskDirective>(&D); 1413 CodeGenFunction CGF(CGM, true); 1414 CGOpenMPTaskOutlinedRegionInfo CGInfo(*CS, ThreadIDVar, CodeGen, 1415 InnermostKind, 1416 TD ? TD->hasCancel() : false, Action); 1417 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 1418 llvm::Function *Res = CGF.GenerateCapturedStmtFunction(*CS); 1419 if (!Tied) 1420 NumberOfParts = Action.getNumberOfParts(); 1421 return Res; 1422 } 1423 1424 static void buildStructValue(ConstantStructBuilder &Fields, CodeGenModule &CGM, 1425 const RecordDecl *RD, const CGRecordLayout &RL, 1426 ArrayRef<llvm::Constant *> Data) { 1427 llvm::StructType *StructTy = RL.getLLVMType(); 1428 unsigned PrevIdx = 0; 1429 ConstantInitBuilder CIBuilder(CGM); 1430 auto DI = Data.begin(); 1431 for (const FieldDecl *FD : RD->fields()) { 1432 unsigned Idx = RL.getLLVMFieldNo(FD); 1433 // Fill the alignment. 1434 for (unsigned I = PrevIdx; I < Idx; ++I) 1435 Fields.add(llvm::Constant::getNullValue(StructTy->getElementType(I))); 1436 PrevIdx = Idx + 1; 1437 Fields.add(*DI); 1438 ++DI; 1439 } 1440 } 1441 1442 template <class... As> 1443 static llvm::GlobalVariable * 1444 createGlobalStruct(CodeGenModule &CGM, QualType Ty, bool IsConstant, 1445 ArrayRef<llvm::Constant *> Data, const Twine &Name, 1446 As &&... Args) { 1447 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1448 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1449 ConstantInitBuilder CIBuilder(CGM); 1450 ConstantStructBuilder Fields = CIBuilder.beginStruct(RL.getLLVMType()); 1451 buildStructValue(Fields, CGM, RD, RL, Data); 1452 return Fields.finishAndCreateGlobal( 1453 Name, CGM.getContext().getAlignOfGlobalVarInChars(Ty), IsConstant, 1454 std::forward<As>(Args)...); 1455 } 1456 1457 template <typename T> 1458 static void 1459 createConstantGlobalStructAndAddToParent(CodeGenModule &CGM, QualType Ty, 1460 ArrayRef<llvm::Constant *> Data, 1461 T &Parent) { 1462 const auto *RD = cast<RecordDecl>(Ty->getAsTagDecl()); 1463 const CGRecordLayout &RL = CGM.getTypes().getCGRecordLayout(RD); 1464 ConstantStructBuilder Fields = Parent.beginStruct(RL.getLLVMType()); 1465 buildStructValue(Fields, CGM, RD, RL, Data); 1466 Fields.finishAndAddTo(Parent); 1467 } 1468 1469 Address CGOpenMPRuntime::getOrCreateDefaultLocation(unsigned Flags) { 1470 CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy); 1471 unsigned Reserved2Flags = getDefaultLocationReserved2Flags(); 1472 FlagsTy FlagsKey(Flags, Reserved2Flags); 1473 llvm::Value *Entry = OpenMPDefaultLocMap.lookup(FlagsKey); 1474 if (!Entry) { 1475 if (!DefaultOpenMPPSource) { 1476 // Initialize default location for psource field of ident_t structure of 1477 // all ident_t objects. Format is ";file;function;line;column;;". 1478 // Taken from 1479 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp_str.cpp 1480 DefaultOpenMPPSource = 1481 CGM.GetAddrOfConstantCString(";unknown;unknown;0;0;;").getPointer(); 1482 DefaultOpenMPPSource = 1483 llvm::ConstantExpr::getBitCast(DefaultOpenMPPSource, CGM.Int8PtrTy); 1484 } 1485 1486 llvm::Constant *Data[] = { 1487 llvm::ConstantInt::getNullValue(CGM.Int32Ty), 1488 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 1489 llvm::ConstantInt::get(CGM.Int32Ty, Reserved2Flags), 1490 llvm::ConstantInt::getNullValue(CGM.Int32Ty), DefaultOpenMPPSource}; 1491 llvm::GlobalValue *DefaultOpenMPLocation = 1492 createGlobalStruct(CGM, IdentQTy, isDefaultLocationConstant(), Data, "", 1493 llvm::GlobalValue::PrivateLinkage); 1494 DefaultOpenMPLocation->setUnnamedAddr( 1495 llvm::GlobalValue::UnnamedAddr::Global); 1496 1497 OpenMPDefaultLocMap[FlagsKey] = Entry = DefaultOpenMPLocation; 1498 } 1499 return Address(Entry, Align); 1500 } 1501 1502 void CGOpenMPRuntime::setLocThreadIdInsertPt(CodeGenFunction &CGF, 1503 bool AtCurrentPoint) { 1504 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1505 assert(!Elem.second.ServiceInsertPt && "Insert point is set already."); 1506 1507 llvm::Value *Undef = llvm::UndefValue::get(CGF.Int32Ty); 1508 if (AtCurrentPoint) { 1509 Elem.second.ServiceInsertPt = new llvm::BitCastInst( 1510 Undef, CGF.Int32Ty, "svcpt", CGF.Builder.GetInsertBlock()); 1511 } else { 1512 Elem.second.ServiceInsertPt = 1513 new llvm::BitCastInst(Undef, CGF.Int32Ty, "svcpt"); 1514 Elem.second.ServiceInsertPt->insertAfter(CGF.AllocaInsertPt); 1515 } 1516 } 1517 1518 void CGOpenMPRuntime::clearLocThreadIdInsertPt(CodeGenFunction &CGF) { 1519 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1520 if (Elem.second.ServiceInsertPt) { 1521 llvm::Instruction *Ptr = Elem.second.ServiceInsertPt; 1522 Elem.second.ServiceInsertPt = nullptr; 1523 Ptr->eraseFromParent(); 1524 } 1525 } 1526 1527 llvm::Value *CGOpenMPRuntime::emitUpdateLocation(CodeGenFunction &CGF, 1528 SourceLocation Loc, 1529 unsigned Flags) { 1530 Flags |= OMP_IDENT_KMPC; 1531 // If no debug info is generated - return global default location. 1532 if (CGM.getCodeGenOpts().getDebugInfo() == codegenoptions::NoDebugInfo || 1533 Loc.isInvalid()) 1534 return getOrCreateDefaultLocation(Flags).getPointer(); 1535 1536 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1537 1538 CharUnits Align = CGM.getContext().getTypeAlignInChars(IdentQTy); 1539 Address LocValue = Address::invalid(); 1540 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1541 if (I != OpenMPLocThreadIDMap.end()) 1542 LocValue = Address(I->second.DebugLoc, Align); 1543 1544 // OpenMPLocThreadIDMap may have null DebugLoc and non-null ThreadID, if 1545 // GetOpenMPThreadID was called before this routine. 1546 if (!LocValue.isValid()) { 1547 // Generate "ident_t .kmpc_loc.addr;" 1548 Address AI = CGF.CreateMemTemp(IdentQTy, ".kmpc_loc.addr"); 1549 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1550 Elem.second.DebugLoc = AI.getPointer(); 1551 LocValue = AI; 1552 1553 if (!Elem.second.ServiceInsertPt) 1554 setLocThreadIdInsertPt(CGF); 1555 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1556 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1557 CGF.Builder.CreateMemCpy(LocValue, getOrCreateDefaultLocation(Flags), 1558 CGF.getTypeSize(IdentQTy)); 1559 } 1560 1561 // char **psource = &.kmpc_loc_<flags>.addr.psource; 1562 LValue Base = CGF.MakeAddrLValue(LocValue, IdentQTy); 1563 auto Fields = cast<RecordDecl>(IdentQTy->getAsTagDecl())->field_begin(); 1564 LValue PSource = 1565 CGF.EmitLValueForField(Base, *std::next(Fields, IdentField_PSource)); 1566 1567 llvm::Value *OMPDebugLoc = OpenMPDebugLocMap.lookup(Loc.getRawEncoding()); 1568 if (OMPDebugLoc == nullptr) { 1569 SmallString<128> Buffer2; 1570 llvm::raw_svector_ostream OS2(Buffer2); 1571 // Build debug location 1572 PresumedLoc PLoc = CGF.getContext().getSourceManager().getPresumedLoc(Loc); 1573 OS2 << ";" << PLoc.getFilename() << ";"; 1574 if (const auto *FD = dyn_cast_or_null<FunctionDecl>(CGF.CurFuncDecl)) 1575 OS2 << FD->getQualifiedNameAsString(); 1576 OS2 << ";" << PLoc.getLine() << ";" << PLoc.getColumn() << ";;"; 1577 OMPDebugLoc = CGF.Builder.CreateGlobalStringPtr(OS2.str()); 1578 OpenMPDebugLocMap[Loc.getRawEncoding()] = OMPDebugLoc; 1579 } 1580 // *psource = ";<File>;<Function>;<Line>;<Column>;;"; 1581 CGF.EmitStoreOfScalar(OMPDebugLoc, PSource); 1582 1583 // Our callers always pass this to a runtime function, so for 1584 // convenience, go ahead and return a naked pointer. 1585 return LocValue.getPointer(); 1586 } 1587 1588 llvm::Value *CGOpenMPRuntime::getThreadID(CodeGenFunction &CGF, 1589 SourceLocation Loc) { 1590 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1591 1592 llvm::Value *ThreadID = nullptr; 1593 // Check whether we've already cached a load of the thread id in this 1594 // function. 1595 auto I = OpenMPLocThreadIDMap.find(CGF.CurFn); 1596 if (I != OpenMPLocThreadIDMap.end()) { 1597 ThreadID = I->second.ThreadID; 1598 if (ThreadID != nullptr) 1599 return ThreadID; 1600 } 1601 // If exceptions are enabled, do not use parameter to avoid possible crash. 1602 if (!CGF.EHStack.requiresLandingPad() || !CGF.getLangOpts().Exceptions || 1603 !CGF.getLangOpts().CXXExceptions || 1604 CGF.Builder.GetInsertBlock() == CGF.AllocaInsertPt->getParent()) { 1605 if (auto *OMPRegionInfo = 1606 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 1607 if (OMPRegionInfo->getThreadIDVariable()) { 1608 // Check if this an outlined function with thread id passed as argument. 1609 LValue LVal = OMPRegionInfo->getThreadIDVariableLValue(CGF); 1610 ThreadID = CGF.EmitLoadOfScalar(LVal, Loc); 1611 // If value loaded in entry block, cache it and use it everywhere in 1612 // function. 1613 if (CGF.Builder.GetInsertBlock() == CGF.AllocaInsertPt->getParent()) { 1614 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1615 Elem.second.ThreadID = ThreadID; 1616 } 1617 return ThreadID; 1618 } 1619 } 1620 } 1621 1622 // This is not an outlined function region - need to call __kmpc_int32 1623 // kmpc_global_thread_num(ident_t *loc). 1624 // Generate thread id value and cache this value for use across the 1625 // function. 1626 auto &Elem = OpenMPLocThreadIDMap.FindAndConstruct(CGF.CurFn); 1627 if (!Elem.second.ServiceInsertPt) 1628 setLocThreadIdInsertPt(CGF); 1629 CGBuilderTy::InsertPointGuard IPG(CGF.Builder); 1630 CGF.Builder.SetInsertPoint(Elem.second.ServiceInsertPt); 1631 llvm::CallInst *Call = CGF.Builder.CreateCall( 1632 createRuntimeFunction(OMPRTL__kmpc_global_thread_num), 1633 emitUpdateLocation(CGF, Loc)); 1634 Call->setCallingConv(CGF.getRuntimeCC()); 1635 Elem.second.ThreadID = Call; 1636 return Call; 1637 } 1638 1639 void CGOpenMPRuntime::functionFinished(CodeGenFunction &CGF) { 1640 assert(CGF.CurFn && "No function in current CodeGenFunction."); 1641 if (OpenMPLocThreadIDMap.count(CGF.CurFn)) { 1642 clearLocThreadIdInsertPt(CGF); 1643 OpenMPLocThreadIDMap.erase(CGF.CurFn); 1644 } 1645 if (FunctionUDRMap.count(CGF.CurFn) > 0) { 1646 for(auto *D : FunctionUDRMap[CGF.CurFn]) 1647 UDRMap.erase(D); 1648 FunctionUDRMap.erase(CGF.CurFn); 1649 } 1650 } 1651 1652 llvm::Type *CGOpenMPRuntime::getIdentTyPointerTy() { 1653 return IdentTy->getPointerTo(); 1654 } 1655 1656 llvm::Type *CGOpenMPRuntime::getKmpc_MicroPointerTy() { 1657 if (!Kmpc_MicroTy) { 1658 // Build void (*kmpc_micro)(kmp_int32 *global_tid, kmp_int32 *bound_tid,...) 1659 llvm::Type *MicroParams[] = {llvm::PointerType::getUnqual(CGM.Int32Ty), 1660 llvm::PointerType::getUnqual(CGM.Int32Ty)}; 1661 Kmpc_MicroTy = llvm::FunctionType::get(CGM.VoidTy, MicroParams, true); 1662 } 1663 return llvm::PointerType::getUnqual(Kmpc_MicroTy); 1664 } 1665 1666 llvm::FunctionCallee CGOpenMPRuntime::createRuntimeFunction(unsigned Function) { 1667 llvm::FunctionCallee RTLFn = nullptr; 1668 switch (static_cast<OpenMPRTLFunction>(Function)) { 1669 case OMPRTL__kmpc_fork_call: { 1670 // Build void __kmpc_fork_call(ident_t *loc, kmp_int32 argc, kmpc_micro 1671 // microtask, ...); 1672 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1673 getKmpc_MicroPointerTy()}; 1674 auto *FnTy = 1675 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true); 1676 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_call"); 1677 if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) { 1678 if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) { 1679 llvm::LLVMContext &Ctx = F->getContext(); 1680 llvm::MDBuilder MDB(Ctx); 1681 // Annotate the callback behavior of the __kmpc_fork_call: 1682 // - The callback callee is argument number 2 (microtask). 1683 // - The first two arguments of the callback callee are unknown (-1). 1684 // - All variadic arguments to the __kmpc_fork_call are passed to the 1685 // callback callee. 1686 F->addMetadata( 1687 llvm::LLVMContext::MD_callback, 1688 *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding( 1689 2, {-1, -1}, 1690 /* VarArgsArePassed */ true)})); 1691 } 1692 } 1693 break; 1694 } 1695 case OMPRTL__kmpc_global_thread_num: { 1696 // Build kmp_int32 __kmpc_global_thread_num(ident_t *loc); 1697 llvm::Type *TypeParams[] = {getIdentTyPointerTy()}; 1698 auto *FnTy = 1699 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1700 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_global_thread_num"); 1701 break; 1702 } 1703 case OMPRTL__kmpc_threadprivate_cached: { 1704 // Build void *__kmpc_threadprivate_cached(ident_t *loc, 1705 // kmp_int32 global_tid, void *data, size_t size, void ***cache); 1706 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1707 CGM.VoidPtrTy, CGM.SizeTy, 1708 CGM.VoidPtrTy->getPointerTo()->getPointerTo()}; 1709 auto *FnTy = 1710 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg*/ false); 1711 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_cached"); 1712 break; 1713 } 1714 case OMPRTL__kmpc_critical: { 1715 // Build void __kmpc_critical(ident_t *loc, kmp_int32 global_tid, 1716 // kmp_critical_name *crit); 1717 llvm::Type *TypeParams[] = { 1718 getIdentTyPointerTy(), CGM.Int32Ty, 1719 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1720 auto *FnTy = 1721 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1722 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical"); 1723 break; 1724 } 1725 case OMPRTL__kmpc_critical_with_hint: { 1726 // Build void __kmpc_critical_with_hint(ident_t *loc, kmp_int32 global_tid, 1727 // kmp_critical_name *crit, uintptr_t hint); 1728 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1729 llvm::PointerType::getUnqual(KmpCriticalNameTy), 1730 CGM.IntPtrTy}; 1731 auto *FnTy = 1732 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1733 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_critical_with_hint"); 1734 break; 1735 } 1736 case OMPRTL__kmpc_threadprivate_register: { 1737 // Build void __kmpc_threadprivate_register(ident_t *, void *data, 1738 // kmpc_ctor ctor, kmpc_cctor cctor, kmpc_dtor dtor); 1739 // typedef void *(*kmpc_ctor)(void *); 1740 auto *KmpcCtorTy = 1741 llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 1742 /*isVarArg*/ false)->getPointerTo(); 1743 // typedef void *(*kmpc_cctor)(void *, void *); 1744 llvm::Type *KmpcCopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 1745 auto *KmpcCopyCtorTy = 1746 llvm::FunctionType::get(CGM.VoidPtrTy, KmpcCopyCtorTyArgs, 1747 /*isVarArg*/ false) 1748 ->getPointerTo(); 1749 // typedef void (*kmpc_dtor)(void *); 1750 auto *KmpcDtorTy = 1751 llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, /*isVarArg*/ false) 1752 ->getPointerTo(); 1753 llvm::Type *FnTyArgs[] = {getIdentTyPointerTy(), CGM.VoidPtrTy, KmpcCtorTy, 1754 KmpcCopyCtorTy, KmpcDtorTy}; 1755 auto *FnTy = llvm::FunctionType::get(CGM.VoidTy, FnTyArgs, 1756 /*isVarArg*/ false); 1757 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_threadprivate_register"); 1758 break; 1759 } 1760 case OMPRTL__kmpc_end_critical: { 1761 // Build void __kmpc_end_critical(ident_t *loc, kmp_int32 global_tid, 1762 // kmp_critical_name *crit); 1763 llvm::Type *TypeParams[] = { 1764 getIdentTyPointerTy(), CGM.Int32Ty, 1765 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1766 auto *FnTy = 1767 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1768 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_critical"); 1769 break; 1770 } 1771 case OMPRTL__kmpc_cancel_barrier: { 1772 // Build kmp_int32 __kmpc_cancel_barrier(ident_t *loc, kmp_int32 1773 // global_tid); 1774 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1775 auto *FnTy = 1776 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 1777 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_cancel_barrier"); 1778 break; 1779 } 1780 case OMPRTL__kmpc_barrier: { 1781 // Build void __kmpc_barrier(ident_t *loc, kmp_int32 global_tid); 1782 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1783 auto *FnTy = 1784 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1785 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name*/ "__kmpc_barrier"); 1786 break; 1787 } 1788 case OMPRTL__kmpc_for_static_fini: { 1789 // Build void __kmpc_for_static_fini(ident_t *loc, kmp_int32 global_tid); 1790 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1791 auto *FnTy = 1792 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1793 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_for_static_fini"); 1794 break; 1795 } 1796 case OMPRTL__kmpc_push_num_threads: { 1797 // Build void __kmpc_push_num_threads(ident_t *loc, kmp_int32 global_tid, 1798 // kmp_int32 num_threads) 1799 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1800 CGM.Int32Ty}; 1801 auto *FnTy = 1802 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1803 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_threads"); 1804 break; 1805 } 1806 case OMPRTL__kmpc_serialized_parallel: { 1807 // Build void __kmpc_serialized_parallel(ident_t *loc, kmp_int32 1808 // global_tid); 1809 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1810 auto *FnTy = 1811 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1812 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_serialized_parallel"); 1813 break; 1814 } 1815 case OMPRTL__kmpc_end_serialized_parallel: { 1816 // Build void __kmpc_end_serialized_parallel(ident_t *loc, kmp_int32 1817 // global_tid); 1818 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1819 auto *FnTy = 1820 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1821 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_serialized_parallel"); 1822 break; 1823 } 1824 case OMPRTL__kmpc_flush: { 1825 // Build void __kmpc_flush(ident_t *loc); 1826 llvm::Type *TypeParams[] = {getIdentTyPointerTy()}; 1827 auto *FnTy = 1828 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 1829 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_flush"); 1830 break; 1831 } 1832 case OMPRTL__kmpc_master: { 1833 // Build kmp_int32 __kmpc_master(ident_t *loc, kmp_int32 global_tid); 1834 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1835 auto *FnTy = 1836 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1837 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_master"); 1838 break; 1839 } 1840 case OMPRTL__kmpc_end_master: { 1841 // Build void __kmpc_end_master(ident_t *loc, kmp_int32 global_tid); 1842 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1843 auto *FnTy = 1844 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1845 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_master"); 1846 break; 1847 } 1848 case OMPRTL__kmpc_omp_taskyield: { 1849 // Build kmp_int32 __kmpc_omp_taskyield(ident_t *, kmp_int32 global_tid, 1850 // int end_part); 1851 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 1852 auto *FnTy = 1853 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1854 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_taskyield"); 1855 break; 1856 } 1857 case OMPRTL__kmpc_single: { 1858 // Build kmp_int32 __kmpc_single(ident_t *loc, kmp_int32 global_tid); 1859 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1860 auto *FnTy = 1861 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1862 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_single"); 1863 break; 1864 } 1865 case OMPRTL__kmpc_end_single: { 1866 // Build void __kmpc_end_single(ident_t *loc, kmp_int32 global_tid); 1867 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1868 auto *FnTy = 1869 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1870 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_single"); 1871 break; 1872 } 1873 case OMPRTL__kmpc_omp_task_alloc: { 1874 // Build kmp_task_t *__kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 1875 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 1876 // kmp_routine_entry_t *task_entry); 1877 assert(KmpRoutineEntryPtrTy != nullptr && 1878 "Type kmp_routine_entry_t must be created."); 1879 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 1880 CGM.SizeTy, CGM.SizeTy, KmpRoutineEntryPtrTy}; 1881 // Return void * and then cast to particular kmp_task_t type. 1882 auto *FnTy = 1883 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 1884 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_alloc"); 1885 break; 1886 } 1887 case OMPRTL__kmpc_omp_task: { 1888 // Build kmp_int32 __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 1889 // *new_task); 1890 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1891 CGM.VoidPtrTy}; 1892 auto *FnTy = 1893 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1894 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task"); 1895 break; 1896 } 1897 case OMPRTL__kmpc_copyprivate: { 1898 // Build void __kmpc_copyprivate(ident_t *loc, kmp_int32 global_tid, 1899 // size_t cpy_size, void *cpy_data, void(*cpy_func)(void *, void *), 1900 // kmp_int32 didit); 1901 llvm::Type *CpyTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 1902 auto *CpyFnTy = 1903 llvm::FunctionType::get(CGM.VoidTy, CpyTypeParams, /*isVarArg=*/false); 1904 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.SizeTy, 1905 CGM.VoidPtrTy, CpyFnTy->getPointerTo(), 1906 CGM.Int32Ty}; 1907 auto *FnTy = 1908 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1909 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_copyprivate"); 1910 break; 1911 } 1912 case OMPRTL__kmpc_reduce: { 1913 // Build kmp_int32 __kmpc_reduce(ident_t *loc, kmp_int32 global_tid, 1914 // kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void 1915 // (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck); 1916 llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 1917 auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams, 1918 /*isVarArg=*/false); 1919 llvm::Type *TypeParams[] = { 1920 getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy, 1921 CGM.VoidPtrTy, ReduceFnTy->getPointerTo(), 1922 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1923 auto *FnTy = 1924 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1925 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce"); 1926 break; 1927 } 1928 case OMPRTL__kmpc_reduce_nowait: { 1929 // Build kmp_int32 __kmpc_reduce_nowait(ident_t *loc, kmp_int32 1930 // global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, 1931 // void (*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name 1932 // *lck); 1933 llvm::Type *ReduceTypeParams[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 1934 auto *ReduceFnTy = llvm::FunctionType::get(CGM.VoidTy, ReduceTypeParams, 1935 /*isVarArg=*/false); 1936 llvm::Type *TypeParams[] = { 1937 getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, CGM.SizeTy, 1938 CGM.VoidPtrTy, ReduceFnTy->getPointerTo(), 1939 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1940 auto *FnTy = 1941 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 1942 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_reduce_nowait"); 1943 break; 1944 } 1945 case OMPRTL__kmpc_end_reduce: { 1946 // Build void __kmpc_end_reduce(ident_t *loc, kmp_int32 global_tid, 1947 // kmp_critical_name *lck); 1948 llvm::Type *TypeParams[] = { 1949 getIdentTyPointerTy(), CGM.Int32Ty, 1950 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1951 auto *FnTy = 1952 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1953 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce"); 1954 break; 1955 } 1956 case OMPRTL__kmpc_end_reduce_nowait: { 1957 // Build __kmpc_end_reduce_nowait(ident_t *loc, kmp_int32 global_tid, 1958 // kmp_critical_name *lck); 1959 llvm::Type *TypeParams[] = { 1960 getIdentTyPointerTy(), CGM.Int32Ty, 1961 llvm::PointerType::getUnqual(KmpCriticalNameTy)}; 1962 auto *FnTy = 1963 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1964 RTLFn = 1965 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_end_reduce_nowait"); 1966 break; 1967 } 1968 case OMPRTL__kmpc_omp_task_begin_if0: { 1969 // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 1970 // *new_task); 1971 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1972 CGM.VoidPtrTy}; 1973 auto *FnTy = 1974 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1975 RTLFn = 1976 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_begin_if0"); 1977 break; 1978 } 1979 case OMPRTL__kmpc_omp_task_complete_if0: { 1980 // Build void __kmpc_omp_task(ident_t *, kmp_int32 gtid, kmp_task_t 1981 // *new_task); 1982 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 1983 CGM.VoidPtrTy}; 1984 auto *FnTy = 1985 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1986 RTLFn = CGM.CreateRuntimeFunction(FnTy, 1987 /*Name=*/"__kmpc_omp_task_complete_if0"); 1988 break; 1989 } 1990 case OMPRTL__kmpc_ordered: { 1991 // Build void __kmpc_ordered(ident_t *loc, kmp_int32 global_tid); 1992 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 1993 auto *FnTy = 1994 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 1995 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_ordered"); 1996 break; 1997 } 1998 case OMPRTL__kmpc_end_ordered: { 1999 // Build void __kmpc_end_ordered(ident_t *loc, kmp_int32 global_tid); 2000 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2001 auto *FnTy = 2002 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2003 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_ordered"); 2004 break; 2005 } 2006 case OMPRTL__kmpc_omp_taskwait: { 2007 // Build kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 global_tid); 2008 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2009 auto *FnTy = 2010 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2011 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_omp_taskwait"); 2012 break; 2013 } 2014 case OMPRTL__kmpc_taskgroup: { 2015 // Build void __kmpc_taskgroup(ident_t *loc, kmp_int32 global_tid); 2016 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2017 auto *FnTy = 2018 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2019 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_taskgroup"); 2020 break; 2021 } 2022 case OMPRTL__kmpc_end_taskgroup: { 2023 // Build void __kmpc_end_taskgroup(ident_t *loc, kmp_int32 global_tid); 2024 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2025 auto *FnTy = 2026 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2027 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_end_taskgroup"); 2028 break; 2029 } 2030 case OMPRTL__kmpc_push_proc_bind: { 2031 // Build void __kmpc_push_proc_bind(ident_t *loc, kmp_int32 global_tid, 2032 // int proc_bind) 2033 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2034 auto *FnTy = 2035 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2036 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_proc_bind"); 2037 break; 2038 } 2039 case OMPRTL__kmpc_omp_task_with_deps: { 2040 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 2041 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 2042 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list); 2043 llvm::Type *TypeParams[] = { 2044 getIdentTyPointerTy(), CGM.Int32Ty, CGM.VoidPtrTy, CGM.Int32Ty, 2045 CGM.VoidPtrTy, CGM.Int32Ty, CGM.VoidPtrTy}; 2046 auto *FnTy = 2047 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg=*/false); 2048 RTLFn = 2049 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_task_with_deps"); 2050 break; 2051 } 2052 case OMPRTL__kmpc_omp_wait_deps: { 2053 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 2054 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 ndeps_noalias, 2055 // kmp_depend_info_t *noalias_dep_list); 2056 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2057 CGM.Int32Ty, CGM.VoidPtrTy, 2058 CGM.Int32Ty, CGM.VoidPtrTy}; 2059 auto *FnTy = 2060 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2061 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_omp_wait_deps"); 2062 break; 2063 } 2064 case OMPRTL__kmpc_cancellationpoint: { 2065 // Build kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 2066 // global_tid, kmp_int32 cncl_kind) 2067 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2068 auto *FnTy = 2069 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2070 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancellationpoint"); 2071 break; 2072 } 2073 case OMPRTL__kmpc_cancel: { 2074 // Build kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 2075 // kmp_int32 cncl_kind) 2076 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.IntTy}; 2077 auto *FnTy = 2078 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2079 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_cancel"); 2080 break; 2081 } 2082 case OMPRTL__kmpc_push_num_teams: { 2083 // Build void kmpc_push_num_teams (ident_t loc, kmp_int32 global_tid, 2084 // kmp_int32 num_teams, kmp_int32 num_threads) 2085 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, CGM.Int32Ty, 2086 CGM.Int32Ty}; 2087 auto *FnTy = 2088 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2089 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_num_teams"); 2090 break; 2091 } 2092 case OMPRTL__kmpc_fork_teams: { 2093 // Build void __kmpc_fork_teams(ident_t *loc, kmp_int32 argc, kmpc_micro 2094 // microtask, ...); 2095 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2096 getKmpc_MicroPointerTy()}; 2097 auto *FnTy = 2098 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ true); 2099 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_fork_teams"); 2100 if (auto *F = dyn_cast<llvm::Function>(RTLFn.getCallee())) { 2101 if (!F->hasMetadata(llvm::LLVMContext::MD_callback)) { 2102 llvm::LLVMContext &Ctx = F->getContext(); 2103 llvm::MDBuilder MDB(Ctx); 2104 // Annotate the callback behavior of the __kmpc_fork_teams: 2105 // - The callback callee is argument number 2 (microtask). 2106 // - The first two arguments of the callback callee are unknown (-1). 2107 // - All variadic arguments to the __kmpc_fork_teams are passed to the 2108 // callback callee. 2109 F->addMetadata( 2110 llvm::LLVMContext::MD_callback, 2111 *llvm::MDNode::get(Ctx, {MDB.createCallbackEncoding( 2112 2, {-1, -1}, 2113 /* VarArgsArePassed */ true)})); 2114 } 2115 } 2116 break; 2117 } 2118 case OMPRTL__kmpc_taskloop: { 2119 // Build void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 2120 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 2121 // sched, kmp_uint64 grainsize, void *task_dup); 2122 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), 2123 CGM.IntTy, 2124 CGM.VoidPtrTy, 2125 CGM.IntTy, 2126 CGM.Int64Ty->getPointerTo(), 2127 CGM.Int64Ty->getPointerTo(), 2128 CGM.Int64Ty, 2129 CGM.IntTy, 2130 CGM.IntTy, 2131 CGM.Int64Ty, 2132 CGM.VoidPtrTy}; 2133 auto *FnTy = 2134 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2135 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_taskloop"); 2136 break; 2137 } 2138 case OMPRTL__kmpc_doacross_init: { 2139 // Build void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, kmp_int32 2140 // num_dims, struct kmp_dim *dims); 2141 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), 2142 CGM.Int32Ty, 2143 CGM.Int32Ty, 2144 CGM.VoidPtrTy}; 2145 auto *FnTy = 2146 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2147 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_init"); 2148 break; 2149 } 2150 case OMPRTL__kmpc_doacross_fini: { 2151 // Build void __kmpc_doacross_fini(ident_t *loc, kmp_int32 gtid); 2152 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty}; 2153 auto *FnTy = 2154 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2155 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_fini"); 2156 break; 2157 } 2158 case OMPRTL__kmpc_doacross_post: { 2159 // Build void __kmpc_doacross_post(ident_t *loc, kmp_int32 gtid, kmp_int64 2160 // *vec); 2161 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2162 CGM.Int64Ty->getPointerTo()}; 2163 auto *FnTy = 2164 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2165 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_post"); 2166 break; 2167 } 2168 case OMPRTL__kmpc_doacross_wait: { 2169 // Build void __kmpc_doacross_wait(ident_t *loc, kmp_int32 gtid, kmp_int64 2170 // *vec); 2171 llvm::Type *TypeParams[] = {getIdentTyPointerTy(), CGM.Int32Ty, 2172 CGM.Int64Ty->getPointerTo()}; 2173 auto *FnTy = 2174 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2175 RTLFn = CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_doacross_wait"); 2176 break; 2177 } 2178 case OMPRTL__kmpc_task_reduction_init: { 2179 // Build void *__kmpc_task_reduction_init(int gtid, int num_data, void 2180 // *data); 2181 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.IntTy, CGM.VoidPtrTy}; 2182 auto *FnTy = 2183 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2184 RTLFn = 2185 CGM.CreateRuntimeFunction(FnTy, /*Name=*/"__kmpc_task_reduction_init"); 2186 break; 2187 } 2188 case OMPRTL__kmpc_task_reduction_get_th_data: { 2189 // Build void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 2190 // *d); 2191 llvm::Type *TypeParams[] = {CGM.IntTy, CGM.VoidPtrTy, CGM.VoidPtrTy}; 2192 auto *FnTy = 2193 llvm::FunctionType::get(CGM.VoidPtrTy, TypeParams, /*isVarArg=*/false); 2194 RTLFn = CGM.CreateRuntimeFunction( 2195 FnTy, /*Name=*/"__kmpc_task_reduction_get_th_data"); 2196 break; 2197 } 2198 case OMPRTL__kmpc_push_target_tripcount: { 2199 // Build void __kmpc_push_target_tripcount(int64_t device_id, kmp_uint64 2200 // size); 2201 llvm::Type *TypeParams[] = {CGM.Int64Ty, CGM.Int64Ty}; 2202 llvm::FunctionType *FnTy = 2203 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2204 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__kmpc_push_target_tripcount"); 2205 break; 2206 } 2207 case OMPRTL__tgt_target: { 2208 // Build int32_t __tgt_target(int64_t device_id, void *host_ptr, int32_t 2209 // arg_num, void** args_base, void **args, size_t *arg_sizes, int64_t 2210 // *arg_types); 2211 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2212 CGM.VoidPtrTy, 2213 CGM.Int32Ty, 2214 CGM.VoidPtrPtrTy, 2215 CGM.VoidPtrPtrTy, 2216 CGM.SizeTy->getPointerTo(), 2217 CGM.Int64Ty->getPointerTo()}; 2218 auto *FnTy = 2219 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2220 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target"); 2221 break; 2222 } 2223 case OMPRTL__tgt_target_nowait: { 2224 // Build int32_t __tgt_target_nowait(int64_t device_id, void *host_ptr, 2225 // int32_t arg_num, void** args_base, void **args, size_t *arg_sizes, 2226 // int64_t *arg_types); 2227 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2228 CGM.VoidPtrTy, 2229 CGM.Int32Ty, 2230 CGM.VoidPtrPtrTy, 2231 CGM.VoidPtrPtrTy, 2232 CGM.SizeTy->getPointerTo(), 2233 CGM.Int64Ty->getPointerTo()}; 2234 auto *FnTy = 2235 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2236 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_nowait"); 2237 break; 2238 } 2239 case OMPRTL__tgt_target_teams: { 2240 // Build int32_t __tgt_target_teams(int64_t device_id, void *host_ptr, 2241 // int32_t arg_num, void** args_base, void **args, size_t *arg_sizes, 2242 // int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 2243 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2244 CGM.VoidPtrTy, 2245 CGM.Int32Ty, 2246 CGM.VoidPtrPtrTy, 2247 CGM.VoidPtrPtrTy, 2248 CGM.SizeTy->getPointerTo(), 2249 CGM.Int64Ty->getPointerTo(), 2250 CGM.Int32Ty, 2251 CGM.Int32Ty}; 2252 auto *FnTy = 2253 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2254 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams"); 2255 break; 2256 } 2257 case OMPRTL__tgt_target_teams_nowait: { 2258 // Build int32_t __tgt_target_teams_nowait(int64_t device_id, void 2259 // *host_ptr, int32_t arg_num, void** args_base, void **args, size_t 2260 // *arg_sizes, int64_t *arg_types, int32_t num_teams, int32_t thread_limit); 2261 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2262 CGM.VoidPtrTy, 2263 CGM.Int32Ty, 2264 CGM.VoidPtrPtrTy, 2265 CGM.VoidPtrPtrTy, 2266 CGM.SizeTy->getPointerTo(), 2267 CGM.Int64Ty->getPointerTo(), 2268 CGM.Int32Ty, 2269 CGM.Int32Ty}; 2270 auto *FnTy = 2271 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2272 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_teams_nowait"); 2273 break; 2274 } 2275 case OMPRTL__tgt_register_lib: { 2276 // Build void __tgt_register_lib(__tgt_bin_desc *desc); 2277 QualType ParamTy = 2278 CGM.getContext().getPointerType(getTgtBinaryDescriptorQTy()); 2279 llvm::Type *TypeParams[] = {CGM.getTypes().ConvertTypeForMem(ParamTy)}; 2280 auto *FnTy = 2281 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2282 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_register_lib"); 2283 break; 2284 } 2285 case OMPRTL__tgt_unregister_lib: { 2286 // Build void __tgt_unregister_lib(__tgt_bin_desc *desc); 2287 QualType ParamTy = 2288 CGM.getContext().getPointerType(getTgtBinaryDescriptorQTy()); 2289 llvm::Type *TypeParams[] = {CGM.getTypes().ConvertTypeForMem(ParamTy)}; 2290 auto *FnTy = 2291 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2292 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_unregister_lib"); 2293 break; 2294 } 2295 case OMPRTL__tgt_target_data_begin: { 2296 // Build void __tgt_target_data_begin(int64_t device_id, int32_t arg_num, 2297 // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types); 2298 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2299 CGM.Int32Ty, 2300 CGM.VoidPtrPtrTy, 2301 CGM.VoidPtrPtrTy, 2302 CGM.SizeTy->getPointerTo(), 2303 CGM.Int64Ty->getPointerTo()}; 2304 auto *FnTy = 2305 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2306 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin"); 2307 break; 2308 } 2309 case OMPRTL__tgt_target_data_begin_nowait: { 2310 // Build void __tgt_target_data_begin_nowait(int64_t device_id, int32_t 2311 // arg_num, void** args_base, void **args, size_t *arg_sizes, int64_t 2312 // *arg_types); 2313 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2314 CGM.Int32Ty, 2315 CGM.VoidPtrPtrTy, 2316 CGM.VoidPtrPtrTy, 2317 CGM.SizeTy->getPointerTo(), 2318 CGM.Int64Ty->getPointerTo()}; 2319 auto *FnTy = 2320 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2321 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_begin_nowait"); 2322 break; 2323 } 2324 case OMPRTL__tgt_target_data_end: { 2325 // Build void __tgt_target_data_end(int64_t device_id, int32_t arg_num, 2326 // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types); 2327 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2328 CGM.Int32Ty, 2329 CGM.VoidPtrPtrTy, 2330 CGM.VoidPtrPtrTy, 2331 CGM.SizeTy->getPointerTo(), 2332 CGM.Int64Ty->getPointerTo()}; 2333 auto *FnTy = 2334 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2335 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end"); 2336 break; 2337 } 2338 case OMPRTL__tgt_target_data_end_nowait: { 2339 // Build void __tgt_target_data_end_nowait(int64_t device_id, int32_t 2340 // arg_num, void** args_base, void **args, size_t *arg_sizes, int64_t 2341 // *arg_types); 2342 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2343 CGM.Int32Ty, 2344 CGM.VoidPtrPtrTy, 2345 CGM.VoidPtrPtrTy, 2346 CGM.SizeTy->getPointerTo(), 2347 CGM.Int64Ty->getPointerTo()}; 2348 auto *FnTy = 2349 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2350 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_end_nowait"); 2351 break; 2352 } 2353 case OMPRTL__tgt_target_data_update: { 2354 // Build void __tgt_target_data_update(int64_t device_id, int32_t arg_num, 2355 // void** args_base, void **args, size_t *arg_sizes, int64_t *arg_types); 2356 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2357 CGM.Int32Ty, 2358 CGM.VoidPtrPtrTy, 2359 CGM.VoidPtrPtrTy, 2360 CGM.SizeTy->getPointerTo(), 2361 CGM.Int64Ty->getPointerTo()}; 2362 auto *FnTy = 2363 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2364 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update"); 2365 break; 2366 } 2367 case OMPRTL__tgt_target_data_update_nowait: { 2368 // Build void __tgt_target_data_update_nowait(int64_t device_id, int32_t 2369 // arg_num, void** args_base, void **args, size_t *arg_sizes, int64_t 2370 // *arg_types); 2371 llvm::Type *TypeParams[] = {CGM.Int64Ty, 2372 CGM.Int32Ty, 2373 CGM.VoidPtrPtrTy, 2374 CGM.VoidPtrPtrTy, 2375 CGM.SizeTy->getPointerTo(), 2376 CGM.Int64Ty->getPointerTo()}; 2377 auto *FnTy = 2378 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2379 RTLFn = CGM.CreateRuntimeFunction(FnTy, "__tgt_target_data_update_nowait"); 2380 break; 2381 } 2382 } 2383 assert(RTLFn && "Unable to find OpenMP runtime function"); 2384 return RTLFn; 2385 } 2386 2387 llvm::FunctionCallee 2388 CGOpenMPRuntime::createForStaticInitFunction(unsigned IVSize, bool IVSigned) { 2389 assert((IVSize == 32 || IVSize == 64) && 2390 "IV size is not compatible with the omp runtime"); 2391 StringRef Name = IVSize == 32 ? (IVSigned ? "__kmpc_for_static_init_4" 2392 : "__kmpc_for_static_init_4u") 2393 : (IVSigned ? "__kmpc_for_static_init_8" 2394 : "__kmpc_for_static_init_8u"); 2395 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2396 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 2397 llvm::Type *TypeParams[] = { 2398 getIdentTyPointerTy(), // loc 2399 CGM.Int32Ty, // tid 2400 CGM.Int32Ty, // schedtype 2401 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 2402 PtrTy, // p_lower 2403 PtrTy, // p_upper 2404 PtrTy, // p_stride 2405 ITy, // incr 2406 ITy // chunk 2407 }; 2408 auto *FnTy = 2409 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2410 return CGM.CreateRuntimeFunction(FnTy, Name); 2411 } 2412 2413 llvm::FunctionCallee 2414 CGOpenMPRuntime::createDispatchInitFunction(unsigned IVSize, bool IVSigned) { 2415 assert((IVSize == 32 || IVSize == 64) && 2416 "IV size is not compatible with the omp runtime"); 2417 StringRef Name = 2418 IVSize == 32 2419 ? (IVSigned ? "__kmpc_dispatch_init_4" : "__kmpc_dispatch_init_4u") 2420 : (IVSigned ? "__kmpc_dispatch_init_8" : "__kmpc_dispatch_init_8u"); 2421 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2422 llvm::Type *TypeParams[] = { getIdentTyPointerTy(), // loc 2423 CGM.Int32Ty, // tid 2424 CGM.Int32Ty, // schedtype 2425 ITy, // lower 2426 ITy, // upper 2427 ITy, // stride 2428 ITy // chunk 2429 }; 2430 auto *FnTy = 2431 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg*/ false); 2432 return CGM.CreateRuntimeFunction(FnTy, Name); 2433 } 2434 2435 llvm::FunctionCallee 2436 CGOpenMPRuntime::createDispatchFiniFunction(unsigned IVSize, bool IVSigned) { 2437 assert((IVSize == 32 || IVSize == 64) && 2438 "IV size is not compatible with the omp runtime"); 2439 StringRef Name = 2440 IVSize == 32 2441 ? (IVSigned ? "__kmpc_dispatch_fini_4" : "__kmpc_dispatch_fini_4u") 2442 : (IVSigned ? "__kmpc_dispatch_fini_8" : "__kmpc_dispatch_fini_8u"); 2443 llvm::Type *TypeParams[] = { 2444 getIdentTyPointerTy(), // loc 2445 CGM.Int32Ty, // tid 2446 }; 2447 auto *FnTy = 2448 llvm::FunctionType::get(CGM.VoidTy, TypeParams, /*isVarArg=*/false); 2449 return CGM.CreateRuntimeFunction(FnTy, Name); 2450 } 2451 2452 llvm::FunctionCallee 2453 CGOpenMPRuntime::createDispatchNextFunction(unsigned IVSize, bool IVSigned) { 2454 assert((IVSize == 32 || IVSize == 64) && 2455 "IV size is not compatible with the omp runtime"); 2456 StringRef Name = 2457 IVSize == 32 2458 ? (IVSigned ? "__kmpc_dispatch_next_4" : "__kmpc_dispatch_next_4u") 2459 : (IVSigned ? "__kmpc_dispatch_next_8" : "__kmpc_dispatch_next_8u"); 2460 llvm::Type *ITy = IVSize == 32 ? CGM.Int32Ty : CGM.Int64Ty; 2461 auto *PtrTy = llvm::PointerType::getUnqual(ITy); 2462 llvm::Type *TypeParams[] = { 2463 getIdentTyPointerTy(), // loc 2464 CGM.Int32Ty, // tid 2465 llvm::PointerType::getUnqual(CGM.Int32Ty), // p_lastiter 2466 PtrTy, // p_lower 2467 PtrTy, // p_upper 2468 PtrTy // p_stride 2469 }; 2470 auto *FnTy = 2471 llvm::FunctionType::get(CGM.Int32Ty, TypeParams, /*isVarArg*/ false); 2472 return CGM.CreateRuntimeFunction(FnTy, Name); 2473 } 2474 2475 Address CGOpenMPRuntime::getAddrOfDeclareTargetLink(const VarDecl *VD) { 2476 if (CGM.getLangOpts().OpenMPSimd) 2477 return Address::invalid(); 2478 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 2479 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 2480 if (Res && *Res == OMPDeclareTargetDeclAttr::MT_Link) { 2481 SmallString<64> PtrName; 2482 { 2483 llvm::raw_svector_ostream OS(PtrName); 2484 OS << CGM.getMangledName(GlobalDecl(VD)) << "_decl_tgt_link_ptr"; 2485 } 2486 llvm::Value *Ptr = CGM.getModule().getNamedValue(PtrName); 2487 if (!Ptr) { 2488 QualType PtrTy = CGM.getContext().getPointerType(VD->getType()); 2489 Ptr = getOrCreateInternalVariable(CGM.getTypes().ConvertTypeForMem(PtrTy), 2490 PtrName); 2491 if (!CGM.getLangOpts().OpenMPIsDevice) { 2492 auto *GV = cast<llvm::GlobalVariable>(Ptr); 2493 GV->setLinkage(llvm::GlobalValue::ExternalLinkage); 2494 GV->setInitializer(CGM.GetAddrOfGlobal(VD)); 2495 } 2496 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ptr)); 2497 registerTargetGlobalVariable(VD, cast<llvm::Constant>(Ptr)); 2498 } 2499 return Address(Ptr, CGM.getContext().getDeclAlign(VD)); 2500 } 2501 return Address::invalid(); 2502 } 2503 2504 llvm::Constant * 2505 CGOpenMPRuntime::getOrCreateThreadPrivateCache(const VarDecl *VD) { 2506 assert(!CGM.getLangOpts().OpenMPUseTLS || 2507 !CGM.getContext().getTargetInfo().isTLSSupported()); 2508 // Lookup the entry, lazily creating it if necessary. 2509 std::string Suffix = getName({"cache", ""}); 2510 return getOrCreateInternalVariable( 2511 CGM.Int8PtrPtrTy, Twine(CGM.getMangledName(VD)).concat(Suffix)); 2512 } 2513 2514 Address CGOpenMPRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 2515 const VarDecl *VD, 2516 Address VDAddr, 2517 SourceLocation Loc) { 2518 if (CGM.getLangOpts().OpenMPUseTLS && 2519 CGM.getContext().getTargetInfo().isTLSSupported()) 2520 return VDAddr; 2521 2522 llvm::Type *VarTy = VDAddr.getElementType(); 2523 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 2524 CGF.Builder.CreatePointerCast(VDAddr.getPointer(), 2525 CGM.Int8PtrTy), 2526 CGM.getSize(CGM.GetTargetTypeStoreSize(VarTy)), 2527 getOrCreateThreadPrivateCache(VD)}; 2528 return Address(CGF.EmitRuntimeCall( 2529 createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args), 2530 VDAddr.getAlignment()); 2531 } 2532 2533 void CGOpenMPRuntime::emitThreadPrivateVarInit( 2534 CodeGenFunction &CGF, Address VDAddr, llvm::Value *Ctor, 2535 llvm::Value *CopyCtor, llvm::Value *Dtor, SourceLocation Loc) { 2536 // Call kmp_int32 __kmpc_global_thread_num(&loc) to init OpenMP runtime 2537 // library. 2538 llvm::Value *OMPLoc = emitUpdateLocation(CGF, Loc); 2539 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_global_thread_num), 2540 OMPLoc); 2541 // Call __kmpc_threadprivate_register(&loc, &var, ctor, cctor/*NULL*/, dtor) 2542 // to register constructor/destructor for variable. 2543 llvm::Value *Args[] = { 2544 OMPLoc, CGF.Builder.CreatePointerCast(VDAddr.getPointer(), CGM.VoidPtrTy), 2545 Ctor, CopyCtor, Dtor}; 2546 CGF.EmitRuntimeCall( 2547 createRuntimeFunction(OMPRTL__kmpc_threadprivate_register), Args); 2548 } 2549 2550 llvm::Function *CGOpenMPRuntime::emitThreadPrivateVarDefinition( 2551 const VarDecl *VD, Address VDAddr, SourceLocation Loc, 2552 bool PerformInit, CodeGenFunction *CGF) { 2553 if (CGM.getLangOpts().OpenMPUseTLS && 2554 CGM.getContext().getTargetInfo().isTLSSupported()) 2555 return nullptr; 2556 2557 VD = VD->getDefinition(CGM.getContext()); 2558 if (VD && ThreadPrivateWithDefinition.insert(CGM.getMangledName(VD)).second) { 2559 QualType ASTTy = VD->getType(); 2560 2561 llvm::Value *Ctor = nullptr, *CopyCtor = nullptr, *Dtor = nullptr; 2562 const Expr *Init = VD->getAnyInitializer(); 2563 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 2564 // Generate function that re-emits the declaration's initializer into the 2565 // threadprivate copy of the variable VD 2566 CodeGenFunction CtorCGF(CGM); 2567 FunctionArgList Args; 2568 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 2569 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 2570 ImplicitParamDecl::Other); 2571 Args.push_back(&Dst); 2572 2573 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 2574 CGM.getContext().VoidPtrTy, Args); 2575 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2576 std::string Name = getName({"__kmpc_global_ctor_", ""}); 2577 llvm::Function *Fn = 2578 CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc); 2579 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidPtrTy, Fn, FI, 2580 Args, Loc, Loc); 2581 llvm::Value *ArgVal = CtorCGF.EmitLoadOfScalar( 2582 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 2583 CGM.getContext().VoidPtrTy, Dst.getLocation()); 2584 Address Arg = Address(ArgVal, VDAddr.getAlignment()); 2585 Arg = CtorCGF.Builder.CreateElementBitCast( 2586 Arg, CtorCGF.ConvertTypeForMem(ASTTy)); 2587 CtorCGF.EmitAnyExprToMem(Init, Arg, Init->getType().getQualifiers(), 2588 /*IsInitializer=*/true); 2589 ArgVal = CtorCGF.EmitLoadOfScalar( 2590 CtorCGF.GetAddrOfLocalVar(&Dst), /*Volatile=*/false, 2591 CGM.getContext().VoidPtrTy, Dst.getLocation()); 2592 CtorCGF.Builder.CreateStore(ArgVal, CtorCGF.ReturnValue); 2593 CtorCGF.FinishFunction(); 2594 Ctor = Fn; 2595 } 2596 if (VD->getType().isDestructedType() != QualType::DK_none) { 2597 // Generate function that emits destructor call for the threadprivate copy 2598 // of the variable VD 2599 CodeGenFunction DtorCGF(CGM); 2600 FunctionArgList Args; 2601 ImplicitParamDecl Dst(CGM.getContext(), /*DC=*/nullptr, Loc, 2602 /*Id=*/nullptr, CGM.getContext().VoidPtrTy, 2603 ImplicitParamDecl::Other); 2604 Args.push_back(&Dst); 2605 2606 const auto &FI = CGM.getTypes().arrangeBuiltinFunctionDeclaration( 2607 CGM.getContext().VoidTy, Args); 2608 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2609 std::string Name = getName({"__kmpc_global_dtor_", ""}); 2610 llvm::Function *Fn = 2611 CGM.CreateGlobalInitOrDestructFunction(FTy, Name, FI, Loc); 2612 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 2613 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, Args, 2614 Loc, Loc); 2615 // Create a scope with an artificial location for the body of this function. 2616 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 2617 llvm::Value *ArgVal = DtorCGF.EmitLoadOfScalar( 2618 DtorCGF.GetAddrOfLocalVar(&Dst), 2619 /*Volatile=*/false, CGM.getContext().VoidPtrTy, Dst.getLocation()); 2620 DtorCGF.emitDestroy(Address(ArgVal, VDAddr.getAlignment()), ASTTy, 2621 DtorCGF.getDestroyer(ASTTy.isDestructedType()), 2622 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 2623 DtorCGF.FinishFunction(); 2624 Dtor = Fn; 2625 } 2626 // Do not emit init function if it is not required. 2627 if (!Ctor && !Dtor) 2628 return nullptr; 2629 2630 llvm::Type *CopyCtorTyArgs[] = {CGM.VoidPtrTy, CGM.VoidPtrTy}; 2631 auto *CopyCtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CopyCtorTyArgs, 2632 /*isVarArg=*/false) 2633 ->getPointerTo(); 2634 // Copying constructor for the threadprivate variable. 2635 // Must be NULL - reserved by runtime, but currently it requires that this 2636 // parameter is always NULL. Otherwise it fires assertion. 2637 CopyCtor = llvm::Constant::getNullValue(CopyCtorTy); 2638 if (Ctor == nullptr) { 2639 auto *CtorTy = llvm::FunctionType::get(CGM.VoidPtrTy, CGM.VoidPtrTy, 2640 /*isVarArg=*/false) 2641 ->getPointerTo(); 2642 Ctor = llvm::Constant::getNullValue(CtorTy); 2643 } 2644 if (Dtor == nullptr) { 2645 auto *DtorTy = llvm::FunctionType::get(CGM.VoidTy, CGM.VoidPtrTy, 2646 /*isVarArg=*/false) 2647 ->getPointerTo(); 2648 Dtor = llvm::Constant::getNullValue(DtorTy); 2649 } 2650 if (!CGF) { 2651 auto *InitFunctionTy = 2652 llvm::FunctionType::get(CGM.VoidTy, /*isVarArg*/ false); 2653 std::string Name = getName({"__omp_threadprivate_init_", ""}); 2654 llvm::Function *InitFunction = CGM.CreateGlobalInitOrDestructFunction( 2655 InitFunctionTy, Name, CGM.getTypes().arrangeNullaryFunction()); 2656 CodeGenFunction InitCGF(CGM); 2657 FunctionArgList ArgList; 2658 InitCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, InitFunction, 2659 CGM.getTypes().arrangeNullaryFunction(), ArgList, 2660 Loc, Loc); 2661 emitThreadPrivateVarInit(InitCGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 2662 InitCGF.FinishFunction(); 2663 return InitFunction; 2664 } 2665 emitThreadPrivateVarInit(*CGF, VDAddr, Ctor, CopyCtor, Dtor, Loc); 2666 } 2667 return nullptr; 2668 } 2669 2670 /// Obtain information that uniquely identifies a target entry. This 2671 /// consists of the file and device IDs as well as line number associated with 2672 /// the relevant entry source location. 2673 static void getTargetEntryUniqueInfo(ASTContext &C, SourceLocation Loc, 2674 unsigned &DeviceID, unsigned &FileID, 2675 unsigned &LineNum) { 2676 SourceManager &SM = C.getSourceManager(); 2677 2678 // The loc should be always valid and have a file ID (the user cannot use 2679 // #pragma directives in macros) 2680 2681 assert(Loc.isValid() && "Source location is expected to be always valid."); 2682 2683 PresumedLoc PLoc = SM.getPresumedLoc(Loc); 2684 assert(PLoc.isValid() && "Source location is expected to be always valid."); 2685 2686 llvm::sys::fs::UniqueID ID; 2687 if (auto EC = llvm::sys::fs::getUniqueID(PLoc.getFilename(), ID)) 2688 SM.getDiagnostics().Report(diag::err_cannot_open_file) 2689 << PLoc.getFilename() << EC.message(); 2690 2691 DeviceID = ID.getDevice(); 2692 FileID = ID.getFile(); 2693 LineNum = PLoc.getLine(); 2694 } 2695 2696 bool CGOpenMPRuntime::emitDeclareTargetVarDefinition(const VarDecl *VD, 2697 llvm::GlobalVariable *Addr, 2698 bool PerformInit) { 2699 Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 2700 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 2701 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link) 2702 return CGM.getLangOpts().OpenMPIsDevice; 2703 VD = VD->getDefinition(CGM.getContext()); 2704 if (VD && !DeclareTargetWithDefinition.insert(CGM.getMangledName(VD)).second) 2705 return CGM.getLangOpts().OpenMPIsDevice; 2706 2707 QualType ASTTy = VD->getType(); 2708 2709 SourceLocation Loc = VD->getCanonicalDecl()->getBeginLoc(); 2710 // Produce the unique prefix to identify the new target regions. We use 2711 // the source location of the variable declaration which we know to not 2712 // conflict with any target region. 2713 unsigned DeviceID; 2714 unsigned FileID; 2715 unsigned Line; 2716 getTargetEntryUniqueInfo(CGM.getContext(), Loc, DeviceID, FileID, Line); 2717 SmallString<128> Buffer, Out; 2718 { 2719 llvm::raw_svector_ostream OS(Buffer); 2720 OS << "__omp_offloading_" << llvm::format("_%x", DeviceID) 2721 << llvm::format("_%x_", FileID) << VD->getName() << "_l" << Line; 2722 } 2723 2724 const Expr *Init = VD->getAnyInitializer(); 2725 if (CGM.getLangOpts().CPlusPlus && PerformInit) { 2726 llvm::Constant *Ctor; 2727 llvm::Constant *ID; 2728 if (CGM.getLangOpts().OpenMPIsDevice) { 2729 // Generate function that re-emits the declaration's initializer into 2730 // the threadprivate copy of the variable VD 2731 CodeGenFunction CtorCGF(CGM); 2732 2733 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 2734 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2735 llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction( 2736 FTy, Twine(Buffer, "_ctor"), FI, Loc); 2737 auto NL = ApplyDebugLocation::CreateEmpty(CtorCGF); 2738 CtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 2739 FunctionArgList(), Loc, Loc); 2740 auto AL = ApplyDebugLocation::CreateArtificial(CtorCGF); 2741 CtorCGF.EmitAnyExprToMem(Init, 2742 Address(Addr, CGM.getContext().getDeclAlign(VD)), 2743 Init->getType().getQualifiers(), 2744 /*IsInitializer=*/true); 2745 CtorCGF.FinishFunction(); 2746 Ctor = Fn; 2747 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 2748 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Ctor)); 2749 } else { 2750 Ctor = new llvm::GlobalVariable( 2751 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 2752 llvm::GlobalValue::PrivateLinkage, 2753 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_ctor")); 2754 ID = Ctor; 2755 } 2756 2757 // Register the information for the entry associated with the constructor. 2758 Out.clear(); 2759 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 2760 DeviceID, FileID, Twine(Buffer, "_ctor").toStringRef(Out), Line, Ctor, 2761 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryCtor); 2762 } 2763 if (VD->getType().isDestructedType() != QualType::DK_none) { 2764 llvm::Constant *Dtor; 2765 llvm::Constant *ID; 2766 if (CGM.getLangOpts().OpenMPIsDevice) { 2767 // Generate function that emits destructor call for the threadprivate 2768 // copy of the variable VD 2769 CodeGenFunction DtorCGF(CGM); 2770 2771 const CGFunctionInfo &FI = CGM.getTypes().arrangeNullaryFunction(); 2772 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 2773 llvm::Function *Fn = CGM.CreateGlobalInitOrDestructFunction( 2774 FTy, Twine(Buffer, "_dtor"), FI, Loc); 2775 auto NL = ApplyDebugLocation::CreateEmpty(DtorCGF); 2776 DtorCGF.StartFunction(GlobalDecl(), CGM.getContext().VoidTy, Fn, FI, 2777 FunctionArgList(), Loc, Loc); 2778 // Create a scope with an artificial location for the body of this 2779 // function. 2780 auto AL = ApplyDebugLocation::CreateArtificial(DtorCGF); 2781 DtorCGF.emitDestroy(Address(Addr, CGM.getContext().getDeclAlign(VD)), 2782 ASTTy, DtorCGF.getDestroyer(ASTTy.isDestructedType()), 2783 DtorCGF.needsEHCleanup(ASTTy.isDestructedType())); 2784 DtorCGF.FinishFunction(); 2785 Dtor = Fn; 2786 ID = llvm::ConstantExpr::getBitCast(Fn, CGM.Int8PtrTy); 2787 CGM.addUsedGlobal(cast<llvm::GlobalValue>(Dtor)); 2788 } else { 2789 Dtor = new llvm::GlobalVariable( 2790 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 2791 llvm::GlobalValue::PrivateLinkage, 2792 llvm::Constant::getNullValue(CGM.Int8Ty), Twine(Buffer, "_dtor")); 2793 ID = Dtor; 2794 } 2795 // Register the information for the entry associated with the destructor. 2796 Out.clear(); 2797 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 2798 DeviceID, FileID, Twine(Buffer, "_dtor").toStringRef(Out), Line, Dtor, 2799 ID, OffloadEntriesInfoManagerTy::OMPTargetRegionEntryDtor); 2800 } 2801 return CGM.getLangOpts().OpenMPIsDevice; 2802 } 2803 2804 Address CGOpenMPRuntime::getAddrOfArtificialThreadPrivate(CodeGenFunction &CGF, 2805 QualType VarType, 2806 StringRef Name) { 2807 std::string Suffix = getName({"artificial", ""}); 2808 std::string CacheSuffix = getName({"cache", ""}); 2809 llvm::Type *VarLVType = CGF.ConvertTypeForMem(VarType); 2810 llvm::Value *GAddr = 2811 getOrCreateInternalVariable(VarLVType, Twine(Name).concat(Suffix)); 2812 llvm::Value *Args[] = { 2813 emitUpdateLocation(CGF, SourceLocation()), 2814 getThreadID(CGF, SourceLocation()), 2815 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(GAddr, CGM.VoidPtrTy), 2816 CGF.Builder.CreateIntCast(CGF.getTypeSize(VarType), CGM.SizeTy, 2817 /*IsSigned=*/false), 2818 getOrCreateInternalVariable( 2819 CGM.VoidPtrPtrTy, Twine(Name).concat(Suffix).concat(CacheSuffix))}; 2820 return Address( 2821 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 2822 CGF.EmitRuntimeCall( 2823 createRuntimeFunction(OMPRTL__kmpc_threadprivate_cached), Args), 2824 VarLVType->getPointerTo(/*AddrSpace=*/0)), 2825 CGM.getPointerAlign()); 2826 } 2827 2828 void CGOpenMPRuntime::emitOMPIfClause(CodeGenFunction &CGF, const Expr *Cond, 2829 const RegionCodeGenTy &ThenGen, 2830 const RegionCodeGenTy &ElseGen) { 2831 CodeGenFunction::LexicalScope ConditionScope(CGF, Cond->getSourceRange()); 2832 2833 // If the condition constant folds and can be elided, try to avoid emitting 2834 // the condition and the dead arm of the if/else. 2835 bool CondConstant; 2836 if (CGF.ConstantFoldsToSimpleInteger(Cond, CondConstant)) { 2837 if (CondConstant) 2838 ThenGen(CGF); 2839 else 2840 ElseGen(CGF); 2841 return; 2842 } 2843 2844 // Otherwise, the condition did not fold, or we couldn't elide it. Just 2845 // emit the conditional branch. 2846 llvm::BasicBlock *ThenBlock = CGF.createBasicBlock("omp_if.then"); 2847 llvm::BasicBlock *ElseBlock = CGF.createBasicBlock("omp_if.else"); 2848 llvm::BasicBlock *ContBlock = CGF.createBasicBlock("omp_if.end"); 2849 CGF.EmitBranchOnBoolExpr(Cond, ThenBlock, ElseBlock, /*TrueCount=*/0); 2850 2851 // Emit the 'then' code. 2852 CGF.EmitBlock(ThenBlock); 2853 ThenGen(CGF); 2854 CGF.EmitBranch(ContBlock); 2855 // Emit the 'else' code if present. 2856 // There is no need to emit line number for unconditional branch. 2857 (void)ApplyDebugLocation::CreateEmpty(CGF); 2858 CGF.EmitBlock(ElseBlock); 2859 ElseGen(CGF); 2860 // There is no need to emit line number for unconditional branch. 2861 (void)ApplyDebugLocation::CreateEmpty(CGF); 2862 CGF.EmitBranch(ContBlock); 2863 // Emit the continuation block for code after the if. 2864 CGF.EmitBlock(ContBlock, /*IsFinished=*/true); 2865 } 2866 2867 void CGOpenMPRuntime::emitParallelCall(CodeGenFunction &CGF, SourceLocation Loc, 2868 llvm::Function *OutlinedFn, 2869 ArrayRef<llvm::Value *> CapturedVars, 2870 const Expr *IfCond) { 2871 if (!CGF.HaveInsertPoint()) 2872 return; 2873 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 2874 auto &&ThenGen = [OutlinedFn, CapturedVars, RTLoc](CodeGenFunction &CGF, 2875 PrePostActionTy &) { 2876 // Build call __kmpc_fork_call(loc, n, microtask, var1, .., varn); 2877 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 2878 llvm::Value *Args[] = { 2879 RTLoc, 2880 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 2881 CGF.Builder.CreateBitCast(OutlinedFn, RT.getKmpc_MicroPointerTy())}; 2882 llvm::SmallVector<llvm::Value *, 16> RealArgs; 2883 RealArgs.append(std::begin(Args), std::end(Args)); 2884 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 2885 2886 llvm::FunctionCallee RTLFn = 2887 RT.createRuntimeFunction(OMPRTL__kmpc_fork_call); 2888 CGF.EmitRuntimeCall(RTLFn, RealArgs); 2889 }; 2890 auto &&ElseGen = [OutlinedFn, CapturedVars, RTLoc, Loc](CodeGenFunction &CGF, 2891 PrePostActionTy &) { 2892 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 2893 llvm::Value *ThreadID = RT.getThreadID(CGF, Loc); 2894 // Build calls: 2895 // __kmpc_serialized_parallel(&Loc, GTid); 2896 llvm::Value *Args[] = {RTLoc, ThreadID}; 2897 CGF.EmitRuntimeCall( 2898 RT.createRuntimeFunction(OMPRTL__kmpc_serialized_parallel), Args); 2899 2900 // OutlinedFn(>id, &zero, CapturedStruct); 2901 Address ZeroAddr = CGF.CreateDefaultAlignTempAlloca(CGF.Int32Ty, 2902 /*Name*/ ".zero.addr"); 2903 CGF.InitTempAlloca(ZeroAddr, CGF.Builder.getInt32(/*C*/ 0)); 2904 llvm::SmallVector<llvm::Value *, 16> OutlinedFnArgs; 2905 // ThreadId for serialized parallels is 0. 2906 OutlinedFnArgs.push_back(ZeroAddr.getPointer()); 2907 OutlinedFnArgs.push_back(ZeroAddr.getPointer()); 2908 OutlinedFnArgs.append(CapturedVars.begin(), CapturedVars.end()); 2909 RT.emitOutlinedFunctionCall(CGF, Loc, OutlinedFn, OutlinedFnArgs); 2910 2911 // __kmpc_end_serialized_parallel(&Loc, GTid); 2912 llvm::Value *EndArgs[] = {RT.emitUpdateLocation(CGF, Loc), ThreadID}; 2913 CGF.EmitRuntimeCall( 2914 RT.createRuntimeFunction(OMPRTL__kmpc_end_serialized_parallel), 2915 EndArgs); 2916 }; 2917 if (IfCond) { 2918 emitOMPIfClause(CGF, IfCond, ThenGen, ElseGen); 2919 } else { 2920 RegionCodeGenTy ThenRCG(ThenGen); 2921 ThenRCG(CGF); 2922 } 2923 } 2924 2925 // If we're inside an (outlined) parallel region, use the region info's 2926 // thread-ID variable (it is passed in a first argument of the outlined function 2927 // as "kmp_int32 *gtid"). Otherwise, if we're not inside parallel region, but in 2928 // regular serial code region, get thread ID by calling kmp_int32 2929 // kmpc_global_thread_num(ident_t *loc), stash this thread ID in a temporary and 2930 // return the address of that temp. 2931 Address CGOpenMPRuntime::emitThreadIDAddress(CodeGenFunction &CGF, 2932 SourceLocation Loc) { 2933 if (auto *OMPRegionInfo = 2934 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 2935 if (OMPRegionInfo->getThreadIDVariable()) 2936 return OMPRegionInfo->getThreadIDVariableLValue(CGF).getAddress(); 2937 2938 llvm::Value *ThreadID = getThreadID(CGF, Loc); 2939 QualType Int32Ty = 2940 CGF.getContext().getIntTypeForBitwidth(/*DestWidth*/ 32, /*Signed*/ true); 2941 Address ThreadIDTemp = CGF.CreateMemTemp(Int32Ty, /*Name*/ ".threadid_temp."); 2942 CGF.EmitStoreOfScalar(ThreadID, 2943 CGF.MakeAddrLValue(ThreadIDTemp, Int32Ty)); 2944 2945 return ThreadIDTemp; 2946 } 2947 2948 llvm::Constant * 2949 CGOpenMPRuntime::getOrCreateInternalVariable(llvm::Type *Ty, 2950 const llvm::Twine &Name) { 2951 SmallString<256> Buffer; 2952 llvm::raw_svector_ostream Out(Buffer); 2953 Out << Name; 2954 StringRef RuntimeName = Out.str(); 2955 auto &Elem = *InternalVars.try_emplace(RuntimeName, nullptr).first; 2956 if (Elem.second) { 2957 assert(Elem.second->getType()->getPointerElementType() == Ty && 2958 "OMP internal variable has different type than requested"); 2959 return &*Elem.second; 2960 } 2961 2962 return Elem.second = new llvm::GlobalVariable( 2963 CGM.getModule(), Ty, /*IsConstant*/ false, 2964 llvm::GlobalValue::CommonLinkage, llvm::Constant::getNullValue(Ty), 2965 Elem.first()); 2966 } 2967 2968 llvm::Value *CGOpenMPRuntime::getCriticalRegionLock(StringRef CriticalName) { 2969 std::string Prefix = Twine("gomp_critical_user_", CriticalName).str(); 2970 std::string Name = getName({Prefix, "var"}); 2971 return getOrCreateInternalVariable(KmpCriticalNameTy, Name); 2972 } 2973 2974 namespace { 2975 /// Common pre(post)-action for different OpenMP constructs. 2976 class CommonActionTy final : public PrePostActionTy { 2977 llvm::FunctionCallee EnterCallee; 2978 ArrayRef<llvm::Value *> EnterArgs; 2979 llvm::FunctionCallee ExitCallee; 2980 ArrayRef<llvm::Value *> ExitArgs; 2981 bool Conditional; 2982 llvm::BasicBlock *ContBlock = nullptr; 2983 2984 public: 2985 CommonActionTy(llvm::FunctionCallee EnterCallee, 2986 ArrayRef<llvm::Value *> EnterArgs, 2987 llvm::FunctionCallee ExitCallee, 2988 ArrayRef<llvm::Value *> ExitArgs, bool Conditional = false) 2989 : EnterCallee(EnterCallee), EnterArgs(EnterArgs), ExitCallee(ExitCallee), 2990 ExitArgs(ExitArgs), Conditional(Conditional) {} 2991 void Enter(CodeGenFunction &CGF) override { 2992 llvm::Value *EnterRes = CGF.EmitRuntimeCall(EnterCallee, EnterArgs); 2993 if (Conditional) { 2994 llvm::Value *CallBool = CGF.Builder.CreateIsNotNull(EnterRes); 2995 auto *ThenBlock = CGF.createBasicBlock("omp_if.then"); 2996 ContBlock = CGF.createBasicBlock("omp_if.end"); 2997 // Generate the branch (If-stmt) 2998 CGF.Builder.CreateCondBr(CallBool, ThenBlock, ContBlock); 2999 CGF.EmitBlock(ThenBlock); 3000 } 3001 } 3002 void Done(CodeGenFunction &CGF) { 3003 // Emit the rest of blocks/branches 3004 CGF.EmitBranch(ContBlock); 3005 CGF.EmitBlock(ContBlock, true); 3006 } 3007 void Exit(CodeGenFunction &CGF) override { 3008 CGF.EmitRuntimeCall(ExitCallee, ExitArgs); 3009 } 3010 }; 3011 } // anonymous namespace 3012 3013 void CGOpenMPRuntime::emitCriticalRegion(CodeGenFunction &CGF, 3014 StringRef CriticalName, 3015 const RegionCodeGenTy &CriticalOpGen, 3016 SourceLocation Loc, const Expr *Hint) { 3017 // __kmpc_critical[_with_hint](ident_t *, gtid, Lock[, hint]); 3018 // CriticalOpGen(); 3019 // __kmpc_end_critical(ident_t *, gtid, Lock); 3020 // Prepare arguments and build a call to __kmpc_critical 3021 if (!CGF.HaveInsertPoint()) 3022 return; 3023 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3024 getCriticalRegionLock(CriticalName)}; 3025 llvm::SmallVector<llvm::Value *, 4> EnterArgs(std::begin(Args), 3026 std::end(Args)); 3027 if (Hint) { 3028 EnterArgs.push_back(CGF.Builder.CreateIntCast( 3029 CGF.EmitScalarExpr(Hint), CGM.IntPtrTy, /*isSigned=*/false)); 3030 } 3031 CommonActionTy Action( 3032 createRuntimeFunction(Hint ? OMPRTL__kmpc_critical_with_hint 3033 : OMPRTL__kmpc_critical), 3034 EnterArgs, createRuntimeFunction(OMPRTL__kmpc_end_critical), Args); 3035 CriticalOpGen.setAction(Action); 3036 emitInlinedDirective(CGF, OMPD_critical, CriticalOpGen); 3037 } 3038 3039 void CGOpenMPRuntime::emitMasterRegion(CodeGenFunction &CGF, 3040 const RegionCodeGenTy &MasterOpGen, 3041 SourceLocation Loc) { 3042 if (!CGF.HaveInsertPoint()) 3043 return; 3044 // if(__kmpc_master(ident_t *, gtid)) { 3045 // MasterOpGen(); 3046 // __kmpc_end_master(ident_t *, gtid); 3047 // } 3048 // Prepare arguments and build a call to __kmpc_master 3049 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3050 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_master), Args, 3051 createRuntimeFunction(OMPRTL__kmpc_end_master), Args, 3052 /*Conditional=*/true); 3053 MasterOpGen.setAction(Action); 3054 emitInlinedDirective(CGF, OMPD_master, MasterOpGen); 3055 Action.Done(CGF); 3056 } 3057 3058 void CGOpenMPRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 3059 SourceLocation Loc) { 3060 if (!CGF.HaveInsertPoint()) 3061 return; 3062 // Build call __kmpc_omp_taskyield(loc, thread_id, 0); 3063 llvm::Value *Args[] = { 3064 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3065 llvm::ConstantInt::get(CGM.IntTy, /*V=*/0, /*isSigned=*/true)}; 3066 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskyield), Args); 3067 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 3068 Region->emitUntiedSwitch(CGF); 3069 } 3070 3071 void CGOpenMPRuntime::emitTaskgroupRegion(CodeGenFunction &CGF, 3072 const RegionCodeGenTy &TaskgroupOpGen, 3073 SourceLocation Loc) { 3074 if (!CGF.HaveInsertPoint()) 3075 return; 3076 // __kmpc_taskgroup(ident_t *, gtid); 3077 // TaskgroupOpGen(); 3078 // __kmpc_end_taskgroup(ident_t *, gtid); 3079 // Prepare arguments and build a call to __kmpc_taskgroup 3080 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3081 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_taskgroup), Args, 3082 createRuntimeFunction(OMPRTL__kmpc_end_taskgroup), 3083 Args); 3084 TaskgroupOpGen.setAction(Action); 3085 emitInlinedDirective(CGF, OMPD_taskgroup, TaskgroupOpGen); 3086 } 3087 3088 /// Given an array of pointers to variables, project the address of a 3089 /// given variable. 3090 static Address emitAddrOfVarFromArray(CodeGenFunction &CGF, Address Array, 3091 unsigned Index, const VarDecl *Var) { 3092 // Pull out the pointer to the variable. 3093 Address PtrAddr = 3094 CGF.Builder.CreateConstArrayGEP(Array, Index, CGF.getPointerSize()); 3095 llvm::Value *Ptr = CGF.Builder.CreateLoad(PtrAddr); 3096 3097 Address Addr = Address(Ptr, CGF.getContext().getDeclAlign(Var)); 3098 Addr = CGF.Builder.CreateElementBitCast( 3099 Addr, CGF.ConvertTypeForMem(Var->getType())); 3100 return Addr; 3101 } 3102 3103 static llvm::Value *emitCopyprivateCopyFunction( 3104 CodeGenModule &CGM, llvm::Type *ArgsType, 3105 ArrayRef<const Expr *> CopyprivateVars, ArrayRef<const Expr *> DestExprs, 3106 ArrayRef<const Expr *> SrcExprs, ArrayRef<const Expr *> AssignmentOps, 3107 SourceLocation Loc) { 3108 ASTContext &C = CGM.getContext(); 3109 // void copy_func(void *LHSArg, void *RHSArg); 3110 FunctionArgList Args; 3111 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 3112 ImplicitParamDecl::Other); 3113 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 3114 ImplicitParamDecl::Other); 3115 Args.push_back(&LHSArg); 3116 Args.push_back(&RHSArg); 3117 const auto &CGFI = 3118 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 3119 std::string Name = 3120 CGM.getOpenMPRuntime().getName({"omp", "copyprivate", "copy_func"}); 3121 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 3122 llvm::GlobalValue::InternalLinkage, Name, 3123 &CGM.getModule()); 3124 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 3125 Fn->setDoesNotRecurse(); 3126 CodeGenFunction CGF(CGM); 3127 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 3128 // Dest = (void*[n])(LHSArg); 3129 // Src = (void*[n])(RHSArg); 3130 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3131 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 3132 ArgsType), CGF.getPointerAlign()); 3133 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3134 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 3135 ArgsType), CGF.getPointerAlign()); 3136 // *(Type0*)Dst[0] = *(Type0*)Src[0]; 3137 // *(Type1*)Dst[1] = *(Type1*)Src[1]; 3138 // ... 3139 // *(Typen*)Dst[n] = *(Typen*)Src[n]; 3140 for (unsigned I = 0, E = AssignmentOps.size(); I < E; ++I) { 3141 const auto *DestVar = 3142 cast<VarDecl>(cast<DeclRefExpr>(DestExprs[I])->getDecl()); 3143 Address DestAddr = emitAddrOfVarFromArray(CGF, LHS, I, DestVar); 3144 3145 const auto *SrcVar = 3146 cast<VarDecl>(cast<DeclRefExpr>(SrcExprs[I])->getDecl()); 3147 Address SrcAddr = emitAddrOfVarFromArray(CGF, RHS, I, SrcVar); 3148 3149 const auto *VD = cast<DeclRefExpr>(CopyprivateVars[I])->getDecl(); 3150 QualType Type = VD->getType(); 3151 CGF.EmitOMPCopy(Type, DestAddr, SrcAddr, DestVar, SrcVar, AssignmentOps[I]); 3152 } 3153 CGF.FinishFunction(); 3154 return Fn; 3155 } 3156 3157 void CGOpenMPRuntime::emitSingleRegion(CodeGenFunction &CGF, 3158 const RegionCodeGenTy &SingleOpGen, 3159 SourceLocation Loc, 3160 ArrayRef<const Expr *> CopyprivateVars, 3161 ArrayRef<const Expr *> SrcExprs, 3162 ArrayRef<const Expr *> DstExprs, 3163 ArrayRef<const Expr *> AssignmentOps) { 3164 if (!CGF.HaveInsertPoint()) 3165 return; 3166 assert(CopyprivateVars.size() == SrcExprs.size() && 3167 CopyprivateVars.size() == DstExprs.size() && 3168 CopyprivateVars.size() == AssignmentOps.size()); 3169 ASTContext &C = CGM.getContext(); 3170 // int32 did_it = 0; 3171 // if(__kmpc_single(ident_t *, gtid)) { 3172 // SingleOpGen(); 3173 // __kmpc_end_single(ident_t *, gtid); 3174 // did_it = 1; 3175 // } 3176 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 3177 // <copy_func>, did_it); 3178 3179 Address DidIt = Address::invalid(); 3180 if (!CopyprivateVars.empty()) { 3181 // int32 did_it = 0; 3182 QualType KmpInt32Ty = 3183 C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 3184 DidIt = CGF.CreateMemTemp(KmpInt32Ty, ".omp.copyprivate.did_it"); 3185 CGF.Builder.CreateStore(CGF.Builder.getInt32(0), DidIt); 3186 } 3187 // Prepare arguments and build a call to __kmpc_single 3188 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3189 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_single), Args, 3190 createRuntimeFunction(OMPRTL__kmpc_end_single), Args, 3191 /*Conditional=*/true); 3192 SingleOpGen.setAction(Action); 3193 emitInlinedDirective(CGF, OMPD_single, SingleOpGen); 3194 if (DidIt.isValid()) { 3195 // did_it = 1; 3196 CGF.Builder.CreateStore(CGF.Builder.getInt32(1), DidIt); 3197 } 3198 Action.Done(CGF); 3199 // call __kmpc_copyprivate(ident_t *, gtid, <buf_size>, <copyprivate list>, 3200 // <copy_func>, did_it); 3201 if (DidIt.isValid()) { 3202 llvm::APInt ArraySize(/*unsigned int numBits=*/32, CopyprivateVars.size()); 3203 QualType CopyprivateArrayTy = 3204 C.getConstantArrayType(C.VoidPtrTy, ArraySize, ArrayType::Normal, 3205 /*IndexTypeQuals=*/0); 3206 // Create a list of all private variables for copyprivate. 3207 Address CopyprivateList = 3208 CGF.CreateMemTemp(CopyprivateArrayTy, ".omp.copyprivate.cpr_list"); 3209 for (unsigned I = 0, E = CopyprivateVars.size(); I < E; ++I) { 3210 Address Elem = CGF.Builder.CreateConstArrayGEP( 3211 CopyprivateList, I, CGF.getPointerSize()); 3212 CGF.Builder.CreateStore( 3213 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 3214 CGF.EmitLValue(CopyprivateVars[I]).getPointer(), CGF.VoidPtrTy), 3215 Elem); 3216 } 3217 // Build function that copies private values from single region to all other 3218 // threads in the corresponding parallel region. 3219 llvm::Value *CpyFn = emitCopyprivateCopyFunction( 3220 CGM, CGF.ConvertTypeForMem(CopyprivateArrayTy)->getPointerTo(), 3221 CopyprivateVars, SrcExprs, DstExprs, AssignmentOps, Loc); 3222 llvm::Value *BufSize = CGF.getTypeSize(CopyprivateArrayTy); 3223 Address CL = 3224 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(CopyprivateList, 3225 CGF.VoidPtrTy); 3226 llvm::Value *DidItVal = CGF.Builder.CreateLoad(DidIt); 3227 llvm::Value *Args[] = { 3228 emitUpdateLocation(CGF, Loc), // ident_t *<loc> 3229 getThreadID(CGF, Loc), // i32 <gtid> 3230 BufSize, // size_t <buf_size> 3231 CL.getPointer(), // void *<copyprivate list> 3232 CpyFn, // void (*) (void *, void *) <copy_func> 3233 DidItVal // i32 did_it 3234 }; 3235 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_copyprivate), Args); 3236 } 3237 } 3238 3239 void CGOpenMPRuntime::emitOrderedRegion(CodeGenFunction &CGF, 3240 const RegionCodeGenTy &OrderedOpGen, 3241 SourceLocation Loc, bool IsThreads) { 3242 if (!CGF.HaveInsertPoint()) 3243 return; 3244 // __kmpc_ordered(ident_t *, gtid); 3245 // OrderedOpGen(); 3246 // __kmpc_end_ordered(ident_t *, gtid); 3247 // Prepare arguments and build a call to __kmpc_ordered 3248 if (IsThreads) { 3249 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3250 CommonActionTy Action(createRuntimeFunction(OMPRTL__kmpc_ordered), Args, 3251 createRuntimeFunction(OMPRTL__kmpc_end_ordered), 3252 Args); 3253 OrderedOpGen.setAction(Action); 3254 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 3255 return; 3256 } 3257 emitInlinedDirective(CGF, OMPD_ordered, OrderedOpGen); 3258 } 3259 3260 unsigned CGOpenMPRuntime::getDefaultFlagsForBarriers(OpenMPDirectiveKind Kind) { 3261 unsigned Flags; 3262 if (Kind == OMPD_for) 3263 Flags = OMP_IDENT_BARRIER_IMPL_FOR; 3264 else if (Kind == OMPD_sections) 3265 Flags = OMP_IDENT_BARRIER_IMPL_SECTIONS; 3266 else if (Kind == OMPD_single) 3267 Flags = OMP_IDENT_BARRIER_IMPL_SINGLE; 3268 else if (Kind == OMPD_barrier) 3269 Flags = OMP_IDENT_BARRIER_EXPL; 3270 else 3271 Flags = OMP_IDENT_BARRIER_IMPL; 3272 return Flags; 3273 } 3274 3275 void CGOpenMPRuntime::emitBarrierCall(CodeGenFunction &CGF, SourceLocation Loc, 3276 OpenMPDirectiveKind Kind, bool EmitChecks, 3277 bool ForceSimpleCall) { 3278 if (!CGF.HaveInsertPoint()) 3279 return; 3280 // Build call __kmpc_cancel_barrier(loc, thread_id); 3281 // Build call __kmpc_barrier(loc, thread_id); 3282 unsigned Flags = getDefaultFlagsForBarriers(Kind); 3283 // Build call __kmpc_cancel_barrier(loc, thread_id) or __kmpc_barrier(loc, 3284 // thread_id); 3285 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc, Flags), 3286 getThreadID(CGF, Loc)}; 3287 if (auto *OMPRegionInfo = 3288 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 3289 if (!ForceSimpleCall && OMPRegionInfo->hasCancel()) { 3290 llvm::Value *Result = CGF.EmitRuntimeCall( 3291 createRuntimeFunction(OMPRTL__kmpc_cancel_barrier), Args); 3292 if (EmitChecks) { 3293 // if (__kmpc_cancel_barrier()) { 3294 // exit from construct; 3295 // } 3296 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 3297 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 3298 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 3299 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 3300 CGF.EmitBlock(ExitBB); 3301 // exit from construct; 3302 CodeGenFunction::JumpDest CancelDestination = 3303 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 3304 CGF.EmitBranchThroughCleanup(CancelDestination); 3305 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 3306 } 3307 return; 3308 } 3309 } 3310 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_barrier), Args); 3311 } 3312 3313 /// Map the OpenMP loop schedule to the runtime enumeration. 3314 static OpenMPSchedType getRuntimeSchedule(OpenMPScheduleClauseKind ScheduleKind, 3315 bool Chunked, bool Ordered) { 3316 switch (ScheduleKind) { 3317 case OMPC_SCHEDULE_static: 3318 return Chunked ? (Ordered ? OMP_ord_static_chunked : OMP_sch_static_chunked) 3319 : (Ordered ? OMP_ord_static : OMP_sch_static); 3320 case OMPC_SCHEDULE_dynamic: 3321 return Ordered ? OMP_ord_dynamic_chunked : OMP_sch_dynamic_chunked; 3322 case OMPC_SCHEDULE_guided: 3323 return Ordered ? OMP_ord_guided_chunked : OMP_sch_guided_chunked; 3324 case OMPC_SCHEDULE_runtime: 3325 return Ordered ? OMP_ord_runtime : OMP_sch_runtime; 3326 case OMPC_SCHEDULE_auto: 3327 return Ordered ? OMP_ord_auto : OMP_sch_auto; 3328 case OMPC_SCHEDULE_unknown: 3329 assert(!Chunked && "chunk was specified but schedule kind not known"); 3330 return Ordered ? OMP_ord_static : OMP_sch_static; 3331 } 3332 llvm_unreachable("Unexpected runtime schedule"); 3333 } 3334 3335 /// Map the OpenMP distribute schedule to the runtime enumeration. 3336 static OpenMPSchedType 3337 getRuntimeSchedule(OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) { 3338 // only static is allowed for dist_schedule 3339 return Chunked ? OMP_dist_sch_static_chunked : OMP_dist_sch_static; 3340 } 3341 3342 bool CGOpenMPRuntime::isStaticNonchunked(OpenMPScheduleClauseKind ScheduleKind, 3343 bool Chunked) const { 3344 OpenMPSchedType Schedule = 3345 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 3346 return Schedule == OMP_sch_static; 3347 } 3348 3349 bool CGOpenMPRuntime::isStaticNonchunked( 3350 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 3351 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 3352 return Schedule == OMP_dist_sch_static; 3353 } 3354 3355 bool CGOpenMPRuntime::isStaticChunked(OpenMPScheduleClauseKind ScheduleKind, 3356 bool Chunked) const { 3357 OpenMPSchedType Schedule = 3358 getRuntimeSchedule(ScheduleKind, Chunked, /*Ordered=*/false); 3359 return Schedule == OMP_sch_static_chunked; 3360 } 3361 3362 bool CGOpenMPRuntime::isStaticChunked( 3363 OpenMPDistScheduleClauseKind ScheduleKind, bool Chunked) const { 3364 OpenMPSchedType Schedule = getRuntimeSchedule(ScheduleKind, Chunked); 3365 return Schedule == OMP_dist_sch_static_chunked; 3366 } 3367 3368 bool CGOpenMPRuntime::isDynamic(OpenMPScheduleClauseKind ScheduleKind) const { 3369 OpenMPSchedType Schedule = 3370 getRuntimeSchedule(ScheduleKind, /*Chunked=*/false, /*Ordered=*/false); 3371 assert(Schedule != OMP_sch_static_chunked && "cannot be chunked here"); 3372 return Schedule != OMP_sch_static; 3373 } 3374 3375 static int addMonoNonMonoModifier(OpenMPSchedType Schedule, 3376 OpenMPScheduleClauseModifier M1, 3377 OpenMPScheduleClauseModifier M2) { 3378 int Modifier = 0; 3379 switch (M1) { 3380 case OMPC_SCHEDULE_MODIFIER_monotonic: 3381 Modifier = OMP_sch_modifier_monotonic; 3382 break; 3383 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 3384 Modifier = OMP_sch_modifier_nonmonotonic; 3385 break; 3386 case OMPC_SCHEDULE_MODIFIER_simd: 3387 if (Schedule == OMP_sch_static_chunked) 3388 Schedule = OMP_sch_static_balanced_chunked; 3389 break; 3390 case OMPC_SCHEDULE_MODIFIER_last: 3391 case OMPC_SCHEDULE_MODIFIER_unknown: 3392 break; 3393 } 3394 switch (M2) { 3395 case OMPC_SCHEDULE_MODIFIER_monotonic: 3396 Modifier = OMP_sch_modifier_monotonic; 3397 break; 3398 case OMPC_SCHEDULE_MODIFIER_nonmonotonic: 3399 Modifier = OMP_sch_modifier_nonmonotonic; 3400 break; 3401 case OMPC_SCHEDULE_MODIFIER_simd: 3402 if (Schedule == OMP_sch_static_chunked) 3403 Schedule = OMP_sch_static_balanced_chunked; 3404 break; 3405 case OMPC_SCHEDULE_MODIFIER_last: 3406 case OMPC_SCHEDULE_MODIFIER_unknown: 3407 break; 3408 } 3409 return Schedule | Modifier; 3410 } 3411 3412 void CGOpenMPRuntime::emitForDispatchInit( 3413 CodeGenFunction &CGF, SourceLocation Loc, 3414 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 3415 bool Ordered, const DispatchRTInput &DispatchValues) { 3416 if (!CGF.HaveInsertPoint()) 3417 return; 3418 OpenMPSchedType Schedule = getRuntimeSchedule( 3419 ScheduleKind.Schedule, DispatchValues.Chunk != nullptr, Ordered); 3420 assert(Ordered || 3421 (Schedule != OMP_sch_static && Schedule != OMP_sch_static_chunked && 3422 Schedule != OMP_ord_static && Schedule != OMP_ord_static_chunked && 3423 Schedule != OMP_sch_static_balanced_chunked)); 3424 // Call __kmpc_dispatch_init( 3425 // ident_t *loc, kmp_int32 tid, kmp_int32 schedule, 3426 // kmp_int[32|64] lower, kmp_int[32|64] upper, 3427 // kmp_int[32|64] stride, kmp_int[32|64] chunk); 3428 3429 // If the Chunk was not specified in the clause - use default value 1. 3430 llvm::Value *Chunk = DispatchValues.Chunk ? DispatchValues.Chunk 3431 : CGF.Builder.getIntN(IVSize, 1); 3432 llvm::Value *Args[] = { 3433 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3434 CGF.Builder.getInt32(addMonoNonMonoModifier( 3435 Schedule, ScheduleKind.M1, ScheduleKind.M2)), // Schedule type 3436 DispatchValues.LB, // Lower 3437 DispatchValues.UB, // Upper 3438 CGF.Builder.getIntN(IVSize, 1), // Stride 3439 Chunk // Chunk 3440 }; 3441 CGF.EmitRuntimeCall(createDispatchInitFunction(IVSize, IVSigned), Args); 3442 } 3443 3444 static void emitForStaticInitCall( 3445 CodeGenFunction &CGF, llvm::Value *UpdateLocation, llvm::Value *ThreadId, 3446 llvm::FunctionCallee ForStaticInitFunction, OpenMPSchedType Schedule, 3447 OpenMPScheduleClauseModifier M1, OpenMPScheduleClauseModifier M2, 3448 const CGOpenMPRuntime::StaticRTInput &Values) { 3449 if (!CGF.HaveInsertPoint()) 3450 return; 3451 3452 assert(!Values.Ordered); 3453 assert(Schedule == OMP_sch_static || Schedule == OMP_sch_static_chunked || 3454 Schedule == OMP_sch_static_balanced_chunked || 3455 Schedule == OMP_ord_static || Schedule == OMP_ord_static_chunked || 3456 Schedule == OMP_dist_sch_static || 3457 Schedule == OMP_dist_sch_static_chunked); 3458 3459 // Call __kmpc_for_static_init( 3460 // ident_t *loc, kmp_int32 tid, kmp_int32 schedtype, 3461 // kmp_int32 *p_lastiter, kmp_int[32|64] *p_lower, 3462 // kmp_int[32|64] *p_upper, kmp_int[32|64] *p_stride, 3463 // kmp_int[32|64] incr, kmp_int[32|64] chunk); 3464 llvm::Value *Chunk = Values.Chunk; 3465 if (Chunk == nullptr) { 3466 assert((Schedule == OMP_sch_static || Schedule == OMP_ord_static || 3467 Schedule == OMP_dist_sch_static) && 3468 "expected static non-chunked schedule"); 3469 // If the Chunk was not specified in the clause - use default value 1. 3470 Chunk = CGF.Builder.getIntN(Values.IVSize, 1); 3471 } else { 3472 assert((Schedule == OMP_sch_static_chunked || 3473 Schedule == OMP_sch_static_balanced_chunked || 3474 Schedule == OMP_ord_static_chunked || 3475 Schedule == OMP_dist_sch_static_chunked) && 3476 "expected static chunked schedule"); 3477 } 3478 llvm::Value *Args[] = { 3479 UpdateLocation, 3480 ThreadId, 3481 CGF.Builder.getInt32(addMonoNonMonoModifier(Schedule, M1, 3482 M2)), // Schedule type 3483 Values.IL.getPointer(), // &isLastIter 3484 Values.LB.getPointer(), // &LB 3485 Values.UB.getPointer(), // &UB 3486 Values.ST.getPointer(), // &Stride 3487 CGF.Builder.getIntN(Values.IVSize, 1), // Incr 3488 Chunk // Chunk 3489 }; 3490 CGF.EmitRuntimeCall(ForStaticInitFunction, Args); 3491 } 3492 3493 void CGOpenMPRuntime::emitForStaticInit(CodeGenFunction &CGF, 3494 SourceLocation Loc, 3495 OpenMPDirectiveKind DKind, 3496 const OpenMPScheduleTy &ScheduleKind, 3497 const StaticRTInput &Values) { 3498 OpenMPSchedType ScheduleNum = getRuntimeSchedule( 3499 ScheduleKind.Schedule, Values.Chunk != nullptr, Values.Ordered); 3500 assert(isOpenMPWorksharingDirective(DKind) && 3501 "Expected loop-based or sections-based directive."); 3502 llvm::Value *UpdatedLocation = emitUpdateLocation(CGF, Loc, 3503 isOpenMPLoopDirective(DKind) 3504 ? OMP_IDENT_WORK_LOOP 3505 : OMP_IDENT_WORK_SECTIONS); 3506 llvm::Value *ThreadId = getThreadID(CGF, Loc); 3507 llvm::FunctionCallee StaticInitFunction = 3508 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 3509 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 3510 ScheduleNum, ScheduleKind.M1, ScheduleKind.M2, Values); 3511 } 3512 3513 void CGOpenMPRuntime::emitDistributeStaticInit( 3514 CodeGenFunction &CGF, SourceLocation Loc, 3515 OpenMPDistScheduleClauseKind SchedKind, 3516 const CGOpenMPRuntime::StaticRTInput &Values) { 3517 OpenMPSchedType ScheduleNum = 3518 getRuntimeSchedule(SchedKind, Values.Chunk != nullptr); 3519 llvm::Value *UpdatedLocation = 3520 emitUpdateLocation(CGF, Loc, OMP_IDENT_WORK_DISTRIBUTE); 3521 llvm::Value *ThreadId = getThreadID(CGF, Loc); 3522 llvm::FunctionCallee StaticInitFunction = 3523 createForStaticInitFunction(Values.IVSize, Values.IVSigned); 3524 emitForStaticInitCall(CGF, UpdatedLocation, ThreadId, StaticInitFunction, 3525 ScheduleNum, OMPC_SCHEDULE_MODIFIER_unknown, 3526 OMPC_SCHEDULE_MODIFIER_unknown, Values); 3527 } 3528 3529 void CGOpenMPRuntime::emitForStaticFinish(CodeGenFunction &CGF, 3530 SourceLocation Loc, 3531 OpenMPDirectiveKind DKind) { 3532 if (!CGF.HaveInsertPoint()) 3533 return; 3534 // Call __kmpc_for_static_fini(ident_t *loc, kmp_int32 tid); 3535 llvm::Value *Args[] = { 3536 emitUpdateLocation(CGF, Loc, 3537 isOpenMPDistributeDirective(DKind) 3538 ? OMP_IDENT_WORK_DISTRIBUTE 3539 : isOpenMPLoopDirective(DKind) 3540 ? OMP_IDENT_WORK_LOOP 3541 : OMP_IDENT_WORK_SECTIONS), 3542 getThreadID(CGF, Loc)}; 3543 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_for_static_fini), 3544 Args); 3545 } 3546 3547 void CGOpenMPRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 3548 SourceLocation Loc, 3549 unsigned IVSize, 3550 bool IVSigned) { 3551 if (!CGF.HaveInsertPoint()) 3552 return; 3553 // Call __kmpc_for_dynamic_fini_(4|8)[u](ident_t *loc, kmp_int32 tid); 3554 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 3555 CGF.EmitRuntimeCall(createDispatchFiniFunction(IVSize, IVSigned), Args); 3556 } 3557 3558 llvm::Value *CGOpenMPRuntime::emitForNext(CodeGenFunction &CGF, 3559 SourceLocation Loc, unsigned IVSize, 3560 bool IVSigned, Address IL, 3561 Address LB, Address UB, 3562 Address ST) { 3563 // Call __kmpc_dispatch_next( 3564 // ident_t *loc, kmp_int32 tid, kmp_int32 *p_lastiter, 3565 // kmp_int[32|64] *p_lower, kmp_int[32|64] *p_upper, 3566 // kmp_int[32|64] *p_stride); 3567 llvm::Value *Args[] = { 3568 emitUpdateLocation(CGF, Loc), 3569 getThreadID(CGF, Loc), 3570 IL.getPointer(), // &isLastIter 3571 LB.getPointer(), // &Lower 3572 UB.getPointer(), // &Upper 3573 ST.getPointer() // &Stride 3574 }; 3575 llvm::Value *Call = 3576 CGF.EmitRuntimeCall(createDispatchNextFunction(IVSize, IVSigned), Args); 3577 return CGF.EmitScalarConversion( 3578 Call, CGF.getContext().getIntTypeForBitwidth(32, /*Signed=*/1), 3579 CGF.getContext().BoolTy, Loc); 3580 } 3581 3582 void CGOpenMPRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 3583 llvm::Value *NumThreads, 3584 SourceLocation Loc) { 3585 if (!CGF.HaveInsertPoint()) 3586 return; 3587 // Build call __kmpc_push_num_threads(&loc, global_tid, num_threads) 3588 llvm::Value *Args[] = { 3589 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3590 CGF.Builder.CreateIntCast(NumThreads, CGF.Int32Ty, /*isSigned*/ true)}; 3591 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_threads), 3592 Args); 3593 } 3594 3595 void CGOpenMPRuntime::emitProcBindClause(CodeGenFunction &CGF, 3596 OpenMPProcBindClauseKind ProcBind, 3597 SourceLocation Loc) { 3598 if (!CGF.HaveInsertPoint()) 3599 return; 3600 // Constants for proc bind value accepted by the runtime. 3601 enum ProcBindTy { 3602 ProcBindFalse = 0, 3603 ProcBindTrue, 3604 ProcBindMaster, 3605 ProcBindClose, 3606 ProcBindSpread, 3607 ProcBindIntel, 3608 ProcBindDefault 3609 } RuntimeProcBind; 3610 switch (ProcBind) { 3611 case OMPC_PROC_BIND_master: 3612 RuntimeProcBind = ProcBindMaster; 3613 break; 3614 case OMPC_PROC_BIND_close: 3615 RuntimeProcBind = ProcBindClose; 3616 break; 3617 case OMPC_PROC_BIND_spread: 3618 RuntimeProcBind = ProcBindSpread; 3619 break; 3620 case OMPC_PROC_BIND_unknown: 3621 llvm_unreachable("Unsupported proc_bind value."); 3622 } 3623 // Build call __kmpc_push_proc_bind(&loc, global_tid, proc_bind) 3624 llvm::Value *Args[] = { 3625 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 3626 llvm::ConstantInt::get(CGM.IntTy, RuntimeProcBind, /*isSigned=*/true)}; 3627 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_proc_bind), Args); 3628 } 3629 3630 void CGOpenMPRuntime::emitFlush(CodeGenFunction &CGF, ArrayRef<const Expr *>, 3631 SourceLocation Loc) { 3632 if (!CGF.HaveInsertPoint()) 3633 return; 3634 // Build call void __kmpc_flush(ident_t *loc) 3635 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_flush), 3636 emitUpdateLocation(CGF, Loc)); 3637 } 3638 3639 namespace { 3640 /// Indexes of fields for type kmp_task_t. 3641 enum KmpTaskTFields { 3642 /// List of shared variables. 3643 KmpTaskTShareds, 3644 /// Task routine. 3645 KmpTaskTRoutine, 3646 /// Partition id for the untied tasks. 3647 KmpTaskTPartId, 3648 /// Function with call of destructors for private variables. 3649 Data1, 3650 /// Task priority. 3651 Data2, 3652 /// (Taskloops only) Lower bound. 3653 KmpTaskTLowerBound, 3654 /// (Taskloops only) Upper bound. 3655 KmpTaskTUpperBound, 3656 /// (Taskloops only) Stride. 3657 KmpTaskTStride, 3658 /// (Taskloops only) Is last iteration flag. 3659 KmpTaskTLastIter, 3660 /// (Taskloops only) Reduction data. 3661 KmpTaskTReductions, 3662 }; 3663 } // anonymous namespace 3664 3665 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::empty() const { 3666 return OffloadEntriesTargetRegion.empty() && 3667 OffloadEntriesDeviceGlobalVar.empty(); 3668 } 3669 3670 /// Initialize target region entry. 3671 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3672 initializeTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3673 StringRef ParentName, unsigned LineNum, 3674 unsigned Order) { 3675 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3676 "only required for the device " 3677 "code generation."); 3678 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = 3679 OffloadEntryInfoTargetRegion(Order, /*Addr=*/nullptr, /*ID=*/nullptr, 3680 OMPTargetRegionEntryTargetRegion); 3681 ++OffloadingEntriesNum; 3682 } 3683 3684 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3685 registerTargetRegionEntryInfo(unsigned DeviceID, unsigned FileID, 3686 StringRef ParentName, unsigned LineNum, 3687 llvm::Constant *Addr, llvm::Constant *ID, 3688 OMPTargetRegionEntryKind Flags) { 3689 // If we are emitting code for a target, the entry is already initialized, 3690 // only has to be registered. 3691 if (CGM.getLangOpts().OpenMPIsDevice) { 3692 if (!hasTargetRegionEntryInfo(DeviceID, FileID, ParentName, LineNum)) { 3693 unsigned DiagID = CGM.getDiags().getCustomDiagID( 3694 DiagnosticsEngine::Error, 3695 "Unable to find target region on line '%0' in the device code."); 3696 CGM.getDiags().Report(DiagID) << LineNum; 3697 return; 3698 } 3699 auto &Entry = 3700 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum]; 3701 assert(Entry.isValid() && "Entry not initialized!"); 3702 Entry.setAddress(Addr); 3703 Entry.setID(ID); 3704 Entry.setFlags(Flags); 3705 } else { 3706 OffloadEntryInfoTargetRegion Entry(OffloadingEntriesNum, Addr, ID, Flags); 3707 OffloadEntriesTargetRegion[DeviceID][FileID][ParentName][LineNum] = Entry; 3708 ++OffloadingEntriesNum; 3709 } 3710 } 3711 3712 bool CGOpenMPRuntime::OffloadEntriesInfoManagerTy::hasTargetRegionEntryInfo( 3713 unsigned DeviceID, unsigned FileID, StringRef ParentName, 3714 unsigned LineNum) const { 3715 auto PerDevice = OffloadEntriesTargetRegion.find(DeviceID); 3716 if (PerDevice == OffloadEntriesTargetRegion.end()) 3717 return false; 3718 auto PerFile = PerDevice->second.find(FileID); 3719 if (PerFile == PerDevice->second.end()) 3720 return false; 3721 auto PerParentName = PerFile->second.find(ParentName); 3722 if (PerParentName == PerFile->second.end()) 3723 return false; 3724 auto PerLine = PerParentName->second.find(LineNum); 3725 if (PerLine == PerParentName->second.end()) 3726 return false; 3727 // Fail if this entry is already registered. 3728 if (PerLine->second.getAddress() || PerLine->second.getID()) 3729 return false; 3730 return true; 3731 } 3732 3733 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy::actOnTargetRegionEntriesInfo( 3734 const OffloadTargetRegionEntryInfoActTy &Action) { 3735 // Scan all target region entries and perform the provided action. 3736 for (const auto &D : OffloadEntriesTargetRegion) 3737 for (const auto &F : D.second) 3738 for (const auto &P : F.second) 3739 for (const auto &L : P.second) 3740 Action(D.first, F.first, P.first(), L.first, L.second); 3741 } 3742 3743 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3744 initializeDeviceGlobalVarEntryInfo(StringRef Name, 3745 OMPTargetGlobalVarEntryKind Flags, 3746 unsigned Order) { 3747 assert(CGM.getLangOpts().OpenMPIsDevice && "Initialization of entries is " 3748 "only required for the device " 3749 "code generation."); 3750 OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags); 3751 ++OffloadingEntriesNum; 3752 } 3753 3754 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3755 registerDeviceGlobalVarEntryInfo(StringRef VarName, llvm::Constant *Addr, 3756 CharUnits VarSize, 3757 OMPTargetGlobalVarEntryKind Flags, 3758 llvm::GlobalValue::LinkageTypes Linkage) { 3759 if (CGM.getLangOpts().OpenMPIsDevice) { 3760 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName]; 3761 assert(Entry.isValid() && Entry.getFlags() == Flags && 3762 "Entry not initialized!"); 3763 assert((!Entry.getAddress() || Entry.getAddress() == Addr) && 3764 "Resetting with the new address."); 3765 if (Entry.getAddress() && hasDeviceGlobalVarEntryInfo(VarName)) 3766 return; 3767 Entry.setAddress(Addr); 3768 Entry.setVarSize(VarSize); 3769 Entry.setLinkage(Linkage); 3770 } else { 3771 if (hasDeviceGlobalVarEntryInfo(VarName)) 3772 return; 3773 OffloadEntriesDeviceGlobalVar.try_emplace( 3774 VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage); 3775 ++OffloadingEntriesNum; 3776 } 3777 } 3778 3779 void CGOpenMPRuntime::OffloadEntriesInfoManagerTy:: 3780 actOnDeviceGlobalVarEntriesInfo( 3781 const OffloadDeviceGlobalVarEntryInfoActTy &Action) { 3782 // Scan all target region entries and perform the provided action. 3783 for (const auto &E : OffloadEntriesDeviceGlobalVar) 3784 Action(E.getKey(), E.getValue()); 3785 } 3786 3787 llvm::Function * 3788 CGOpenMPRuntime::createOffloadingBinaryDescriptorRegistration() { 3789 // If we don't have entries or if we are emitting code for the device, we 3790 // don't need to do anything. 3791 if (CGM.getLangOpts().OpenMPIsDevice || OffloadEntriesInfoManager.empty()) 3792 return nullptr; 3793 3794 llvm::Module &M = CGM.getModule(); 3795 ASTContext &C = CGM.getContext(); 3796 3797 // Get list of devices we care about 3798 const std::vector<llvm::Triple> &Devices = CGM.getLangOpts().OMPTargetTriples; 3799 3800 // We should be creating an offloading descriptor only if there are devices 3801 // specified. 3802 assert(!Devices.empty() && "No OpenMP offloading devices??"); 3803 3804 // Create the external variables that will point to the begin and end of the 3805 // host entries section. These will be defined by the linker. 3806 llvm::Type *OffloadEntryTy = 3807 CGM.getTypes().ConvertTypeForMem(getTgtOffloadEntryQTy()); 3808 std::string EntriesBeginName = getName({"omp_offloading", "entries_begin"}); 3809 auto *HostEntriesBegin = new llvm::GlobalVariable( 3810 M, OffloadEntryTy, /*isConstant=*/true, 3811 llvm::GlobalValue::ExternalLinkage, /*Initializer=*/nullptr, 3812 EntriesBeginName); 3813 std::string EntriesEndName = getName({"omp_offloading", "entries_end"}); 3814 auto *HostEntriesEnd = 3815 new llvm::GlobalVariable(M, OffloadEntryTy, /*isConstant=*/true, 3816 llvm::GlobalValue::ExternalLinkage, 3817 /*Initializer=*/nullptr, EntriesEndName); 3818 3819 // Create all device images 3820 auto *DeviceImageTy = cast<llvm::StructType>( 3821 CGM.getTypes().ConvertTypeForMem(getTgtDeviceImageQTy())); 3822 ConstantInitBuilder DeviceImagesBuilder(CGM); 3823 ConstantArrayBuilder DeviceImagesEntries = 3824 DeviceImagesBuilder.beginArray(DeviceImageTy); 3825 3826 for (const llvm::Triple &Device : Devices) { 3827 StringRef T = Device.getTriple(); 3828 std::string BeginName = getName({"omp_offloading", "img_start", ""}); 3829 auto *ImgBegin = new llvm::GlobalVariable( 3830 M, CGM.Int8Ty, /*isConstant=*/true, 3831 llvm::GlobalValue::ExternalWeakLinkage, 3832 /*Initializer=*/nullptr, Twine(BeginName).concat(T)); 3833 std::string EndName = getName({"omp_offloading", "img_end", ""}); 3834 auto *ImgEnd = new llvm::GlobalVariable( 3835 M, CGM.Int8Ty, /*isConstant=*/true, 3836 llvm::GlobalValue::ExternalWeakLinkage, 3837 /*Initializer=*/nullptr, Twine(EndName).concat(T)); 3838 3839 llvm::Constant *Data[] = {ImgBegin, ImgEnd, HostEntriesBegin, 3840 HostEntriesEnd}; 3841 createConstantGlobalStructAndAddToParent(CGM, getTgtDeviceImageQTy(), Data, 3842 DeviceImagesEntries); 3843 } 3844 3845 // Create device images global array. 3846 std::string ImagesName = getName({"omp_offloading", "device_images"}); 3847 llvm::GlobalVariable *DeviceImages = 3848 DeviceImagesEntries.finishAndCreateGlobal(ImagesName, 3849 CGM.getPointerAlign(), 3850 /*isConstant=*/true); 3851 DeviceImages->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 3852 3853 // This is a Zero array to be used in the creation of the constant expressions 3854 llvm::Constant *Index[] = {llvm::Constant::getNullValue(CGM.Int32Ty), 3855 llvm::Constant::getNullValue(CGM.Int32Ty)}; 3856 3857 // Create the target region descriptor. 3858 llvm::Constant *Data[] = { 3859 llvm::ConstantInt::get(CGM.Int32Ty, Devices.size()), 3860 llvm::ConstantExpr::getGetElementPtr(DeviceImages->getValueType(), 3861 DeviceImages, Index), 3862 HostEntriesBegin, HostEntriesEnd}; 3863 std::string Descriptor = getName({"omp_offloading", "descriptor"}); 3864 llvm::GlobalVariable *Desc = createGlobalStruct( 3865 CGM, getTgtBinaryDescriptorQTy(), /*IsConstant=*/true, Data, Descriptor); 3866 3867 // Emit code to register or unregister the descriptor at execution 3868 // startup or closing, respectively. 3869 3870 llvm::Function *UnRegFn; 3871 { 3872 FunctionArgList Args; 3873 ImplicitParamDecl DummyPtr(C, C.VoidPtrTy, ImplicitParamDecl::Other); 3874 Args.push_back(&DummyPtr); 3875 3876 CodeGenFunction CGF(CGM); 3877 // Disable debug info for global (de-)initializer because they are not part 3878 // of some particular construct. 3879 CGF.disableDebugInfo(); 3880 const auto &FI = 3881 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 3882 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 3883 std::string UnregName = getName({"omp_offloading", "descriptor_unreg"}); 3884 UnRegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, UnregName, FI); 3885 CGF.StartFunction(GlobalDecl(), C.VoidTy, UnRegFn, FI, Args); 3886 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_unregister_lib), 3887 Desc); 3888 CGF.FinishFunction(); 3889 } 3890 llvm::Function *RegFn; 3891 { 3892 CodeGenFunction CGF(CGM); 3893 // Disable debug info for global (de-)initializer because they are not part 3894 // of some particular construct. 3895 CGF.disableDebugInfo(); 3896 const auto &FI = CGM.getTypes().arrangeNullaryFunction(); 3897 llvm::FunctionType *FTy = CGM.getTypes().GetFunctionType(FI); 3898 3899 // Encode offload target triples into the registration function name. It 3900 // will serve as a comdat key for the registration/unregistration code for 3901 // this particular combination of offloading targets. 3902 SmallVector<StringRef, 4U> RegFnNameParts(Devices.size() + 2U); 3903 RegFnNameParts[0] = "omp_offloading"; 3904 RegFnNameParts[1] = "descriptor_reg"; 3905 llvm::transform(Devices, std::next(RegFnNameParts.begin(), 2), 3906 [](const llvm::Triple &T) -> const std::string& { 3907 return T.getTriple(); 3908 }); 3909 llvm::sort(std::next(RegFnNameParts.begin(), 2), RegFnNameParts.end()); 3910 std::string Descriptor = getName(RegFnNameParts); 3911 RegFn = CGM.CreateGlobalInitOrDestructFunction(FTy, Descriptor, FI); 3912 CGF.StartFunction(GlobalDecl(), C.VoidTy, RegFn, FI, FunctionArgList()); 3913 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_register_lib), Desc); 3914 // Create a variable to drive the registration and unregistration of the 3915 // descriptor, so we can reuse the logic that emits Ctors and Dtors. 3916 ImplicitParamDecl RegUnregVar(C, C.getTranslationUnitDecl(), 3917 SourceLocation(), nullptr, C.CharTy, 3918 ImplicitParamDecl::Other); 3919 CGM.getCXXABI().registerGlobalDtor(CGF, RegUnregVar, UnRegFn, Desc); 3920 CGF.FinishFunction(); 3921 } 3922 if (CGM.supportsCOMDAT()) { 3923 // It is sufficient to call registration function only once, so create a 3924 // COMDAT group for registration/unregistration functions and associated 3925 // data. That would reduce startup time and code size. Registration 3926 // function serves as a COMDAT group key. 3927 llvm::Comdat *ComdatKey = M.getOrInsertComdat(RegFn->getName()); 3928 RegFn->setLinkage(llvm::GlobalValue::LinkOnceAnyLinkage); 3929 RegFn->setVisibility(llvm::GlobalValue::HiddenVisibility); 3930 RegFn->setComdat(ComdatKey); 3931 UnRegFn->setComdat(ComdatKey); 3932 DeviceImages->setComdat(ComdatKey); 3933 Desc->setComdat(ComdatKey); 3934 } 3935 return RegFn; 3936 } 3937 3938 void CGOpenMPRuntime::createOffloadEntry( 3939 llvm::Constant *ID, llvm::Constant *Addr, uint64_t Size, int32_t Flags, 3940 llvm::GlobalValue::LinkageTypes Linkage) { 3941 StringRef Name = Addr->getName(); 3942 llvm::Module &M = CGM.getModule(); 3943 llvm::LLVMContext &C = M.getContext(); 3944 3945 // Create constant string with the name. 3946 llvm::Constant *StrPtrInit = llvm::ConstantDataArray::getString(C, Name); 3947 3948 std::string StringName = getName({"omp_offloading", "entry_name"}); 3949 auto *Str = new llvm::GlobalVariable( 3950 M, StrPtrInit->getType(), /*isConstant=*/true, 3951 llvm::GlobalValue::InternalLinkage, StrPtrInit, StringName); 3952 Str->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 3953 3954 llvm::Constant *Data[] = {llvm::ConstantExpr::getBitCast(ID, CGM.VoidPtrTy), 3955 llvm::ConstantExpr::getBitCast(Str, CGM.Int8PtrTy), 3956 llvm::ConstantInt::get(CGM.SizeTy, Size), 3957 llvm::ConstantInt::get(CGM.Int32Ty, Flags), 3958 llvm::ConstantInt::get(CGM.Int32Ty, 0)}; 3959 std::string EntryName = getName({"omp_offloading", "entry", ""}); 3960 llvm::GlobalVariable *Entry = createGlobalStruct( 3961 CGM, getTgtOffloadEntryQTy(), /*IsConstant=*/true, Data, 3962 Twine(EntryName).concat(Name), llvm::GlobalValue::WeakAnyLinkage); 3963 3964 // The entry has to be created in the section the linker expects it to be. 3965 std::string Section = getName({"omp_offloading", "entries"}); 3966 Entry->setSection(Section); 3967 } 3968 3969 void CGOpenMPRuntime::createOffloadEntriesAndInfoMetadata() { 3970 // Emit the offloading entries and metadata so that the device codegen side 3971 // can easily figure out what to emit. The produced metadata looks like 3972 // this: 3973 // 3974 // !omp_offload.info = !{!1, ...} 3975 // 3976 // Right now we only generate metadata for function that contain target 3977 // regions. 3978 3979 // If we do not have entries, we don't need to do anything. 3980 if (OffloadEntriesInfoManager.empty()) 3981 return; 3982 3983 llvm::Module &M = CGM.getModule(); 3984 llvm::LLVMContext &C = M.getContext(); 3985 SmallVector<const OffloadEntriesInfoManagerTy::OffloadEntryInfo *, 16> 3986 OrderedEntries(OffloadEntriesInfoManager.size()); 3987 llvm::SmallVector<StringRef, 16> ParentFunctions( 3988 OffloadEntriesInfoManager.size()); 3989 3990 // Auxiliary methods to create metadata values and strings. 3991 auto &&GetMDInt = [this](unsigned V) { 3992 return llvm::ConstantAsMetadata::get( 3993 llvm::ConstantInt::get(CGM.Int32Ty, V)); 3994 }; 3995 3996 auto &&GetMDString = [&C](StringRef V) { return llvm::MDString::get(C, V); }; 3997 3998 // Create the offloading info metadata node. 3999 llvm::NamedMDNode *MD = M.getOrInsertNamedMetadata("omp_offload.info"); 4000 4001 // Create function that emits metadata for each target region entry; 4002 auto &&TargetRegionMetadataEmitter = 4003 [&C, MD, &OrderedEntries, &ParentFunctions, &GetMDInt, &GetMDString]( 4004 unsigned DeviceID, unsigned FileID, StringRef ParentName, 4005 unsigned Line, 4006 const OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion &E) { 4007 // Generate metadata for target regions. Each entry of this metadata 4008 // contains: 4009 // - Entry 0 -> Kind of this type of metadata (0). 4010 // - Entry 1 -> Device ID of the file where the entry was identified. 4011 // - Entry 2 -> File ID of the file where the entry was identified. 4012 // - Entry 3 -> Mangled name of the function where the entry was 4013 // identified. 4014 // - Entry 4 -> Line in the file where the entry was identified. 4015 // - Entry 5 -> Order the entry was created. 4016 // The first element of the metadata node is the kind. 4017 llvm::Metadata *Ops[] = {GetMDInt(E.getKind()), GetMDInt(DeviceID), 4018 GetMDInt(FileID), GetMDString(ParentName), 4019 GetMDInt(Line), GetMDInt(E.getOrder())}; 4020 4021 // Save this entry in the right position of the ordered entries array. 4022 OrderedEntries[E.getOrder()] = &E; 4023 ParentFunctions[E.getOrder()] = ParentName; 4024 4025 // Add metadata to the named metadata node. 4026 MD->addOperand(llvm::MDNode::get(C, Ops)); 4027 }; 4028 4029 OffloadEntriesInfoManager.actOnTargetRegionEntriesInfo( 4030 TargetRegionMetadataEmitter); 4031 4032 // Create function that emits metadata for each device global variable entry; 4033 auto &&DeviceGlobalVarMetadataEmitter = 4034 [&C, &OrderedEntries, &GetMDInt, &GetMDString, 4035 MD](StringRef MangledName, 4036 const OffloadEntriesInfoManagerTy::OffloadEntryInfoDeviceGlobalVar 4037 &E) { 4038 // Generate metadata for global variables. Each entry of this metadata 4039 // contains: 4040 // - Entry 0 -> Kind of this type of metadata (1). 4041 // - Entry 1 -> Mangled name of the variable. 4042 // - Entry 2 -> Declare target kind. 4043 // - Entry 3 -> Order the entry was created. 4044 // The first element of the metadata node is the kind. 4045 llvm::Metadata *Ops[] = { 4046 GetMDInt(E.getKind()), GetMDString(MangledName), 4047 GetMDInt(E.getFlags()), GetMDInt(E.getOrder())}; 4048 4049 // Save this entry in the right position of the ordered entries array. 4050 OrderedEntries[E.getOrder()] = &E; 4051 4052 // Add metadata to the named metadata node. 4053 MD->addOperand(llvm::MDNode::get(C, Ops)); 4054 }; 4055 4056 OffloadEntriesInfoManager.actOnDeviceGlobalVarEntriesInfo( 4057 DeviceGlobalVarMetadataEmitter); 4058 4059 for (const auto *E : OrderedEntries) { 4060 assert(E && "All ordered entries must exist!"); 4061 if (const auto *CE = 4062 dyn_cast<OffloadEntriesInfoManagerTy::OffloadEntryInfoTargetRegion>( 4063 E)) { 4064 if (!CE->getID() || !CE->getAddress()) { 4065 // Do not blame the entry if the parent funtion is not emitted. 4066 StringRef FnName = ParentFunctions[CE->getOrder()]; 4067 if (!CGM.GetGlobalValue(FnName)) 4068 continue; 4069 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4070 DiagnosticsEngine::Error, 4071 "Offloading entry for target region is incorrect: either the " 4072 "address or the ID is invalid."); 4073 CGM.getDiags().Report(DiagID); 4074 continue; 4075 } 4076 createOffloadEntry(CE->getID(), CE->getAddress(), /*Size=*/0, 4077 CE->getFlags(), llvm::GlobalValue::WeakAnyLinkage); 4078 } else if (const auto *CE = 4079 dyn_cast<OffloadEntriesInfoManagerTy:: 4080 OffloadEntryInfoDeviceGlobalVar>(E)) { 4081 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags = 4082 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 4083 CE->getFlags()); 4084 switch (Flags) { 4085 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo: { 4086 if (!CE->getAddress()) { 4087 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4088 DiagnosticsEngine::Error, 4089 "Offloading entry for declare target variable is incorrect: the " 4090 "address is invalid."); 4091 CGM.getDiags().Report(DiagID); 4092 continue; 4093 } 4094 // The vaiable has no definition - no need to add the entry. 4095 if (CE->getVarSize().isZero()) 4096 continue; 4097 break; 4098 } 4099 case OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink: 4100 assert(((CGM.getLangOpts().OpenMPIsDevice && !CE->getAddress()) || 4101 (!CGM.getLangOpts().OpenMPIsDevice && CE->getAddress())) && 4102 "Declaret target link address is set."); 4103 if (CGM.getLangOpts().OpenMPIsDevice) 4104 continue; 4105 if (!CE->getAddress()) { 4106 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4107 DiagnosticsEngine::Error, 4108 "Offloading entry for declare target variable is incorrect: the " 4109 "address is invalid."); 4110 CGM.getDiags().Report(DiagID); 4111 continue; 4112 } 4113 break; 4114 } 4115 createOffloadEntry(CE->getAddress(), CE->getAddress(), 4116 CE->getVarSize().getQuantity(), Flags, 4117 CE->getLinkage()); 4118 } else { 4119 llvm_unreachable("Unsupported entry kind."); 4120 } 4121 } 4122 } 4123 4124 /// Loads all the offload entries information from the host IR 4125 /// metadata. 4126 void CGOpenMPRuntime::loadOffloadInfoMetadata() { 4127 // If we are in target mode, load the metadata from the host IR. This code has 4128 // to match the metadaata creation in createOffloadEntriesAndInfoMetadata(). 4129 4130 if (!CGM.getLangOpts().OpenMPIsDevice) 4131 return; 4132 4133 if (CGM.getLangOpts().OMPHostIRFile.empty()) 4134 return; 4135 4136 auto Buf = llvm::MemoryBuffer::getFile(CGM.getLangOpts().OMPHostIRFile); 4137 if (auto EC = Buf.getError()) { 4138 CGM.getDiags().Report(diag::err_cannot_open_file) 4139 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 4140 return; 4141 } 4142 4143 llvm::LLVMContext C; 4144 auto ME = expectedToErrorOrAndEmitErrors( 4145 C, llvm::parseBitcodeFile(Buf.get()->getMemBufferRef(), C)); 4146 4147 if (auto EC = ME.getError()) { 4148 unsigned DiagID = CGM.getDiags().getCustomDiagID( 4149 DiagnosticsEngine::Error, "Unable to parse host IR file '%0':'%1'"); 4150 CGM.getDiags().Report(DiagID) 4151 << CGM.getLangOpts().OMPHostIRFile << EC.message(); 4152 return; 4153 } 4154 4155 llvm::NamedMDNode *MD = ME.get()->getNamedMetadata("omp_offload.info"); 4156 if (!MD) 4157 return; 4158 4159 for (llvm::MDNode *MN : MD->operands()) { 4160 auto &&GetMDInt = [MN](unsigned Idx) { 4161 auto *V = cast<llvm::ConstantAsMetadata>(MN->getOperand(Idx)); 4162 return cast<llvm::ConstantInt>(V->getValue())->getZExtValue(); 4163 }; 4164 4165 auto &&GetMDString = [MN](unsigned Idx) { 4166 auto *V = cast<llvm::MDString>(MN->getOperand(Idx)); 4167 return V->getString(); 4168 }; 4169 4170 switch (GetMDInt(0)) { 4171 default: 4172 llvm_unreachable("Unexpected metadata!"); 4173 break; 4174 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 4175 OffloadingEntryInfoTargetRegion: 4176 OffloadEntriesInfoManager.initializeTargetRegionEntryInfo( 4177 /*DeviceID=*/GetMDInt(1), /*FileID=*/GetMDInt(2), 4178 /*ParentName=*/GetMDString(3), /*Line=*/GetMDInt(4), 4179 /*Order=*/GetMDInt(5)); 4180 break; 4181 case OffloadEntriesInfoManagerTy::OffloadEntryInfo:: 4182 OffloadingEntryInfoDeviceGlobalVar: 4183 OffloadEntriesInfoManager.initializeDeviceGlobalVarEntryInfo( 4184 /*MangledName=*/GetMDString(1), 4185 static_cast<OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind>( 4186 /*Flags=*/GetMDInt(2)), 4187 /*Order=*/GetMDInt(3)); 4188 break; 4189 } 4190 } 4191 } 4192 4193 void CGOpenMPRuntime::emitKmpRoutineEntryT(QualType KmpInt32Ty) { 4194 if (!KmpRoutineEntryPtrTy) { 4195 // Build typedef kmp_int32 (* kmp_routine_entry_t)(kmp_int32, void *); type. 4196 ASTContext &C = CGM.getContext(); 4197 QualType KmpRoutineEntryTyArgs[] = {KmpInt32Ty, C.VoidPtrTy}; 4198 FunctionProtoType::ExtProtoInfo EPI; 4199 KmpRoutineEntryPtrQTy = C.getPointerType( 4200 C.getFunctionType(KmpInt32Ty, KmpRoutineEntryTyArgs, EPI)); 4201 KmpRoutineEntryPtrTy = CGM.getTypes().ConvertType(KmpRoutineEntryPtrQTy); 4202 } 4203 } 4204 4205 QualType CGOpenMPRuntime::getTgtOffloadEntryQTy() { 4206 // Make sure the type of the entry is already created. This is the type we 4207 // have to create: 4208 // struct __tgt_offload_entry{ 4209 // void *addr; // Pointer to the offload entry info. 4210 // // (function or global) 4211 // char *name; // Name of the function or global. 4212 // size_t size; // Size of the entry info (0 if it a function). 4213 // int32_t flags; // Flags associated with the entry, e.g. 'link'. 4214 // int32_t reserved; // Reserved, to use by the runtime library. 4215 // }; 4216 if (TgtOffloadEntryQTy.isNull()) { 4217 ASTContext &C = CGM.getContext(); 4218 RecordDecl *RD = C.buildImplicitRecord("__tgt_offload_entry"); 4219 RD->startDefinition(); 4220 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4221 addFieldToRecordDecl(C, RD, C.getPointerType(C.CharTy)); 4222 addFieldToRecordDecl(C, RD, C.getSizeType()); 4223 addFieldToRecordDecl( 4224 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4225 addFieldToRecordDecl( 4226 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4227 RD->completeDefinition(); 4228 RD->addAttr(PackedAttr::CreateImplicit(C)); 4229 TgtOffloadEntryQTy = C.getRecordType(RD); 4230 } 4231 return TgtOffloadEntryQTy; 4232 } 4233 4234 QualType CGOpenMPRuntime::getTgtDeviceImageQTy() { 4235 // These are the types we need to build: 4236 // struct __tgt_device_image{ 4237 // void *ImageStart; // Pointer to the target code start. 4238 // void *ImageEnd; // Pointer to the target code end. 4239 // // We also add the host entries to the device image, as it may be useful 4240 // // for the target runtime to have access to that information. 4241 // __tgt_offload_entry *EntriesBegin; // Begin of the table with all 4242 // // the entries. 4243 // __tgt_offload_entry *EntriesEnd; // End of the table with all the 4244 // // entries (non inclusive). 4245 // }; 4246 if (TgtDeviceImageQTy.isNull()) { 4247 ASTContext &C = CGM.getContext(); 4248 RecordDecl *RD = C.buildImplicitRecord("__tgt_device_image"); 4249 RD->startDefinition(); 4250 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4251 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4252 addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy())); 4253 addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy())); 4254 RD->completeDefinition(); 4255 TgtDeviceImageQTy = C.getRecordType(RD); 4256 } 4257 return TgtDeviceImageQTy; 4258 } 4259 4260 QualType CGOpenMPRuntime::getTgtBinaryDescriptorQTy() { 4261 // struct __tgt_bin_desc{ 4262 // int32_t NumDevices; // Number of devices supported. 4263 // __tgt_device_image *DeviceImages; // Arrays of device images 4264 // // (one per device). 4265 // __tgt_offload_entry *EntriesBegin; // Begin of the table with all the 4266 // // entries. 4267 // __tgt_offload_entry *EntriesEnd; // End of the table with all the 4268 // // entries (non inclusive). 4269 // }; 4270 if (TgtBinaryDescriptorQTy.isNull()) { 4271 ASTContext &C = CGM.getContext(); 4272 RecordDecl *RD = C.buildImplicitRecord("__tgt_bin_desc"); 4273 RD->startDefinition(); 4274 addFieldToRecordDecl( 4275 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/true)); 4276 addFieldToRecordDecl(C, RD, C.getPointerType(getTgtDeviceImageQTy())); 4277 addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy())); 4278 addFieldToRecordDecl(C, RD, C.getPointerType(getTgtOffloadEntryQTy())); 4279 RD->completeDefinition(); 4280 TgtBinaryDescriptorQTy = C.getRecordType(RD); 4281 } 4282 return TgtBinaryDescriptorQTy; 4283 } 4284 4285 namespace { 4286 struct PrivateHelpersTy { 4287 PrivateHelpersTy(const VarDecl *Original, const VarDecl *PrivateCopy, 4288 const VarDecl *PrivateElemInit) 4289 : Original(Original), PrivateCopy(PrivateCopy), 4290 PrivateElemInit(PrivateElemInit) {} 4291 const VarDecl *Original; 4292 const VarDecl *PrivateCopy; 4293 const VarDecl *PrivateElemInit; 4294 }; 4295 typedef std::pair<CharUnits /*Align*/, PrivateHelpersTy> PrivateDataTy; 4296 } // anonymous namespace 4297 4298 static RecordDecl * 4299 createPrivatesRecordDecl(CodeGenModule &CGM, ArrayRef<PrivateDataTy> Privates) { 4300 if (!Privates.empty()) { 4301 ASTContext &C = CGM.getContext(); 4302 // Build struct .kmp_privates_t. { 4303 // /* private vars */ 4304 // }; 4305 RecordDecl *RD = C.buildImplicitRecord(".kmp_privates.t"); 4306 RD->startDefinition(); 4307 for (const auto &Pair : Privates) { 4308 const VarDecl *VD = Pair.second.Original; 4309 QualType Type = VD->getType().getNonReferenceType(); 4310 FieldDecl *FD = addFieldToRecordDecl(C, RD, Type); 4311 if (VD->hasAttrs()) { 4312 for (specific_attr_iterator<AlignedAttr> I(VD->getAttrs().begin()), 4313 E(VD->getAttrs().end()); 4314 I != E; ++I) 4315 FD->addAttr(*I); 4316 } 4317 } 4318 RD->completeDefinition(); 4319 return RD; 4320 } 4321 return nullptr; 4322 } 4323 4324 static RecordDecl * 4325 createKmpTaskTRecordDecl(CodeGenModule &CGM, OpenMPDirectiveKind Kind, 4326 QualType KmpInt32Ty, 4327 QualType KmpRoutineEntryPointerQTy) { 4328 ASTContext &C = CGM.getContext(); 4329 // Build struct kmp_task_t { 4330 // void * shareds; 4331 // kmp_routine_entry_t routine; 4332 // kmp_int32 part_id; 4333 // kmp_cmplrdata_t data1; 4334 // kmp_cmplrdata_t data2; 4335 // For taskloops additional fields: 4336 // kmp_uint64 lb; 4337 // kmp_uint64 ub; 4338 // kmp_int64 st; 4339 // kmp_int32 liter; 4340 // void * reductions; 4341 // }; 4342 RecordDecl *UD = C.buildImplicitRecord("kmp_cmplrdata_t", TTK_Union); 4343 UD->startDefinition(); 4344 addFieldToRecordDecl(C, UD, KmpInt32Ty); 4345 addFieldToRecordDecl(C, UD, KmpRoutineEntryPointerQTy); 4346 UD->completeDefinition(); 4347 QualType KmpCmplrdataTy = C.getRecordType(UD); 4348 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t"); 4349 RD->startDefinition(); 4350 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4351 addFieldToRecordDecl(C, RD, KmpRoutineEntryPointerQTy); 4352 addFieldToRecordDecl(C, RD, KmpInt32Ty); 4353 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 4354 addFieldToRecordDecl(C, RD, KmpCmplrdataTy); 4355 if (isOpenMPTaskLoopDirective(Kind)) { 4356 QualType KmpUInt64Ty = 4357 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/0); 4358 QualType KmpInt64Ty = 4359 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 4360 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 4361 addFieldToRecordDecl(C, RD, KmpUInt64Ty); 4362 addFieldToRecordDecl(C, RD, KmpInt64Ty); 4363 addFieldToRecordDecl(C, RD, KmpInt32Ty); 4364 addFieldToRecordDecl(C, RD, C.VoidPtrTy); 4365 } 4366 RD->completeDefinition(); 4367 return RD; 4368 } 4369 4370 static RecordDecl * 4371 createKmpTaskTWithPrivatesRecordDecl(CodeGenModule &CGM, QualType KmpTaskTQTy, 4372 ArrayRef<PrivateDataTy> Privates) { 4373 ASTContext &C = CGM.getContext(); 4374 // Build struct kmp_task_t_with_privates { 4375 // kmp_task_t task_data; 4376 // .kmp_privates_t. privates; 4377 // }; 4378 RecordDecl *RD = C.buildImplicitRecord("kmp_task_t_with_privates"); 4379 RD->startDefinition(); 4380 addFieldToRecordDecl(C, RD, KmpTaskTQTy); 4381 if (const RecordDecl *PrivateRD = createPrivatesRecordDecl(CGM, Privates)) 4382 addFieldToRecordDecl(C, RD, C.getRecordType(PrivateRD)); 4383 RD->completeDefinition(); 4384 return RD; 4385 } 4386 4387 /// Emit a proxy function which accepts kmp_task_t as the second 4388 /// argument. 4389 /// \code 4390 /// kmp_int32 .omp_task_entry.(kmp_int32 gtid, kmp_task_t *tt) { 4391 /// TaskFunction(gtid, tt->part_id, &tt->privates, task_privates_map, tt, 4392 /// For taskloops: 4393 /// tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 4394 /// tt->reductions, tt->shareds); 4395 /// return 0; 4396 /// } 4397 /// \endcode 4398 static llvm::Function * 4399 emitProxyTaskFunction(CodeGenModule &CGM, SourceLocation Loc, 4400 OpenMPDirectiveKind Kind, QualType KmpInt32Ty, 4401 QualType KmpTaskTWithPrivatesPtrQTy, 4402 QualType KmpTaskTWithPrivatesQTy, QualType KmpTaskTQTy, 4403 QualType SharedsPtrTy, llvm::Function *TaskFunction, 4404 llvm::Value *TaskPrivatesMap) { 4405 ASTContext &C = CGM.getContext(); 4406 FunctionArgList Args; 4407 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 4408 ImplicitParamDecl::Other); 4409 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4410 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 4411 ImplicitParamDecl::Other); 4412 Args.push_back(&GtidArg); 4413 Args.push_back(&TaskTypeArg); 4414 const auto &TaskEntryFnInfo = 4415 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 4416 llvm::FunctionType *TaskEntryTy = 4417 CGM.getTypes().GetFunctionType(TaskEntryFnInfo); 4418 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_entry", ""}); 4419 auto *TaskEntry = llvm::Function::Create( 4420 TaskEntryTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 4421 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskEntry, TaskEntryFnInfo); 4422 TaskEntry->setDoesNotRecurse(); 4423 CodeGenFunction CGF(CGM); 4424 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, TaskEntry, TaskEntryFnInfo, Args, 4425 Loc, Loc); 4426 4427 // TaskFunction(gtid, tt->task_data.part_id, &tt->privates, task_privates_map, 4428 // tt, 4429 // For taskloops: 4430 // tt->task_data.lb, tt->task_data.ub, tt->task_data.st, tt->task_data.liter, 4431 // tt->task_data.shareds); 4432 llvm::Value *GtidParam = CGF.EmitLoadOfScalar( 4433 CGF.GetAddrOfLocalVar(&GtidArg), /*Volatile=*/false, KmpInt32Ty, Loc); 4434 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4435 CGF.GetAddrOfLocalVar(&TaskTypeArg), 4436 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4437 const auto *KmpTaskTWithPrivatesQTyRD = 4438 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 4439 LValue Base = 4440 CGF.EmitLValueForField(TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4441 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 4442 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 4443 LValue PartIdLVal = CGF.EmitLValueForField(Base, *PartIdFI); 4444 llvm::Value *PartidParam = PartIdLVal.getPointer(); 4445 4446 auto SharedsFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTShareds); 4447 LValue SharedsLVal = CGF.EmitLValueForField(Base, *SharedsFI); 4448 llvm::Value *SharedsParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4449 CGF.EmitLoadOfScalar(SharedsLVal, Loc), 4450 CGF.ConvertTypeForMem(SharedsPtrTy)); 4451 4452 auto PrivatesFI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 4453 llvm::Value *PrivatesParam; 4454 if (PrivatesFI != KmpTaskTWithPrivatesQTyRD->field_end()) { 4455 LValue PrivatesLVal = CGF.EmitLValueForField(TDBase, *PrivatesFI); 4456 PrivatesParam = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4457 PrivatesLVal.getPointer(), CGF.VoidPtrTy); 4458 } else { 4459 PrivatesParam = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 4460 } 4461 4462 llvm::Value *CommonArgs[] = {GtidParam, PartidParam, PrivatesParam, 4463 TaskPrivatesMap, 4464 CGF.Builder 4465 .CreatePointerBitCastOrAddrSpaceCast( 4466 TDBase.getAddress(), CGF.VoidPtrTy) 4467 .getPointer()}; 4468 SmallVector<llvm::Value *, 16> CallArgs(std::begin(CommonArgs), 4469 std::end(CommonArgs)); 4470 if (isOpenMPTaskLoopDirective(Kind)) { 4471 auto LBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound); 4472 LValue LBLVal = CGF.EmitLValueForField(Base, *LBFI); 4473 llvm::Value *LBParam = CGF.EmitLoadOfScalar(LBLVal, Loc); 4474 auto UBFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound); 4475 LValue UBLVal = CGF.EmitLValueForField(Base, *UBFI); 4476 llvm::Value *UBParam = CGF.EmitLoadOfScalar(UBLVal, Loc); 4477 auto StFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTStride); 4478 LValue StLVal = CGF.EmitLValueForField(Base, *StFI); 4479 llvm::Value *StParam = CGF.EmitLoadOfScalar(StLVal, Loc); 4480 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 4481 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 4482 llvm::Value *LIParam = CGF.EmitLoadOfScalar(LILVal, Loc); 4483 auto RFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTReductions); 4484 LValue RLVal = CGF.EmitLValueForField(Base, *RFI); 4485 llvm::Value *RParam = CGF.EmitLoadOfScalar(RLVal, Loc); 4486 CallArgs.push_back(LBParam); 4487 CallArgs.push_back(UBParam); 4488 CallArgs.push_back(StParam); 4489 CallArgs.push_back(LIParam); 4490 CallArgs.push_back(RParam); 4491 } 4492 CallArgs.push_back(SharedsParam); 4493 4494 CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskFunction, 4495 CallArgs); 4496 CGF.EmitStoreThroughLValue(RValue::get(CGF.Builder.getInt32(/*C=*/0)), 4497 CGF.MakeAddrLValue(CGF.ReturnValue, KmpInt32Ty)); 4498 CGF.FinishFunction(); 4499 return TaskEntry; 4500 } 4501 4502 static llvm::Value *emitDestructorsFunction(CodeGenModule &CGM, 4503 SourceLocation Loc, 4504 QualType KmpInt32Ty, 4505 QualType KmpTaskTWithPrivatesPtrQTy, 4506 QualType KmpTaskTWithPrivatesQTy) { 4507 ASTContext &C = CGM.getContext(); 4508 FunctionArgList Args; 4509 ImplicitParamDecl GtidArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, KmpInt32Ty, 4510 ImplicitParamDecl::Other); 4511 ImplicitParamDecl TaskTypeArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4512 KmpTaskTWithPrivatesPtrQTy.withRestrict(), 4513 ImplicitParamDecl::Other); 4514 Args.push_back(&GtidArg); 4515 Args.push_back(&TaskTypeArg); 4516 const auto &DestructorFnInfo = 4517 CGM.getTypes().arrangeBuiltinFunctionDeclaration(KmpInt32Ty, Args); 4518 llvm::FunctionType *DestructorFnTy = 4519 CGM.getTypes().GetFunctionType(DestructorFnInfo); 4520 std::string Name = 4521 CGM.getOpenMPRuntime().getName({"omp_task_destructor", ""}); 4522 auto *DestructorFn = 4523 llvm::Function::Create(DestructorFnTy, llvm::GlobalValue::InternalLinkage, 4524 Name, &CGM.getModule()); 4525 CGM.SetInternalFunctionAttributes(GlobalDecl(), DestructorFn, 4526 DestructorFnInfo); 4527 DestructorFn->setDoesNotRecurse(); 4528 CodeGenFunction CGF(CGM); 4529 CGF.StartFunction(GlobalDecl(), KmpInt32Ty, DestructorFn, DestructorFnInfo, 4530 Args, Loc, Loc); 4531 4532 LValue Base = CGF.EmitLoadOfPointerLValue( 4533 CGF.GetAddrOfLocalVar(&TaskTypeArg), 4534 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4535 const auto *KmpTaskTWithPrivatesQTyRD = 4536 cast<RecordDecl>(KmpTaskTWithPrivatesQTy->getAsTagDecl()); 4537 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4538 Base = CGF.EmitLValueForField(Base, *FI); 4539 for (const auto *Field : 4540 cast<RecordDecl>(FI->getType()->getAsTagDecl())->fields()) { 4541 if (QualType::DestructionKind DtorKind = 4542 Field->getType().isDestructedType()) { 4543 LValue FieldLValue = CGF.EmitLValueForField(Base, Field); 4544 CGF.pushDestroy(DtorKind, FieldLValue.getAddress(), Field->getType()); 4545 } 4546 } 4547 CGF.FinishFunction(); 4548 return DestructorFn; 4549 } 4550 4551 /// Emit a privates mapping function for correct handling of private and 4552 /// firstprivate variables. 4553 /// \code 4554 /// void .omp_task_privates_map.(const .privates. *noalias privs, <ty1> 4555 /// **noalias priv1,..., <tyn> **noalias privn) { 4556 /// *priv1 = &.privates.priv1; 4557 /// ...; 4558 /// *privn = &.privates.privn; 4559 /// } 4560 /// \endcode 4561 static llvm::Value * 4562 emitTaskPrivateMappingFunction(CodeGenModule &CGM, SourceLocation Loc, 4563 ArrayRef<const Expr *> PrivateVars, 4564 ArrayRef<const Expr *> FirstprivateVars, 4565 ArrayRef<const Expr *> LastprivateVars, 4566 QualType PrivatesQTy, 4567 ArrayRef<PrivateDataTy> Privates) { 4568 ASTContext &C = CGM.getContext(); 4569 FunctionArgList Args; 4570 ImplicitParamDecl TaskPrivatesArg( 4571 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4572 C.getPointerType(PrivatesQTy).withConst().withRestrict(), 4573 ImplicitParamDecl::Other); 4574 Args.push_back(&TaskPrivatesArg); 4575 llvm::DenseMap<const VarDecl *, unsigned> PrivateVarsPos; 4576 unsigned Counter = 1; 4577 for (const Expr *E : PrivateVars) { 4578 Args.push_back(ImplicitParamDecl::Create( 4579 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4580 C.getPointerType(C.getPointerType(E->getType())) 4581 .withConst() 4582 .withRestrict(), 4583 ImplicitParamDecl::Other)); 4584 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4585 PrivateVarsPos[VD] = Counter; 4586 ++Counter; 4587 } 4588 for (const Expr *E : FirstprivateVars) { 4589 Args.push_back(ImplicitParamDecl::Create( 4590 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4591 C.getPointerType(C.getPointerType(E->getType())) 4592 .withConst() 4593 .withRestrict(), 4594 ImplicitParamDecl::Other)); 4595 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4596 PrivateVarsPos[VD] = Counter; 4597 ++Counter; 4598 } 4599 for (const Expr *E : LastprivateVars) { 4600 Args.push_back(ImplicitParamDecl::Create( 4601 C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4602 C.getPointerType(C.getPointerType(E->getType())) 4603 .withConst() 4604 .withRestrict(), 4605 ImplicitParamDecl::Other)); 4606 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4607 PrivateVarsPos[VD] = Counter; 4608 ++Counter; 4609 } 4610 const auto &TaskPrivatesMapFnInfo = 4611 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4612 llvm::FunctionType *TaskPrivatesMapTy = 4613 CGM.getTypes().GetFunctionType(TaskPrivatesMapFnInfo); 4614 std::string Name = 4615 CGM.getOpenMPRuntime().getName({"omp_task_privates_map", ""}); 4616 auto *TaskPrivatesMap = llvm::Function::Create( 4617 TaskPrivatesMapTy, llvm::GlobalValue::InternalLinkage, Name, 4618 &CGM.getModule()); 4619 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskPrivatesMap, 4620 TaskPrivatesMapFnInfo); 4621 TaskPrivatesMap->removeFnAttr(llvm::Attribute::NoInline); 4622 TaskPrivatesMap->removeFnAttr(llvm::Attribute::OptimizeNone); 4623 TaskPrivatesMap->addFnAttr(llvm::Attribute::AlwaysInline); 4624 CodeGenFunction CGF(CGM); 4625 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskPrivatesMap, 4626 TaskPrivatesMapFnInfo, Args, Loc, Loc); 4627 4628 // *privi = &.privates.privi; 4629 LValue Base = CGF.EmitLoadOfPointerLValue( 4630 CGF.GetAddrOfLocalVar(&TaskPrivatesArg), 4631 TaskPrivatesArg.getType()->castAs<PointerType>()); 4632 const auto *PrivatesQTyRD = cast<RecordDecl>(PrivatesQTy->getAsTagDecl()); 4633 Counter = 0; 4634 for (const FieldDecl *Field : PrivatesQTyRD->fields()) { 4635 LValue FieldLVal = CGF.EmitLValueForField(Base, Field); 4636 const VarDecl *VD = Args[PrivateVarsPos[Privates[Counter].second.Original]]; 4637 LValue RefLVal = 4638 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(VD), VD->getType()); 4639 LValue RefLoadLVal = CGF.EmitLoadOfPointerLValue( 4640 RefLVal.getAddress(), RefLVal.getType()->castAs<PointerType>()); 4641 CGF.EmitStoreOfScalar(FieldLVal.getPointer(), RefLoadLVal); 4642 ++Counter; 4643 } 4644 CGF.FinishFunction(); 4645 return TaskPrivatesMap; 4646 } 4647 4648 static bool stable_sort_comparator(const PrivateDataTy P1, 4649 const PrivateDataTy P2) { 4650 return P1.first > P2.first; 4651 } 4652 4653 /// Emit initialization for private variables in task-based directives. 4654 static void emitPrivatesInit(CodeGenFunction &CGF, 4655 const OMPExecutableDirective &D, 4656 Address KmpTaskSharedsPtr, LValue TDBase, 4657 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 4658 QualType SharedsTy, QualType SharedsPtrTy, 4659 const OMPTaskDataTy &Data, 4660 ArrayRef<PrivateDataTy> Privates, bool ForDup) { 4661 ASTContext &C = CGF.getContext(); 4662 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4663 LValue PrivatesBase = CGF.EmitLValueForField(TDBase, *FI); 4664 OpenMPDirectiveKind Kind = isOpenMPTaskLoopDirective(D.getDirectiveKind()) 4665 ? OMPD_taskloop 4666 : OMPD_task; 4667 const CapturedStmt &CS = *D.getCapturedStmt(Kind); 4668 CodeGenFunction::CGCapturedStmtInfo CapturesInfo(CS); 4669 LValue SrcBase; 4670 bool IsTargetTask = 4671 isOpenMPTargetDataManagementDirective(D.getDirectiveKind()) || 4672 isOpenMPTargetExecutionDirective(D.getDirectiveKind()); 4673 // For target-based directives skip 3 firstprivate arrays BasePointersArray, 4674 // PointersArray and SizesArray. The original variables for these arrays are 4675 // not captured and we get their addresses explicitly. 4676 if ((!IsTargetTask && !Data.FirstprivateVars.empty()) || 4677 (IsTargetTask && KmpTaskSharedsPtr.isValid())) { 4678 SrcBase = CGF.MakeAddrLValue( 4679 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4680 KmpTaskSharedsPtr, CGF.ConvertTypeForMem(SharedsPtrTy)), 4681 SharedsTy); 4682 } 4683 FI = cast<RecordDecl>(FI->getType()->getAsTagDecl())->field_begin(); 4684 for (const PrivateDataTy &Pair : Privates) { 4685 const VarDecl *VD = Pair.second.PrivateCopy; 4686 const Expr *Init = VD->getAnyInitializer(); 4687 if (Init && (!ForDup || (isa<CXXConstructExpr>(Init) && 4688 !CGF.isTrivialInitializer(Init)))) { 4689 LValue PrivateLValue = CGF.EmitLValueForField(PrivatesBase, *FI); 4690 if (const VarDecl *Elem = Pair.second.PrivateElemInit) { 4691 const VarDecl *OriginalVD = Pair.second.Original; 4692 // Check if the variable is the target-based BasePointersArray, 4693 // PointersArray or SizesArray. 4694 LValue SharedRefLValue; 4695 QualType Type = OriginalVD->getType(); 4696 const FieldDecl *SharedField = CapturesInfo.lookup(OriginalVD); 4697 if (IsTargetTask && !SharedField) { 4698 assert(isa<ImplicitParamDecl>(OriginalVD) && 4699 isa<CapturedDecl>(OriginalVD->getDeclContext()) && 4700 cast<CapturedDecl>(OriginalVD->getDeclContext()) 4701 ->getNumParams() == 0 && 4702 isa<TranslationUnitDecl>( 4703 cast<CapturedDecl>(OriginalVD->getDeclContext()) 4704 ->getDeclContext()) && 4705 "Expected artificial target data variable."); 4706 SharedRefLValue = 4707 CGF.MakeAddrLValue(CGF.GetAddrOfLocalVar(OriginalVD), Type); 4708 } else { 4709 SharedRefLValue = CGF.EmitLValueForField(SrcBase, SharedField); 4710 SharedRefLValue = CGF.MakeAddrLValue( 4711 Address(SharedRefLValue.getPointer(), C.getDeclAlign(OriginalVD)), 4712 SharedRefLValue.getType(), LValueBaseInfo(AlignmentSource::Decl), 4713 SharedRefLValue.getTBAAInfo()); 4714 } 4715 if (Type->isArrayType()) { 4716 // Initialize firstprivate array. 4717 if (!isa<CXXConstructExpr>(Init) || CGF.isTrivialInitializer(Init)) { 4718 // Perform simple memcpy. 4719 CGF.EmitAggregateAssign(PrivateLValue, SharedRefLValue, Type); 4720 } else { 4721 // Initialize firstprivate array using element-by-element 4722 // initialization. 4723 CGF.EmitOMPAggregateAssign( 4724 PrivateLValue.getAddress(), SharedRefLValue.getAddress(), Type, 4725 [&CGF, Elem, Init, &CapturesInfo](Address DestElement, 4726 Address SrcElement) { 4727 // Clean up any temporaries needed by the initialization. 4728 CodeGenFunction::OMPPrivateScope InitScope(CGF); 4729 InitScope.addPrivate( 4730 Elem, [SrcElement]() -> Address { return SrcElement; }); 4731 (void)InitScope.Privatize(); 4732 // Emit initialization for single element. 4733 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII( 4734 CGF, &CapturesInfo); 4735 CGF.EmitAnyExprToMem(Init, DestElement, 4736 Init->getType().getQualifiers(), 4737 /*IsInitializer=*/false); 4738 }); 4739 } 4740 } else { 4741 CodeGenFunction::OMPPrivateScope InitScope(CGF); 4742 InitScope.addPrivate(Elem, [SharedRefLValue]() -> Address { 4743 return SharedRefLValue.getAddress(); 4744 }); 4745 (void)InitScope.Privatize(); 4746 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CapturesInfo); 4747 CGF.EmitExprAsInit(Init, VD, PrivateLValue, 4748 /*capturedByInit=*/false); 4749 } 4750 } else { 4751 CGF.EmitExprAsInit(Init, VD, PrivateLValue, /*capturedByInit=*/false); 4752 } 4753 } 4754 ++FI; 4755 } 4756 } 4757 4758 /// Check if duplication function is required for taskloops. 4759 static bool checkInitIsRequired(CodeGenFunction &CGF, 4760 ArrayRef<PrivateDataTy> Privates) { 4761 bool InitRequired = false; 4762 for (const PrivateDataTy &Pair : Privates) { 4763 const VarDecl *VD = Pair.second.PrivateCopy; 4764 const Expr *Init = VD->getAnyInitializer(); 4765 InitRequired = InitRequired || (Init && isa<CXXConstructExpr>(Init) && 4766 !CGF.isTrivialInitializer(Init)); 4767 if (InitRequired) 4768 break; 4769 } 4770 return InitRequired; 4771 } 4772 4773 4774 /// Emit task_dup function (for initialization of 4775 /// private/firstprivate/lastprivate vars and last_iter flag) 4776 /// \code 4777 /// void __task_dup_entry(kmp_task_t *task_dst, const kmp_task_t *task_src, int 4778 /// lastpriv) { 4779 /// // setup lastprivate flag 4780 /// task_dst->last = lastpriv; 4781 /// // could be constructor calls here... 4782 /// } 4783 /// \endcode 4784 static llvm::Value * 4785 emitTaskDupFunction(CodeGenModule &CGM, SourceLocation Loc, 4786 const OMPExecutableDirective &D, 4787 QualType KmpTaskTWithPrivatesPtrQTy, 4788 const RecordDecl *KmpTaskTWithPrivatesQTyRD, 4789 const RecordDecl *KmpTaskTQTyRD, QualType SharedsTy, 4790 QualType SharedsPtrTy, const OMPTaskDataTy &Data, 4791 ArrayRef<PrivateDataTy> Privates, bool WithLastIter) { 4792 ASTContext &C = CGM.getContext(); 4793 FunctionArgList Args; 4794 ImplicitParamDecl DstArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4795 KmpTaskTWithPrivatesPtrQTy, 4796 ImplicitParamDecl::Other); 4797 ImplicitParamDecl SrcArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 4798 KmpTaskTWithPrivatesPtrQTy, 4799 ImplicitParamDecl::Other); 4800 ImplicitParamDecl LastprivArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.IntTy, 4801 ImplicitParamDecl::Other); 4802 Args.push_back(&DstArg); 4803 Args.push_back(&SrcArg); 4804 Args.push_back(&LastprivArg); 4805 const auto &TaskDupFnInfo = 4806 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 4807 llvm::FunctionType *TaskDupTy = CGM.getTypes().GetFunctionType(TaskDupFnInfo); 4808 std::string Name = CGM.getOpenMPRuntime().getName({"omp_task_dup", ""}); 4809 auto *TaskDup = llvm::Function::Create( 4810 TaskDupTy, llvm::GlobalValue::InternalLinkage, Name, &CGM.getModule()); 4811 CGM.SetInternalFunctionAttributes(GlobalDecl(), TaskDup, TaskDupFnInfo); 4812 TaskDup->setDoesNotRecurse(); 4813 CodeGenFunction CGF(CGM); 4814 CGF.StartFunction(GlobalDecl(), C.VoidTy, TaskDup, TaskDupFnInfo, Args, Loc, 4815 Loc); 4816 4817 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4818 CGF.GetAddrOfLocalVar(&DstArg), 4819 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4820 // task_dst->liter = lastpriv; 4821 if (WithLastIter) { 4822 auto LIFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTLastIter); 4823 LValue Base = CGF.EmitLValueForField( 4824 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4825 LValue LILVal = CGF.EmitLValueForField(Base, *LIFI); 4826 llvm::Value *Lastpriv = CGF.EmitLoadOfScalar( 4827 CGF.GetAddrOfLocalVar(&LastprivArg), /*Volatile=*/false, C.IntTy, Loc); 4828 CGF.EmitStoreOfScalar(Lastpriv, LILVal); 4829 } 4830 4831 // Emit initial values for private copies (if any). 4832 assert(!Privates.empty()); 4833 Address KmpTaskSharedsPtr = Address::invalid(); 4834 if (!Data.FirstprivateVars.empty()) { 4835 LValue TDBase = CGF.EmitLoadOfPointerLValue( 4836 CGF.GetAddrOfLocalVar(&SrcArg), 4837 KmpTaskTWithPrivatesPtrQTy->castAs<PointerType>()); 4838 LValue Base = CGF.EmitLValueForField( 4839 TDBase, *KmpTaskTWithPrivatesQTyRD->field_begin()); 4840 KmpTaskSharedsPtr = Address( 4841 CGF.EmitLoadOfScalar(CGF.EmitLValueForField( 4842 Base, *std::next(KmpTaskTQTyRD->field_begin(), 4843 KmpTaskTShareds)), 4844 Loc), 4845 CGF.getNaturalTypeAlignment(SharedsTy)); 4846 } 4847 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, TDBase, KmpTaskTWithPrivatesQTyRD, 4848 SharedsTy, SharedsPtrTy, Data, Privates, /*ForDup=*/true); 4849 CGF.FinishFunction(); 4850 return TaskDup; 4851 } 4852 4853 /// Checks if destructor function is required to be generated. 4854 /// \return true if cleanups are required, false otherwise. 4855 static bool 4856 checkDestructorsRequired(const RecordDecl *KmpTaskTWithPrivatesQTyRD) { 4857 bool NeedsCleanup = false; 4858 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin(), 1); 4859 const auto *PrivateRD = cast<RecordDecl>(FI->getType()->getAsTagDecl()); 4860 for (const FieldDecl *FD : PrivateRD->fields()) { 4861 NeedsCleanup = NeedsCleanup || FD->getType().isDestructedType(); 4862 if (NeedsCleanup) 4863 break; 4864 } 4865 return NeedsCleanup; 4866 } 4867 4868 CGOpenMPRuntime::TaskResultTy 4869 CGOpenMPRuntime::emitTaskInit(CodeGenFunction &CGF, SourceLocation Loc, 4870 const OMPExecutableDirective &D, 4871 llvm::Function *TaskFunction, QualType SharedsTy, 4872 Address Shareds, const OMPTaskDataTy &Data) { 4873 ASTContext &C = CGM.getContext(); 4874 llvm::SmallVector<PrivateDataTy, 4> Privates; 4875 // Aggregate privates and sort them by the alignment. 4876 auto I = Data.PrivateCopies.begin(); 4877 for (const Expr *E : Data.PrivateVars) { 4878 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4879 Privates.emplace_back( 4880 C.getDeclAlign(VD), 4881 PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4882 /*PrivateElemInit=*/nullptr)); 4883 ++I; 4884 } 4885 I = Data.FirstprivateCopies.begin(); 4886 auto IElemInitRef = Data.FirstprivateInits.begin(); 4887 for (const Expr *E : Data.FirstprivateVars) { 4888 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4889 Privates.emplace_back( 4890 C.getDeclAlign(VD), 4891 PrivateHelpersTy( 4892 VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4893 cast<VarDecl>(cast<DeclRefExpr>(*IElemInitRef)->getDecl()))); 4894 ++I; 4895 ++IElemInitRef; 4896 } 4897 I = Data.LastprivateCopies.begin(); 4898 for (const Expr *E : Data.LastprivateVars) { 4899 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(E)->getDecl()); 4900 Privates.emplace_back( 4901 C.getDeclAlign(VD), 4902 PrivateHelpersTy(VD, cast<VarDecl>(cast<DeclRefExpr>(*I)->getDecl()), 4903 /*PrivateElemInit=*/nullptr)); 4904 ++I; 4905 } 4906 std::stable_sort(Privates.begin(), Privates.end(), stable_sort_comparator); 4907 QualType KmpInt32Ty = C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/1); 4908 // Build type kmp_routine_entry_t (if not built yet). 4909 emitKmpRoutineEntryT(KmpInt32Ty); 4910 // Build type kmp_task_t (if not built yet). 4911 if (isOpenMPTaskLoopDirective(D.getDirectiveKind())) { 4912 if (SavedKmpTaskloopTQTy.isNull()) { 4913 SavedKmpTaskloopTQTy = C.getRecordType(createKmpTaskTRecordDecl( 4914 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 4915 } 4916 KmpTaskTQTy = SavedKmpTaskloopTQTy; 4917 } else { 4918 assert((D.getDirectiveKind() == OMPD_task || 4919 isOpenMPTargetExecutionDirective(D.getDirectiveKind()) || 4920 isOpenMPTargetDataManagementDirective(D.getDirectiveKind())) && 4921 "Expected taskloop, task or target directive"); 4922 if (SavedKmpTaskTQTy.isNull()) { 4923 SavedKmpTaskTQTy = C.getRecordType(createKmpTaskTRecordDecl( 4924 CGM, D.getDirectiveKind(), KmpInt32Ty, KmpRoutineEntryPtrQTy)); 4925 } 4926 KmpTaskTQTy = SavedKmpTaskTQTy; 4927 } 4928 const auto *KmpTaskTQTyRD = cast<RecordDecl>(KmpTaskTQTy->getAsTagDecl()); 4929 // Build particular struct kmp_task_t for the given task. 4930 const RecordDecl *KmpTaskTWithPrivatesQTyRD = 4931 createKmpTaskTWithPrivatesRecordDecl(CGM, KmpTaskTQTy, Privates); 4932 QualType KmpTaskTWithPrivatesQTy = C.getRecordType(KmpTaskTWithPrivatesQTyRD); 4933 QualType KmpTaskTWithPrivatesPtrQTy = 4934 C.getPointerType(KmpTaskTWithPrivatesQTy); 4935 llvm::Type *KmpTaskTWithPrivatesTy = CGF.ConvertType(KmpTaskTWithPrivatesQTy); 4936 llvm::Type *KmpTaskTWithPrivatesPtrTy = 4937 KmpTaskTWithPrivatesTy->getPointerTo(); 4938 llvm::Value *KmpTaskTWithPrivatesTySize = 4939 CGF.getTypeSize(KmpTaskTWithPrivatesQTy); 4940 QualType SharedsPtrTy = C.getPointerType(SharedsTy); 4941 4942 // Emit initial values for private copies (if any). 4943 llvm::Value *TaskPrivatesMap = nullptr; 4944 llvm::Type *TaskPrivatesMapTy = 4945 std::next(TaskFunction->arg_begin(), 3)->getType(); 4946 if (!Privates.empty()) { 4947 auto FI = std::next(KmpTaskTWithPrivatesQTyRD->field_begin()); 4948 TaskPrivatesMap = emitTaskPrivateMappingFunction( 4949 CGM, Loc, Data.PrivateVars, Data.FirstprivateVars, Data.LastprivateVars, 4950 FI->getType(), Privates); 4951 TaskPrivatesMap = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4952 TaskPrivatesMap, TaskPrivatesMapTy); 4953 } else { 4954 TaskPrivatesMap = llvm::ConstantPointerNull::get( 4955 cast<llvm::PointerType>(TaskPrivatesMapTy)); 4956 } 4957 // Build a proxy function kmp_int32 .omp_task_entry.(kmp_int32 gtid, 4958 // kmp_task_t *tt); 4959 llvm::Function *TaskEntry = emitProxyTaskFunction( 4960 CGM, Loc, D.getDirectiveKind(), KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 4961 KmpTaskTWithPrivatesQTy, KmpTaskTQTy, SharedsPtrTy, TaskFunction, 4962 TaskPrivatesMap); 4963 4964 // Build call kmp_task_t * __kmpc_omp_task_alloc(ident_t *, kmp_int32 gtid, 4965 // kmp_int32 flags, size_t sizeof_kmp_task_t, size_t sizeof_shareds, 4966 // kmp_routine_entry_t *task_entry); 4967 // Task flags. Format is taken from 4968 // https://github.com/llvm/llvm-project/blob/master/openmp/runtime/src/kmp.h, 4969 // description of kmp_tasking_flags struct. 4970 enum { 4971 TiedFlag = 0x1, 4972 FinalFlag = 0x2, 4973 DestructorsFlag = 0x8, 4974 PriorityFlag = 0x20 4975 }; 4976 unsigned Flags = Data.Tied ? TiedFlag : 0; 4977 bool NeedsCleanup = false; 4978 if (!Privates.empty()) { 4979 NeedsCleanup = checkDestructorsRequired(KmpTaskTWithPrivatesQTyRD); 4980 if (NeedsCleanup) 4981 Flags = Flags | DestructorsFlag; 4982 } 4983 if (Data.Priority.getInt()) 4984 Flags = Flags | PriorityFlag; 4985 llvm::Value *TaskFlags = 4986 Data.Final.getPointer() 4987 ? CGF.Builder.CreateSelect(Data.Final.getPointer(), 4988 CGF.Builder.getInt32(FinalFlag), 4989 CGF.Builder.getInt32(/*C=*/0)) 4990 : CGF.Builder.getInt32(Data.Final.getInt() ? FinalFlag : 0); 4991 TaskFlags = CGF.Builder.CreateOr(TaskFlags, CGF.Builder.getInt32(Flags)); 4992 llvm::Value *SharedsSize = CGM.getSize(C.getTypeSizeInChars(SharedsTy)); 4993 llvm::Value *AllocArgs[] = {emitUpdateLocation(CGF, Loc), 4994 getThreadID(CGF, Loc), TaskFlags, 4995 KmpTaskTWithPrivatesTySize, SharedsSize, 4996 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 4997 TaskEntry, KmpRoutineEntryPtrTy)}; 4998 llvm::Value *NewTask = CGF.EmitRuntimeCall( 4999 createRuntimeFunction(OMPRTL__kmpc_omp_task_alloc), AllocArgs); 5000 llvm::Value *NewTaskNewTaskTTy = 5001 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5002 NewTask, KmpTaskTWithPrivatesPtrTy); 5003 LValue Base = CGF.MakeNaturalAlignAddrLValue(NewTaskNewTaskTTy, 5004 KmpTaskTWithPrivatesQTy); 5005 LValue TDBase = 5006 CGF.EmitLValueForField(Base, *KmpTaskTWithPrivatesQTyRD->field_begin()); 5007 // Fill the data in the resulting kmp_task_t record. 5008 // Copy shareds if there are any. 5009 Address KmpTaskSharedsPtr = Address::invalid(); 5010 if (!SharedsTy->getAsStructureType()->getDecl()->field_empty()) { 5011 KmpTaskSharedsPtr = 5012 Address(CGF.EmitLoadOfScalar( 5013 CGF.EmitLValueForField( 5014 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), 5015 KmpTaskTShareds)), 5016 Loc), 5017 CGF.getNaturalTypeAlignment(SharedsTy)); 5018 LValue Dest = CGF.MakeAddrLValue(KmpTaskSharedsPtr, SharedsTy); 5019 LValue Src = CGF.MakeAddrLValue(Shareds, SharedsTy); 5020 CGF.EmitAggregateCopy(Dest, Src, SharedsTy, AggValueSlot::DoesNotOverlap); 5021 } 5022 // Emit initial values for private copies (if any). 5023 TaskResultTy Result; 5024 if (!Privates.empty()) { 5025 emitPrivatesInit(CGF, D, KmpTaskSharedsPtr, Base, KmpTaskTWithPrivatesQTyRD, 5026 SharedsTy, SharedsPtrTy, Data, Privates, 5027 /*ForDup=*/false); 5028 if (isOpenMPTaskLoopDirective(D.getDirectiveKind()) && 5029 (!Data.LastprivateVars.empty() || checkInitIsRequired(CGF, Privates))) { 5030 Result.TaskDupFn = emitTaskDupFunction( 5031 CGM, Loc, D, KmpTaskTWithPrivatesPtrQTy, KmpTaskTWithPrivatesQTyRD, 5032 KmpTaskTQTyRD, SharedsTy, SharedsPtrTy, Data, Privates, 5033 /*WithLastIter=*/!Data.LastprivateVars.empty()); 5034 } 5035 } 5036 // Fields of union "kmp_cmplrdata_t" for destructors and priority. 5037 enum { Priority = 0, Destructors = 1 }; 5038 // Provide pointer to function with destructors for privates. 5039 auto FI = std::next(KmpTaskTQTyRD->field_begin(), Data1); 5040 const RecordDecl *KmpCmplrdataUD = 5041 (*FI)->getType()->getAsUnionType()->getDecl(); 5042 if (NeedsCleanup) { 5043 llvm::Value *DestructorFn = emitDestructorsFunction( 5044 CGM, Loc, KmpInt32Ty, KmpTaskTWithPrivatesPtrQTy, 5045 KmpTaskTWithPrivatesQTy); 5046 LValue Data1LV = CGF.EmitLValueForField(TDBase, *FI); 5047 LValue DestructorsLV = CGF.EmitLValueForField( 5048 Data1LV, *std::next(KmpCmplrdataUD->field_begin(), Destructors)); 5049 CGF.EmitStoreOfScalar(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5050 DestructorFn, KmpRoutineEntryPtrTy), 5051 DestructorsLV); 5052 } 5053 // Set priority. 5054 if (Data.Priority.getInt()) { 5055 LValue Data2LV = CGF.EmitLValueForField( 5056 TDBase, *std::next(KmpTaskTQTyRD->field_begin(), Data2)); 5057 LValue PriorityLV = CGF.EmitLValueForField( 5058 Data2LV, *std::next(KmpCmplrdataUD->field_begin(), Priority)); 5059 CGF.EmitStoreOfScalar(Data.Priority.getPointer(), PriorityLV); 5060 } 5061 Result.NewTask = NewTask; 5062 Result.TaskEntry = TaskEntry; 5063 Result.NewTaskNewTaskTTy = NewTaskNewTaskTTy; 5064 Result.TDBase = TDBase; 5065 Result.KmpTaskTQTyRD = KmpTaskTQTyRD; 5066 return Result; 5067 } 5068 5069 void CGOpenMPRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 5070 const OMPExecutableDirective &D, 5071 llvm::Function *TaskFunction, 5072 QualType SharedsTy, Address Shareds, 5073 const Expr *IfCond, 5074 const OMPTaskDataTy &Data) { 5075 if (!CGF.HaveInsertPoint()) 5076 return; 5077 5078 TaskResultTy Result = 5079 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5080 llvm::Value *NewTask = Result.NewTask; 5081 llvm::Function *TaskEntry = Result.TaskEntry; 5082 llvm::Value *NewTaskNewTaskTTy = Result.NewTaskNewTaskTTy; 5083 LValue TDBase = Result.TDBase; 5084 const RecordDecl *KmpTaskTQTyRD = Result.KmpTaskTQTyRD; 5085 ASTContext &C = CGM.getContext(); 5086 // Process list of dependences. 5087 Address DependenciesArray = Address::invalid(); 5088 unsigned NumDependencies = Data.Dependences.size(); 5089 if (NumDependencies) { 5090 // Dependence kind for RTL. 5091 enum RTLDependenceKindTy { DepIn = 0x01, DepInOut = 0x3, DepMutexInOutSet = 0x4 }; 5092 enum RTLDependInfoFieldsTy { BaseAddr, Len, Flags }; 5093 RecordDecl *KmpDependInfoRD; 5094 QualType FlagsTy = 5095 C.getIntTypeForBitwidth(C.getTypeSize(C.BoolTy), /*Signed=*/false); 5096 llvm::Type *LLVMFlagsTy = CGF.ConvertTypeForMem(FlagsTy); 5097 if (KmpDependInfoTy.isNull()) { 5098 KmpDependInfoRD = C.buildImplicitRecord("kmp_depend_info"); 5099 KmpDependInfoRD->startDefinition(); 5100 addFieldToRecordDecl(C, KmpDependInfoRD, C.getIntPtrType()); 5101 addFieldToRecordDecl(C, KmpDependInfoRD, C.getSizeType()); 5102 addFieldToRecordDecl(C, KmpDependInfoRD, FlagsTy); 5103 KmpDependInfoRD->completeDefinition(); 5104 KmpDependInfoTy = C.getRecordType(KmpDependInfoRD); 5105 } else { 5106 KmpDependInfoRD = cast<RecordDecl>(KmpDependInfoTy->getAsTagDecl()); 5107 } 5108 CharUnits DependencySize = C.getTypeSizeInChars(KmpDependInfoTy); 5109 // Define type kmp_depend_info[<Dependences.size()>]; 5110 QualType KmpDependInfoArrayTy = C.getConstantArrayType( 5111 KmpDependInfoTy, llvm::APInt(/*numBits=*/64, NumDependencies), 5112 ArrayType::Normal, /*IndexTypeQuals=*/0); 5113 // kmp_depend_info[<Dependences.size()>] deps; 5114 DependenciesArray = 5115 CGF.CreateMemTemp(KmpDependInfoArrayTy, ".dep.arr.addr"); 5116 for (unsigned I = 0; I < NumDependencies; ++I) { 5117 const Expr *E = Data.Dependences[I].second; 5118 LValue Addr = CGF.EmitLValue(E); 5119 llvm::Value *Size; 5120 QualType Ty = E->getType(); 5121 if (const auto *ASE = 5122 dyn_cast<OMPArraySectionExpr>(E->IgnoreParenImpCasts())) { 5123 LValue UpAddrLVal = 5124 CGF.EmitOMPArraySectionExpr(ASE, /*LowerBound=*/false); 5125 llvm::Value *UpAddr = 5126 CGF.Builder.CreateConstGEP1_32(UpAddrLVal.getPointer(), /*Idx0=*/1); 5127 llvm::Value *LowIntPtr = 5128 CGF.Builder.CreatePtrToInt(Addr.getPointer(), CGM.SizeTy); 5129 llvm::Value *UpIntPtr = CGF.Builder.CreatePtrToInt(UpAddr, CGM.SizeTy); 5130 Size = CGF.Builder.CreateNUWSub(UpIntPtr, LowIntPtr); 5131 } else { 5132 Size = CGF.getTypeSize(Ty); 5133 } 5134 LValue Base = CGF.MakeAddrLValue( 5135 CGF.Builder.CreateConstArrayGEP(DependenciesArray, I, DependencySize), 5136 KmpDependInfoTy); 5137 // deps[i].base_addr = &<Dependences[i].second>; 5138 LValue BaseAddrLVal = CGF.EmitLValueForField( 5139 Base, *std::next(KmpDependInfoRD->field_begin(), BaseAddr)); 5140 CGF.EmitStoreOfScalar( 5141 CGF.Builder.CreatePtrToInt(Addr.getPointer(), CGF.IntPtrTy), 5142 BaseAddrLVal); 5143 // deps[i].len = sizeof(<Dependences[i].second>); 5144 LValue LenLVal = CGF.EmitLValueForField( 5145 Base, *std::next(KmpDependInfoRD->field_begin(), Len)); 5146 CGF.EmitStoreOfScalar(Size, LenLVal); 5147 // deps[i].flags = <Dependences[i].first>; 5148 RTLDependenceKindTy DepKind; 5149 switch (Data.Dependences[I].first) { 5150 case OMPC_DEPEND_in: 5151 DepKind = DepIn; 5152 break; 5153 // Out and InOut dependencies must use the same code. 5154 case OMPC_DEPEND_out: 5155 case OMPC_DEPEND_inout: 5156 DepKind = DepInOut; 5157 break; 5158 case OMPC_DEPEND_mutexinoutset: 5159 DepKind = DepMutexInOutSet; 5160 break; 5161 case OMPC_DEPEND_source: 5162 case OMPC_DEPEND_sink: 5163 case OMPC_DEPEND_unknown: 5164 llvm_unreachable("Unknown task dependence type"); 5165 } 5166 LValue FlagsLVal = CGF.EmitLValueForField( 5167 Base, *std::next(KmpDependInfoRD->field_begin(), Flags)); 5168 CGF.EmitStoreOfScalar(llvm::ConstantInt::get(LLVMFlagsTy, DepKind), 5169 FlagsLVal); 5170 } 5171 DependenciesArray = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5172 CGF.Builder.CreateStructGEP(DependenciesArray, 0, CharUnits::Zero()), 5173 CGF.VoidPtrTy); 5174 } 5175 5176 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5177 // libcall. 5178 // Build kmp_int32 __kmpc_omp_task_with_deps(ident_t *, kmp_int32 gtid, 5179 // kmp_task_t *new_task, kmp_int32 ndeps, kmp_depend_info_t *dep_list, 5180 // kmp_int32 ndeps_noalias, kmp_depend_info_t *noalias_dep_list) if dependence 5181 // list is not empty 5182 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5183 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5184 llvm::Value *TaskArgs[] = { UpLoc, ThreadID, NewTask }; 5185 llvm::Value *DepTaskArgs[7]; 5186 if (NumDependencies) { 5187 DepTaskArgs[0] = UpLoc; 5188 DepTaskArgs[1] = ThreadID; 5189 DepTaskArgs[2] = NewTask; 5190 DepTaskArgs[3] = CGF.Builder.getInt32(NumDependencies); 5191 DepTaskArgs[4] = DependenciesArray.getPointer(); 5192 DepTaskArgs[5] = CGF.Builder.getInt32(0); 5193 DepTaskArgs[6] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5194 } 5195 auto &&ThenCodeGen = [this, &Data, TDBase, KmpTaskTQTyRD, NumDependencies, 5196 &TaskArgs, 5197 &DepTaskArgs](CodeGenFunction &CGF, PrePostActionTy &) { 5198 if (!Data.Tied) { 5199 auto PartIdFI = std::next(KmpTaskTQTyRD->field_begin(), KmpTaskTPartId); 5200 LValue PartIdLVal = CGF.EmitLValueForField(TDBase, *PartIdFI); 5201 CGF.EmitStoreOfScalar(CGF.Builder.getInt32(0), PartIdLVal); 5202 } 5203 if (NumDependencies) { 5204 CGF.EmitRuntimeCall( 5205 createRuntimeFunction(OMPRTL__kmpc_omp_task_with_deps), DepTaskArgs); 5206 } else { 5207 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_task), 5208 TaskArgs); 5209 } 5210 // Check if parent region is untied and build return for untied task; 5211 if (auto *Region = 5212 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 5213 Region->emitUntiedSwitch(CGF); 5214 }; 5215 5216 llvm::Value *DepWaitTaskArgs[6]; 5217 if (NumDependencies) { 5218 DepWaitTaskArgs[0] = UpLoc; 5219 DepWaitTaskArgs[1] = ThreadID; 5220 DepWaitTaskArgs[2] = CGF.Builder.getInt32(NumDependencies); 5221 DepWaitTaskArgs[3] = DependenciesArray.getPointer(); 5222 DepWaitTaskArgs[4] = CGF.Builder.getInt32(0); 5223 DepWaitTaskArgs[5] = llvm::ConstantPointerNull::get(CGF.VoidPtrTy); 5224 } 5225 auto &&ElseCodeGen = [&TaskArgs, ThreadID, NewTaskNewTaskTTy, TaskEntry, 5226 NumDependencies, &DepWaitTaskArgs, 5227 Loc](CodeGenFunction &CGF, PrePostActionTy &) { 5228 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5229 CodeGenFunction::RunCleanupsScope LocalScope(CGF); 5230 // Build void __kmpc_omp_wait_deps(ident_t *, kmp_int32 gtid, 5231 // kmp_int32 ndeps, kmp_depend_info_t *dep_list, kmp_int32 5232 // ndeps_noalias, kmp_depend_info_t *noalias_dep_list); if dependence info 5233 // is specified. 5234 if (NumDependencies) 5235 CGF.EmitRuntimeCall(RT.createRuntimeFunction(OMPRTL__kmpc_omp_wait_deps), 5236 DepWaitTaskArgs); 5237 // Call proxy_task_entry(gtid, new_task); 5238 auto &&CodeGen = [TaskEntry, ThreadID, NewTaskNewTaskTTy, 5239 Loc](CodeGenFunction &CGF, PrePostActionTy &Action) { 5240 Action.Enter(CGF); 5241 llvm::Value *OutlinedFnArgs[] = {ThreadID, NewTaskNewTaskTTy}; 5242 CGF.CGM.getOpenMPRuntime().emitOutlinedFunctionCall(CGF, Loc, TaskEntry, 5243 OutlinedFnArgs); 5244 }; 5245 5246 // Build void __kmpc_omp_task_begin_if0(ident_t *, kmp_int32 gtid, 5247 // kmp_task_t *new_task); 5248 // Build void __kmpc_omp_task_complete_if0(ident_t *, kmp_int32 gtid, 5249 // kmp_task_t *new_task); 5250 RegionCodeGenTy RCG(CodeGen); 5251 CommonActionTy Action( 5252 RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_begin_if0), TaskArgs, 5253 RT.createRuntimeFunction(OMPRTL__kmpc_omp_task_complete_if0), TaskArgs); 5254 RCG.setAction(Action); 5255 RCG(CGF); 5256 }; 5257 5258 if (IfCond) { 5259 emitOMPIfClause(CGF, IfCond, ThenCodeGen, ElseCodeGen); 5260 } else { 5261 RegionCodeGenTy ThenRCG(ThenCodeGen); 5262 ThenRCG(CGF); 5263 } 5264 } 5265 5266 void CGOpenMPRuntime::emitTaskLoopCall(CodeGenFunction &CGF, SourceLocation Loc, 5267 const OMPLoopDirective &D, 5268 llvm::Function *TaskFunction, 5269 QualType SharedsTy, Address Shareds, 5270 const Expr *IfCond, 5271 const OMPTaskDataTy &Data) { 5272 if (!CGF.HaveInsertPoint()) 5273 return; 5274 TaskResultTy Result = 5275 emitTaskInit(CGF, Loc, D, TaskFunction, SharedsTy, Shareds, Data); 5276 // NOTE: routine and part_id fields are initialized by __kmpc_omp_task_alloc() 5277 // libcall. 5278 // Call to void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int 5279 // if_val, kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, int 5280 // sched, kmp_uint64 grainsize, void *task_dup); 5281 llvm::Value *ThreadID = getThreadID(CGF, Loc); 5282 llvm::Value *UpLoc = emitUpdateLocation(CGF, Loc); 5283 llvm::Value *IfVal; 5284 if (IfCond) { 5285 IfVal = CGF.Builder.CreateIntCast(CGF.EvaluateExprAsBool(IfCond), CGF.IntTy, 5286 /*isSigned=*/true); 5287 } else { 5288 IfVal = llvm::ConstantInt::getSigned(CGF.IntTy, /*V=*/1); 5289 } 5290 5291 LValue LBLVal = CGF.EmitLValueForField( 5292 Result.TDBase, 5293 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTLowerBound)); 5294 const auto *LBVar = 5295 cast<VarDecl>(cast<DeclRefExpr>(D.getLowerBoundVariable())->getDecl()); 5296 CGF.EmitAnyExprToMem(LBVar->getInit(), LBLVal.getAddress(), LBLVal.getQuals(), 5297 /*IsInitializer=*/true); 5298 LValue UBLVal = CGF.EmitLValueForField( 5299 Result.TDBase, 5300 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTUpperBound)); 5301 const auto *UBVar = 5302 cast<VarDecl>(cast<DeclRefExpr>(D.getUpperBoundVariable())->getDecl()); 5303 CGF.EmitAnyExprToMem(UBVar->getInit(), UBLVal.getAddress(), UBLVal.getQuals(), 5304 /*IsInitializer=*/true); 5305 LValue StLVal = CGF.EmitLValueForField( 5306 Result.TDBase, 5307 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTStride)); 5308 const auto *StVar = 5309 cast<VarDecl>(cast<DeclRefExpr>(D.getStrideVariable())->getDecl()); 5310 CGF.EmitAnyExprToMem(StVar->getInit(), StLVal.getAddress(), StLVal.getQuals(), 5311 /*IsInitializer=*/true); 5312 // Store reductions address. 5313 LValue RedLVal = CGF.EmitLValueForField( 5314 Result.TDBase, 5315 *std::next(Result.KmpTaskTQTyRD->field_begin(), KmpTaskTReductions)); 5316 if (Data.Reductions) { 5317 CGF.EmitStoreOfScalar(Data.Reductions, RedLVal); 5318 } else { 5319 CGF.EmitNullInitialization(RedLVal.getAddress(), 5320 CGF.getContext().VoidPtrTy); 5321 } 5322 enum { NoSchedule = 0, Grainsize = 1, NumTasks = 2 }; 5323 llvm::Value *TaskArgs[] = { 5324 UpLoc, 5325 ThreadID, 5326 Result.NewTask, 5327 IfVal, 5328 LBLVal.getPointer(), 5329 UBLVal.getPointer(), 5330 CGF.EmitLoadOfScalar(StLVal, Loc), 5331 llvm::ConstantInt::getSigned( 5332 CGF.IntTy, 1), // Always 1 because taskgroup emitted by the compiler 5333 llvm::ConstantInt::getSigned( 5334 CGF.IntTy, Data.Schedule.getPointer() 5335 ? Data.Schedule.getInt() ? NumTasks : Grainsize 5336 : NoSchedule), 5337 Data.Schedule.getPointer() 5338 ? CGF.Builder.CreateIntCast(Data.Schedule.getPointer(), CGF.Int64Ty, 5339 /*isSigned=*/false) 5340 : llvm::ConstantInt::get(CGF.Int64Ty, /*V=*/0), 5341 Result.TaskDupFn ? CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5342 Result.TaskDupFn, CGF.VoidPtrTy) 5343 : llvm::ConstantPointerNull::get(CGF.VoidPtrTy)}; 5344 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_taskloop), TaskArgs); 5345 } 5346 5347 /// Emit reduction operation for each element of array (required for 5348 /// array sections) LHS op = RHS. 5349 /// \param Type Type of array. 5350 /// \param LHSVar Variable on the left side of the reduction operation 5351 /// (references element of array in original variable). 5352 /// \param RHSVar Variable on the right side of the reduction operation 5353 /// (references element of array in original variable). 5354 /// \param RedOpGen Generator of reduction operation with use of LHSVar and 5355 /// RHSVar. 5356 static void EmitOMPAggregateReduction( 5357 CodeGenFunction &CGF, QualType Type, const VarDecl *LHSVar, 5358 const VarDecl *RHSVar, 5359 const llvm::function_ref<void(CodeGenFunction &CGF, const Expr *, 5360 const Expr *, const Expr *)> &RedOpGen, 5361 const Expr *XExpr = nullptr, const Expr *EExpr = nullptr, 5362 const Expr *UpExpr = nullptr) { 5363 // Perform element-by-element initialization. 5364 QualType ElementTy; 5365 Address LHSAddr = CGF.GetAddrOfLocalVar(LHSVar); 5366 Address RHSAddr = CGF.GetAddrOfLocalVar(RHSVar); 5367 5368 // Drill down to the base element type on both arrays. 5369 const ArrayType *ArrayTy = Type->getAsArrayTypeUnsafe(); 5370 llvm::Value *NumElements = CGF.emitArrayLength(ArrayTy, ElementTy, LHSAddr); 5371 5372 llvm::Value *RHSBegin = RHSAddr.getPointer(); 5373 llvm::Value *LHSBegin = LHSAddr.getPointer(); 5374 // Cast from pointer to array type to pointer to single element. 5375 llvm::Value *LHSEnd = CGF.Builder.CreateGEP(LHSBegin, NumElements); 5376 // The basic structure here is a while-do loop. 5377 llvm::BasicBlock *BodyBB = CGF.createBasicBlock("omp.arraycpy.body"); 5378 llvm::BasicBlock *DoneBB = CGF.createBasicBlock("omp.arraycpy.done"); 5379 llvm::Value *IsEmpty = 5380 CGF.Builder.CreateICmpEQ(LHSBegin, LHSEnd, "omp.arraycpy.isempty"); 5381 CGF.Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB); 5382 5383 // Enter the loop body, making that address the current address. 5384 llvm::BasicBlock *EntryBB = CGF.Builder.GetInsertBlock(); 5385 CGF.EmitBlock(BodyBB); 5386 5387 CharUnits ElementSize = CGF.getContext().getTypeSizeInChars(ElementTy); 5388 5389 llvm::PHINode *RHSElementPHI = CGF.Builder.CreatePHI( 5390 RHSBegin->getType(), 2, "omp.arraycpy.srcElementPast"); 5391 RHSElementPHI->addIncoming(RHSBegin, EntryBB); 5392 Address RHSElementCurrent = 5393 Address(RHSElementPHI, 5394 RHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5395 5396 llvm::PHINode *LHSElementPHI = CGF.Builder.CreatePHI( 5397 LHSBegin->getType(), 2, "omp.arraycpy.destElementPast"); 5398 LHSElementPHI->addIncoming(LHSBegin, EntryBB); 5399 Address LHSElementCurrent = 5400 Address(LHSElementPHI, 5401 LHSAddr.getAlignment().alignmentOfArrayElement(ElementSize)); 5402 5403 // Emit copy. 5404 CodeGenFunction::OMPPrivateScope Scope(CGF); 5405 Scope.addPrivate(LHSVar, [=]() { return LHSElementCurrent; }); 5406 Scope.addPrivate(RHSVar, [=]() { return RHSElementCurrent; }); 5407 Scope.Privatize(); 5408 RedOpGen(CGF, XExpr, EExpr, UpExpr); 5409 Scope.ForceCleanup(); 5410 5411 // Shift the address forward by one element. 5412 llvm::Value *LHSElementNext = CGF.Builder.CreateConstGEP1_32( 5413 LHSElementPHI, /*Idx0=*/1, "omp.arraycpy.dest.element"); 5414 llvm::Value *RHSElementNext = CGF.Builder.CreateConstGEP1_32( 5415 RHSElementPHI, /*Idx0=*/1, "omp.arraycpy.src.element"); 5416 // Check whether we've reached the end. 5417 llvm::Value *Done = 5418 CGF.Builder.CreateICmpEQ(LHSElementNext, LHSEnd, "omp.arraycpy.done"); 5419 CGF.Builder.CreateCondBr(Done, DoneBB, BodyBB); 5420 LHSElementPHI->addIncoming(LHSElementNext, CGF.Builder.GetInsertBlock()); 5421 RHSElementPHI->addIncoming(RHSElementNext, CGF.Builder.GetInsertBlock()); 5422 5423 // Done. 5424 CGF.EmitBlock(DoneBB, /*IsFinished=*/true); 5425 } 5426 5427 /// Emit reduction combiner. If the combiner is a simple expression emit it as 5428 /// is, otherwise consider it as combiner of UDR decl and emit it as a call of 5429 /// UDR combiner function. 5430 static void emitReductionCombiner(CodeGenFunction &CGF, 5431 const Expr *ReductionOp) { 5432 if (const auto *CE = dyn_cast<CallExpr>(ReductionOp)) 5433 if (const auto *OVE = dyn_cast<OpaqueValueExpr>(CE->getCallee())) 5434 if (const auto *DRE = 5435 dyn_cast<DeclRefExpr>(OVE->getSourceExpr()->IgnoreImpCasts())) 5436 if (const auto *DRD = 5437 dyn_cast<OMPDeclareReductionDecl>(DRE->getDecl())) { 5438 std::pair<llvm::Function *, llvm::Function *> Reduction = 5439 CGF.CGM.getOpenMPRuntime().getUserDefinedReduction(DRD); 5440 RValue Func = RValue::get(Reduction.first); 5441 CodeGenFunction::OpaqueValueMapping Map(CGF, OVE, Func); 5442 CGF.EmitIgnoredExpr(ReductionOp); 5443 return; 5444 } 5445 CGF.EmitIgnoredExpr(ReductionOp); 5446 } 5447 5448 llvm::Function *CGOpenMPRuntime::emitReductionFunction( 5449 CodeGenModule &CGM, SourceLocation Loc, llvm::Type *ArgsType, 5450 ArrayRef<const Expr *> Privates, ArrayRef<const Expr *> LHSExprs, 5451 ArrayRef<const Expr *> RHSExprs, ArrayRef<const Expr *> ReductionOps) { 5452 ASTContext &C = CGM.getContext(); 5453 5454 // void reduction_func(void *LHSArg, void *RHSArg); 5455 FunctionArgList Args; 5456 ImplicitParamDecl LHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5457 ImplicitParamDecl::Other); 5458 ImplicitParamDecl RHSArg(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5459 ImplicitParamDecl::Other); 5460 Args.push_back(&LHSArg); 5461 Args.push_back(&RHSArg); 5462 const auto &CGFI = 5463 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5464 std::string Name = getName({"omp", "reduction", "reduction_func"}); 5465 auto *Fn = llvm::Function::Create(CGM.getTypes().GetFunctionType(CGFI), 5466 llvm::GlobalValue::InternalLinkage, Name, 5467 &CGM.getModule()); 5468 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, CGFI); 5469 Fn->setDoesNotRecurse(); 5470 CodeGenFunction CGF(CGM); 5471 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, CGFI, Args, Loc, Loc); 5472 5473 // Dst = (void*[n])(LHSArg); 5474 // Src = (void*[n])(RHSArg); 5475 Address LHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5476 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&LHSArg)), 5477 ArgsType), CGF.getPointerAlign()); 5478 Address RHS(CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5479 CGF.Builder.CreateLoad(CGF.GetAddrOfLocalVar(&RHSArg)), 5480 ArgsType), CGF.getPointerAlign()); 5481 5482 // ... 5483 // *(Type<i>*)lhs[i] = RedOp<i>(*(Type<i>*)lhs[i], *(Type<i>*)rhs[i]); 5484 // ... 5485 CodeGenFunction::OMPPrivateScope Scope(CGF); 5486 auto IPriv = Privates.begin(); 5487 unsigned Idx = 0; 5488 for (unsigned I = 0, E = ReductionOps.size(); I < E; ++I, ++IPriv, ++Idx) { 5489 const auto *RHSVar = 5490 cast<VarDecl>(cast<DeclRefExpr>(RHSExprs[I])->getDecl()); 5491 Scope.addPrivate(RHSVar, [&CGF, RHS, Idx, RHSVar]() { 5492 return emitAddrOfVarFromArray(CGF, RHS, Idx, RHSVar); 5493 }); 5494 const auto *LHSVar = 5495 cast<VarDecl>(cast<DeclRefExpr>(LHSExprs[I])->getDecl()); 5496 Scope.addPrivate(LHSVar, [&CGF, LHS, Idx, LHSVar]() { 5497 return emitAddrOfVarFromArray(CGF, LHS, Idx, LHSVar); 5498 }); 5499 QualType PrivTy = (*IPriv)->getType(); 5500 if (PrivTy->isVariablyModifiedType()) { 5501 // Get array size and emit VLA type. 5502 ++Idx; 5503 Address Elem = 5504 CGF.Builder.CreateConstArrayGEP(LHS, Idx, CGF.getPointerSize()); 5505 llvm::Value *Ptr = CGF.Builder.CreateLoad(Elem); 5506 const VariableArrayType *VLA = 5507 CGF.getContext().getAsVariableArrayType(PrivTy); 5508 const auto *OVE = cast<OpaqueValueExpr>(VLA->getSizeExpr()); 5509 CodeGenFunction::OpaqueValueMapping OpaqueMap( 5510 CGF, OVE, RValue::get(CGF.Builder.CreatePtrToInt(Ptr, CGF.SizeTy))); 5511 CGF.EmitVariablyModifiedType(PrivTy); 5512 } 5513 } 5514 Scope.Privatize(); 5515 IPriv = Privates.begin(); 5516 auto ILHS = LHSExprs.begin(); 5517 auto IRHS = RHSExprs.begin(); 5518 for (const Expr *E : ReductionOps) { 5519 if ((*IPriv)->getType()->isArrayType()) { 5520 // Emit reduction for array section. 5521 const auto *LHSVar = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5522 const auto *RHSVar = cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5523 EmitOMPAggregateReduction( 5524 CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5525 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5526 emitReductionCombiner(CGF, E); 5527 }); 5528 } else { 5529 // Emit reduction for array subscript or single variable. 5530 emitReductionCombiner(CGF, E); 5531 } 5532 ++IPriv; 5533 ++ILHS; 5534 ++IRHS; 5535 } 5536 Scope.ForceCleanup(); 5537 CGF.FinishFunction(); 5538 return Fn; 5539 } 5540 5541 void CGOpenMPRuntime::emitSingleReductionCombiner(CodeGenFunction &CGF, 5542 const Expr *ReductionOp, 5543 const Expr *PrivateRef, 5544 const DeclRefExpr *LHS, 5545 const DeclRefExpr *RHS) { 5546 if (PrivateRef->getType()->isArrayType()) { 5547 // Emit reduction for array section. 5548 const auto *LHSVar = cast<VarDecl>(LHS->getDecl()); 5549 const auto *RHSVar = cast<VarDecl>(RHS->getDecl()); 5550 EmitOMPAggregateReduction( 5551 CGF, PrivateRef->getType(), LHSVar, RHSVar, 5552 [=](CodeGenFunction &CGF, const Expr *, const Expr *, const Expr *) { 5553 emitReductionCombiner(CGF, ReductionOp); 5554 }); 5555 } else { 5556 // Emit reduction for array subscript or single variable. 5557 emitReductionCombiner(CGF, ReductionOp); 5558 } 5559 } 5560 5561 void CGOpenMPRuntime::emitReduction(CodeGenFunction &CGF, SourceLocation Loc, 5562 ArrayRef<const Expr *> Privates, 5563 ArrayRef<const Expr *> LHSExprs, 5564 ArrayRef<const Expr *> RHSExprs, 5565 ArrayRef<const Expr *> ReductionOps, 5566 ReductionOptionsTy Options) { 5567 if (!CGF.HaveInsertPoint()) 5568 return; 5569 5570 bool WithNowait = Options.WithNowait; 5571 bool SimpleReduction = Options.SimpleReduction; 5572 5573 // Next code should be emitted for reduction: 5574 // 5575 // static kmp_critical_name lock = { 0 }; 5576 // 5577 // void reduce_func(void *lhs[<n>], void *rhs[<n>]) { 5578 // *(Type0*)lhs[0] = ReductionOperation0(*(Type0*)lhs[0], *(Type0*)rhs[0]); 5579 // ... 5580 // *(Type<n>-1*)lhs[<n>-1] = ReductionOperation<n>-1(*(Type<n>-1*)lhs[<n>-1], 5581 // *(Type<n>-1*)rhs[<n>-1]); 5582 // } 5583 // 5584 // ... 5585 // void *RedList[<n>] = {&<RHSExprs>[0], ..., &<RHSExprs>[<n>-1]}; 5586 // switch (__kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5587 // RedList, reduce_func, &<lock>)) { 5588 // case 1: 5589 // ... 5590 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5591 // ... 5592 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5593 // break; 5594 // case 2: 5595 // ... 5596 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5597 // ... 5598 // [__kmpc_end_reduce(<loc>, <gtid>, &<lock>);] 5599 // break; 5600 // default:; 5601 // } 5602 // 5603 // if SimpleReduction is true, only the next code is generated: 5604 // ... 5605 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5606 // ... 5607 5608 ASTContext &C = CGM.getContext(); 5609 5610 if (SimpleReduction) { 5611 CodeGenFunction::RunCleanupsScope Scope(CGF); 5612 auto IPriv = Privates.begin(); 5613 auto ILHS = LHSExprs.begin(); 5614 auto IRHS = RHSExprs.begin(); 5615 for (const Expr *E : ReductionOps) { 5616 emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5617 cast<DeclRefExpr>(*IRHS)); 5618 ++IPriv; 5619 ++ILHS; 5620 ++IRHS; 5621 } 5622 return; 5623 } 5624 5625 // 1. Build a list of reduction variables. 5626 // void *RedList[<n>] = {<ReductionVars>[0], ..., <ReductionVars>[<n>-1]}; 5627 auto Size = RHSExprs.size(); 5628 for (const Expr *E : Privates) { 5629 if (E->getType()->isVariablyModifiedType()) 5630 // Reserve place for array size. 5631 ++Size; 5632 } 5633 llvm::APInt ArraySize(/*unsigned int numBits=*/32, Size); 5634 QualType ReductionArrayTy = 5635 C.getConstantArrayType(C.VoidPtrTy, ArraySize, ArrayType::Normal, 5636 /*IndexTypeQuals=*/0); 5637 Address ReductionList = 5638 CGF.CreateMemTemp(ReductionArrayTy, ".omp.reduction.red_list"); 5639 auto IPriv = Privates.begin(); 5640 unsigned Idx = 0; 5641 for (unsigned I = 0, E = RHSExprs.size(); I < E; ++I, ++IPriv, ++Idx) { 5642 Address Elem = 5643 CGF.Builder.CreateConstArrayGEP(ReductionList, Idx, CGF.getPointerSize()); 5644 CGF.Builder.CreateStore( 5645 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5646 CGF.EmitLValue(RHSExprs[I]).getPointer(), CGF.VoidPtrTy), 5647 Elem); 5648 if ((*IPriv)->getType()->isVariablyModifiedType()) { 5649 // Store array size. 5650 ++Idx; 5651 Elem = CGF.Builder.CreateConstArrayGEP(ReductionList, Idx, 5652 CGF.getPointerSize()); 5653 llvm::Value *Size = CGF.Builder.CreateIntCast( 5654 CGF.getVLASize( 5655 CGF.getContext().getAsVariableArrayType((*IPriv)->getType())) 5656 .NumElts, 5657 CGF.SizeTy, /*isSigned=*/false); 5658 CGF.Builder.CreateStore(CGF.Builder.CreateIntToPtr(Size, CGF.VoidPtrTy), 5659 Elem); 5660 } 5661 } 5662 5663 // 2. Emit reduce_func(). 5664 llvm::Function *ReductionFn = emitReductionFunction( 5665 CGM, Loc, CGF.ConvertTypeForMem(ReductionArrayTy)->getPointerTo(), 5666 Privates, LHSExprs, RHSExprs, ReductionOps); 5667 5668 // 3. Create static kmp_critical_name lock = { 0 }; 5669 std::string Name = getName({"reduction"}); 5670 llvm::Value *Lock = getCriticalRegionLock(Name); 5671 5672 // 4. Build res = __kmpc_reduce{_nowait}(<loc>, <gtid>, <n>, sizeof(RedList), 5673 // RedList, reduce_func, &<lock>); 5674 llvm::Value *IdentTLoc = emitUpdateLocation(CGF, Loc, OMP_ATOMIC_REDUCE); 5675 llvm::Value *ThreadId = getThreadID(CGF, Loc); 5676 llvm::Value *ReductionArrayTySize = CGF.getTypeSize(ReductionArrayTy); 5677 llvm::Value *RL = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 5678 ReductionList.getPointer(), CGF.VoidPtrTy); 5679 llvm::Value *Args[] = { 5680 IdentTLoc, // ident_t *<loc> 5681 ThreadId, // i32 <gtid> 5682 CGF.Builder.getInt32(RHSExprs.size()), // i32 <n> 5683 ReductionArrayTySize, // size_type sizeof(RedList) 5684 RL, // void *RedList 5685 ReductionFn, // void (*) (void *, void *) <reduce_func> 5686 Lock // kmp_critical_name *&<lock> 5687 }; 5688 llvm::Value *Res = CGF.EmitRuntimeCall( 5689 createRuntimeFunction(WithNowait ? OMPRTL__kmpc_reduce_nowait 5690 : OMPRTL__kmpc_reduce), 5691 Args); 5692 5693 // 5. Build switch(res) 5694 llvm::BasicBlock *DefaultBB = CGF.createBasicBlock(".omp.reduction.default"); 5695 llvm::SwitchInst *SwInst = 5696 CGF.Builder.CreateSwitch(Res, DefaultBB, /*NumCases=*/2); 5697 5698 // 6. Build case 1: 5699 // ... 5700 // <LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i]); 5701 // ... 5702 // __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5703 // break; 5704 llvm::BasicBlock *Case1BB = CGF.createBasicBlock(".omp.reduction.case1"); 5705 SwInst->addCase(CGF.Builder.getInt32(1), Case1BB); 5706 CGF.EmitBlock(Case1BB); 5707 5708 // Add emission of __kmpc_end_reduce{_nowait}(<loc>, <gtid>, &<lock>); 5709 llvm::Value *EndArgs[] = { 5710 IdentTLoc, // ident_t *<loc> 5711 ThreadId, // i32 <gtid> 5712 Lock // kmp_critical_name *&<lock> 5713 }; 5714 auto &&CodeGen = [Privates, LHSExprs, RHSExprs, ReductionOps]( 5715 CodeGenFunction &CGF, PrePostActionTy &Action) { 5716 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5717 auto IPriv = Privates.begin(); 5718 auto ILHS = LHSExprs.begin(); 5719 auto IRHS = RHSExprs.begin(); 5720 for (const Expr *E : ReductionOps) { 5721 RT.emitSingleReductionCombiner(CGF, E, *IPriv, cast<DeclRefExpr>(*ILHS), 5722 cast<DeclRefExpr>(*IRHS)); 5723 ++IPriv; 5724 ++ILHS; 5725 ++IRHS; 5726 } 5727 }; 5728 RegionCodeGenTy RCG(CodeGen); 5729 CommonActionTy Action( 5730 nullptr, llvm::None, 5731 createRuntimeFunction(WithNowait ? OMPRTL__kmpc_end_reduce_nowait 5732 : OMPRTL__kmpc_end_reduce), 5733 EndArgs); 5734 RCG.setAction(Action); 5735 RCG(CGF); 5736 5737 CGF.EmitBranch(DefaultBB); 5738 5739 // 7. Build case 2: 5740 // ... 5741 // Atomic(<LHSExprs>[i] = RedOp<i>(*<LHSExprs>[i], *<RHSExprs>[i])); 5742 // ... 5743 // break; 5744 llvm::BasicBlock *Case2BB = CGF.createBasicBlock(".omp.reduction.case2"); 5745 SwInst->addCase(CGF.Builder.getInt32(2), Case2BB); 5746 CGF.EmitBlock(Case2BB); 5747 5748 auto &&AtomicCodeGen = [Loc, Privates, LHSExprs, RHSExprs, ReductionOps]( 5749 CodeGenFunction &CGF, PrePostActionTy &Action) { 5750 auto ILHS = LHSExprs.begin(); 5751 auto IRHS = RHSExprs.begin(); 5752 auto IPriv = Privates.begin(); 5753 for (const Expr *E : ReductionOps) { 5754 const Expr *XExpr = nullptr; 5755 const Expr *EExpr = nullptr; 5756 const Expr *UpExpr = nullptr; 5757 BinaryOperatorKind BO = BO_Comma; 5758 if (const auto *BO = dyn_cast<BinaryOperator>(E)) { 5759 if (BO->getOpcode() == BO_Assign) { 5760 XExpr = BO->getLHS(); 5761 UpExpr = BO->getRHS(); 5762 } 5763 } 5764 // Try to emit update expression as a simple atomic. 5765 const Expr *RHSExpr = UpExpr; 5766 if (RHSExpr) { 5767 // Analyze RHS part of the whole expression. 5768 if (const auto *ACO = dyn_cast<AbstractConditionalOperator>( 5769 RHSExpr->IgnoreParenImpCasts())) { 5770 // If this is a conditional operator, analyze its condition for 5771 // min/max reduction operator. 5772 RHSExpr = ACO->getCond(); 5773 } 5774 if (const auto *BORHS = 5775 dyn_cast<BinaryOperator>(RHSExpr->IgnoreParenImpCasts())) { 5776 EExpr = BORHS->getRHS(); 5777 BO = BORHS->getOpcode(); 5778 } 5779 } 5780 if (XExpr) { 5781 const auto *VD = cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5782 auto &&AtomicRedGen = [BO, VD, 5783 Loc](CodeGenFunction &CGF, const Expr *XExpr, 5784 const Expr *EExpr, const Expr *UpExpr) { 5785 LValue X = CGF.EmitLValue(XExpr); 5786 RValue E; 5787 if (EExpr) 5788 E = CGF.EmitAnyExpr(EExpr); 5789 CGF.EmitOMPAtomicSimpleUpdateExpr( 5790 X, E, BO, /*IsXLHSInRHSPart=*/true, 5791 llvm::AtomicOrdering::Monotonic, Loc, 5792 [&CGF, UpExpr, VD, Loc](RValue XRValue) { 5793 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 5794 PrivateScope.addPrivate( 5795 VD, [&CGF, VD, XRValue, Loc]() { 5796 Address LHSTemp = CGF.CreateMemTemp(VD->getType()); 5797 CGF.emitOMPSimpleStore( 5798 CGF.MakeAddrLValue(LHSTemp, VD->getType()), XRValue, 5799 VD->getType().getNonReferenceType(), Loc); 5800 return LHSTemp; 5801 }); 5802 (void)PrivateScope.Privatize(); 5803 return CGF.EmitAnyExpr(UpExpr); 5804 }); 5805 }; 5806 if ((*IPriv)->getType()->isArrayType()) { 5807 // Emit atomic reduction for array section. 5808 const auto *RHSVar = 5809 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5810 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), VD, RHSVar, 5811 AtomicRedGen, XExpr, EExpr, UpExpr); 5812 } else { 5813 // Emit atomic reduction for array subscript or single variable. 5814 AtomicRedGen(CGF, XExpr, EExpr, UpExpr); 5815 } 5816 } else { 5817 // Emit as a critical region. 5818 auto &&CritRedGen = [E, Loc](CodeGenFunction &CGF, const Expr *, 5819 const Expr *, const Expr *) { 5820 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 5821 std::string Name = RT.getName({"atomic_reduction"}); 5822 RT.emitCriticalRegion( 5823 CGF, Name, 5824 [=](CodeGenFunction &CGF, PrePostActionTy &Action) { 5825 Action.Enter(CGF); 5826 emitReductionCombiner(CGF, E); 5827 }, 5828 Loc); 5829 }; 5830 if ((*IPriv)->getType()->isArrayType()) { 5831 const auto *LHSVar = 5832 cast<VarDecl>(cast<DeclRefExpr>(*ILHS)->getDecl()); 5833 const auto *RHSVar = 5834 cast<VarDecl>(cast<DeclRefExpr>(*IRHS)->getDecl()); 5835 EmitOMPAggregateReduction(CGF, (*IPriv)->getType(), LHSVar, RHSVar, 5836 CritRedGen); 5837 } else { 5838 CritRedGen(CGF, nullptr, nullptr, nullptr); 5839 } 5840 } 5841 ++ILHS; 5842 ++IRHS; 5843 ++IPriv; 5844 } 5845 }; 5846 RegionCodeGenTy AtomicRCG(AtomicCodeGen); 5847 if (!WithNowait) { 5848 // Add emission of __kmpc_end_reduce(<loc>, <gtid>, &<lock>); 5849 llvm::Value *EndArgs[] = { 5850 IdentTLoc, // ident_t *<loc> 5851 ThreadId, // i32 <gtid> 5852 Lock // kmp_critical_name *&<lock> 5853 }; 5854 CommonActionTy Action(nullptr, llvm::None, 5855 createRuntimeFunction(OMPRTL__kmpc_end_reduce), 5856 EndArgs); 5857 AtomicRCG.setAction(Action); 5858 AtomicRCG(CGF); 5859 } else { 5860 AtomicRCG(CGF); 5861 } 5862 5863 CGF.EmitBranch(DefaultBB); 5864 CGF.EmitBlock(DefaultBB, /*IsFinished=*/true); 5865 } 5866 5867 /// Generates unique name for artificial threadprivate variables. 5868 /// Format is: <Prefix> "." <Decl_mangled_name> "_" "<Decl_start_loc_raw_enc>" 5869 static std::string generateUniqueName(CodeGenModule &CGM, StringRef Prefix, 5870 const Expr *Ref) { 5871 SmallString<256> Buffer; 5872 llvm::raw_svector_ostream Out(Buffer); 5873 const clang::DeclRefExpr *DE; 5874 const VarDecl *D = ::getBaseDecl(Ref, DE); 5875 if (!D) 5876 D = cast<VarDecl>(cast<DeclRefExpr>(Ref)->getDecl()); 5877 D = D->getCanonicalDecl(); 5878 std::string Name = CGM.getOpenMPRuntime().getName( 5879 {D->isLocalVarDeclOrParm() ? D->getName() : CGM.getMangledName(D)}); 5880 Out << Prefix << Name << "_" 5881 << D->getCanonicalDecl()->getBeginLoc().getRawEncoding(); 5882 return Out.str(); 5883 } 5884 5885 /// Emits reduction initializer function: 5886 /// \code 5887 /// void @.red_init(void* %arg) { 5888 /// %0 = bitcast void* %arg to <type>* 5889 /// store <type> <init>, <type>* %0 5890 /// ret void 5891 /// } 5892 /// \endcode 5893 static llvm::Value *emitReduceInitFunction(CodeGenModule &CGM, 5894 SourceLocation Loc, 5895 ReductionCodeGen &RCG, unsigned N) { 5896 ASTContext &C = CGM.getContext(); 5897 FunctionArgList Args; 5898 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5899 ImplicitParamDecl::Other); 5900 Args.emplace_back(&Param); 5901 const auto &FnInfo = 5902 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5903 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 5904 std::string Name = CGM.getOpenMPRuntime().getName({"red_init", ""}); 5905 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 5906 Name, &CGM.getModule()); 5907 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 5908 Fn->setDoesNotRecurse(); 5909 CodeGenFunction CGF(CGM); 5910 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 5911 Address PrivateAddr = CGF.EmitLoadOfPointer( 5912 CGF.GetAddrOfLocalVar(&Param), 5913 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 5914 llvm::Value *Size = nullptr; 5915 // If the size of the reduction item is non-constant, load it from global 5916 // threadprivate variable. 5917 if (RCG.getSizes(N).second) { 5918 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 5919 CGF, CGM.getContext().getSizeType(), 5920 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 5921 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 5922 CGM.getContext().getSizeType(), Loc); 5923 } 5924 RCG.emitAggregateType(CGF, N, Size); 5925 LValue SharedLVal; 5926 // If initializer uses initializer from declare reduction construct, emit a 5927 // pointer to the address of the original reduction item (reuired by reduction 5928 // initializer) 5929 if (RCG.usesReductionInitializer(N)) { 5930 Address SharedAddr = 5931 CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 5932 CGF, CGM.getContext().VoidPtrTy, 5933 generateUniqueName(CGM, "reduction", RCG.getRefExpr(N))); 5934 SharedAddr = CGF.EmitLoadOfPointer( 5935 SharedAddr, 5936 CGM.getContext().VoidPtrTy.castAs<PointerType>()->getTypePtr()); 5937 SharedLVal = CGF.MakeAddrLValue(SharedAddr, CGM.getContext().VoidPtrTy); 5938 } else { 5939 SharedLVal = CGF.MakeNaturalAlignAddrLValue( 5940 llvm::ConstantPointerNull::get(CGM.VoidPtrTy), 5941 CGM.getContext().VoidPtrTy); 5942 } 5943 // Emit the initializer: 5944 // %0 = bitcast void* %arg to <type>* 5945 // store <type> <init>, <type>* %0 5946 RCG.emitInitialization(CGF, N, PrivateAddr, SharedLVal, 5947 [](CodeGenFunction &) { return false; }); 5948 CGF.FinishFunction(); 5949 return Fn; 5950 } 5951 5952 /// Emits reduction combiner function: 5953 /// \code 5954 /// void @.red_comb(void* %arg0, void* %arg1) { 5955 /// %lhs = bitcast void* %arg0 to <type>* 5956 /// %rhs = bitcast void* %arg1 to <type>* 5957 /// %2 = <ReductionOp>(<type>* %lhs, <type>* %rhs) 5958 /// store <type> %2, <type>* %lhs 5959 /// ret void 5960 /// } 5961 /// \endcode 5962 static llvm::Value *emitReduceCombFunction(CodeGenModule &CGM, 5963 SourceLocation Loc, 5964 ReductionCodeGen &RCG, unsigned N, 5965 const Expr *ReductionOp, 5966 const Expr *LHS, const Expr *RHS, 5967 const Expr *PrivateRef) { 5968 ASTContext &C = CGM.getContext(); 5969 const auto *LHSVD = cast<VarDecl>(cast<DeclRefExpr>(LHS)->getDecl()); 5970 const auto *RHSVD = cast<VarDecl>(cast<DeclRefExpr>(RHS)->getDecl()); 5971 FunctionArgList Args; 5972 ImplicitParamDecl ParamInOut(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, 5973 C.VoidPtrTy, ImplicitParamDecl::Other); 5974 ImplicitParamDecl ParamIn(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 5975 ImplicitParamDecl::Other); 5976 Args.emplace_back(&ParamInOut); 5977 Args.emplace_back(&ParamIn); 5978 const auto &FnInfo = 5979 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 5980 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 5981 std::string Name = CGM.getOpenMPRuntime().getName({"red_comb", ""}); 5982 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 5983 Name, &CGM.getModule()); 5984 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 5985 Fn->setDoesNotRecurse(); 5986 CodeGenFunction CGF(CGM); 5987 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 5988 llvm::Value *Size = nullptr; 5989 // If the size of the reduction item is non-constant, load it from global 5990 // threadprivate variable. 5991 if (RCG.getSizes(N).second) { 5992 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 5993 CGF, CGM.getContext().getSizeType(), 5994 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 5995 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 5996 CGM.getContext().getSizeType(), Loc); 5997 } 5998 RCG.emitAggregateType(CGF, N, Size); 5999 // Remap lhs and rhs variables to the addresses of the function arguments. 6000 // %lhs = bitcast void* %arg0 to <type>* 6001 // %rhs = bitcast void* %arg1 to <type>* 6002 CodeGenFunction::OMPPrivateScope PrivateScope(CGF); 6003 PrivateScope.addPrivate(LHSVD, [&C, &CGF, &ParamInOut, LHSVD]() { 6004 // Pull out the pointer to the variable. 6005 Address PtrAddr = CGF.EmitLoadOfPointer( 6006 CGF.GetAddrOfLocalVar(&ParamInOut), 6007 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6008 return CGF.Builder.CreateElementBitCast( 6009 PtrAddr, CGF.ConvertTypeForMem(LHSVD->getType())); 6010 }); 6011 PrivateScope.addPrivate(RHSVD, [&C, &CGF, &ParamIn, RHSVD]() { 6012 // Pull out the pointer to the variable. 6013 Address PtrAddr = CGF.EmitLoadOfPointer( 6014 CGF.GetAddrOfLocalVar(&ParamIn), 6015 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6016 return CGF.Builder.CreateElementBitCast( 6017 PtrAddr, CGF.ConvertTypeForMem(RHSVD->getType())); 6018 }); 6019 PrivateScope.Privatize(); 6020 // Emit the combiner body: 6021 // %2 = <ReductionOp>(<type> *%lhs, <type> *%rhs) 6022 // store <type> %2, <type>* %lhs 6023 CGM.getOpenMPRuntime().emitSingleReductionCombiner( 6024 CGF, ReductionOp, PrivateRef, cast<DeclRefExpr>(LHS), 6025 cast<DeclRefExpr>(RHS)); 6026 CGF.FinishFunction(); 6027 return Fn; 6028 } 6029 6030 /// Emits reduction finalizer function: 6031 /// \code 6032 /// void @.red_fini(void* %arg) { 6033 /// %0 = bitcast void* %arg to <type>* 6034 /// <destroy>(<type>* %0) 6035 /// ret void 6036 /// } 6037 /// \endcode 6038 static llvm::Value *emitReduceFiniFunction(CodeGenModule &CGM, 6039 SourceLocation Loc, 6040 ReductionCodeGen &RCG, unsigned N) { 6041 if (!RCG.needCleanups(N)) 6042 return nullptr; 6043 ASTContext &C = CGM.getContext(); 6044 FunctionArgList Args; 6045 ImplicitParamDecl Param(C, /*DC=*/nullptr, Loc, /*Id=*/nullptr, C.VoidPtrTy, 6046 ImplicitParamDecl::Other); 6047 Args.emplace_back(&Param); 6048 const auto &FnInfo = 6049 CGM.getTypes().arrangeBuiltinFunctionDeclaration(C.VoidTy, Args); 6050 llvm::FunctionType *FnTy = CGM.getTypes().GetFunctionType(FnInfo); 6051 std::string Name = CGM.getOpenMPRuntime().getName({"red_fini", ""}); 6052 auto *Fn = llvm::Function::Create(FnTy, llvm::GlobalValue::InternalLinkage, 6053 Name, &CGM.getModule()); 6054 CGM.SetInternalFunctionAttributes(GlobalDecl(), Fn, FnInfo); 6055 Fn->setDoesNotRecurse(); 6056 CodeGenFunction CGF(CGM); 6057 CGF.StartFunction(GlobalDecl(), C.VoidTy, Fn, FnInfo, Args, Loc, Loc); 6058 Address PrivateAddr = CGF.EmitLoadOfPointer( 6059 CGF.GetAddrOfLocalVar(&Param), 6060 C.getPointerType(C.VoidPtrTy).castAs<PointerType>()); 6061 llvm::Value *Size = nullptr; 6062 // If the size of the reduction item is non-constant, load it from global 6063 // threadprivate variable. 6064 if (RCG.getSizes(N).second) { 6065 Address SizeAddr = CGM.getOpenMPRuntime().getAddrOfArtificialThreadPrivate( 6066 CGF, CGM.getContext().getSizeType(), 6067 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6068 Size = CGF.EmitLoadOfScalar(SizeAddr, /*Volatile=*/false, 6069 CGM.getContext().getSizeType(), Loc); 6070 } 6071 RCG.emitAggregateType(CGF, N, Size); 6072 // Emit the finalizer body: 6073 // <destroy>(<type>* %0) 6074 RCG.emitCleanups(CGF, N, PrivateAddr); 6075 CGF.FinishFunction(); 6076 return Fn; 6077 } 6078 6079 llvm::Value *CGOpenMPRuntime::emitTaskReductionInit( 6080 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 6081 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 6082 if (!CGF.HaveInsertPoint() || Data.ReductionVars.empty()) 6083 return nullptr; 6084 6085 // Build typedef struct: 6086 // kmp_task_red_input { 6087 // void *reduce_shar; // shared reduction item 6088 // size_t reduce_size; // size of data item 6089 // void *reduce_init; // data initialization routine 6090 // void *reduce_fini; // data finalization routine 6091 // void *reduce_comb; // data combiner routine 6092 // kmp_task_red_flags_t flags; // flags for additional info from compiler 6093 // } kmp_task_red_input_t; 6094 ASTContext &C = CGM.getContext(); 6095 RecordDecl *RD = C.buildImplicitRecord("kmp_task_red_input_t"); 6096 RD->startDefinition(); 6097 const FieldDecl *SharedFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6098 const FieldDecl *SizeFD = addFieldToRecordDecl(C, RD, C.getSizeType()); 6099 const FieldDecl *InitFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6100 const FieldDecl *FiniFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6101 const FieldDecl *CombFD = addFieldToRecordDecl(C, RD, C.VoidPtrTy); 6102 const FieldDecl *FlagsFD = addFieldToRecordDecl( 6103 C, RD, C.getIntTypeForBitwidth(/*DestWidth=*/32, /*Signed=*/false)); 6104 RD->completeDefinition(); 6105 QualType RDType = C.getRecordType(RD); 6106 unsigned Size = Data.ReductionVars.size(); 6107 llvm::APInt ArraySize(/*numBits=*/64, Size); 6108 QualType ArrayRDType = C.getConstantArrayType( 6109 RDType, ArraySize, ArrayType::Normal, /*IndexTypeQuals=*/0); 6110 // kmp_task_red_input_t .rd_input.[Size]; 6111 Address TaskRedInput = CGF.CreateMemTemp(ArrayRDType, ".rd_input."); 6112 ReductionCodeGen RCG(Data.ReductionVars, Data.ReductionCopies, 6113 Data.ReductionOps); 6114 for (unsigned Cnt = 0; Cnt < Size; ++Cnt) { 6115 // kmp_task_red_input_t &ElemLVal = .rd_input.[Cnt]; 6116 llvm::Value *Idxs[] = {llvm::ConstantInt::get(CGM.SizeTy, /*V=*/0), 6117 llvm::ConstantInt::get(CGM.SizeTy, Cnt)}; 6118 llvm::Value *GEP = CGF.EmitCheckedInBoundsGEP( 6119 TaskRedInput.getPointer(), Idxs, 6120 /*SignedIndices=*/false, /*IsSubtraction=*/false, Loc, 6121 ".rd_input.gep."); 6122 LValue ElemLVal = CGF.MakeNaturalAlignAddrLValue(GEP, RDType); 6123 // ElemLVal.reduce_shar = &Shareds[Cnt]; 6124 LValue SharedLVal = CGF.EmitLValueForField(ElemLVal, SharedFD); 6125 RCG.emitSharedLValue(CGF, Cnt); 6126 llvm::Value *CastedShared = 6127 CGF.EmitCastToVoidPtr(RCG.getSharedLValue(Cnt).getPointer()); 6128 CGF.EmitStoreOfScalar(CastedShared, SharedLVal); 6129 RCG.emitAggregateType(CGF, Cnt); 6130 llvm::Value *SizeValInChars; 6131 llvm::Value *SizeVal; 6132 std::tie(SizeValInChars, SizeVal) = RCG.getSizes(Cnt); 6133 // We use delayed creation/initialization for VLAs, array sections and 6134 // custom reduction initializations. It is required because runtime does not 6135 // provide the way to pass the sizes of VLAs/array sections to 6136 // initializer/combiner/finalizer functions and does not pass the pointer to 6137 // original reduction item to the initializer. Instead threadprivate global 6138 // variables are used to store these values and use them in the functions. 6139 bool DelayedCreation = !!SizeVal; 6140 SizeValInChars = CGF.Builder.CreateIntCast(SizeValInChars, CGM.SizeTy, 6141 /*isSigned=*/false); 6142 LValue SizeLVal = CGF.EmitLValueForField(ElemLVal, SizeFD); 6143 CGF.EmitStoreOfScalar(SizeValInChars, SizeLVal); 6144 // ElemLVal.reduce_init = init; 6145 LValue InitLVal = CGF.EmitLValueForField(ElemLVal, InitFD); 6146 llvm::Value *InitAddr = 6147 CGF.EmitCastToVoidPtr(emitReduceInitFunction(CGM, Loc, RCG, Cnt)); 6148 CGF.EmitStoreOfScalar(InitAddr, InitLVal); 6149 DelayedCreation = DelayedCreation || RCG.usesReductionInitializer(Cnt); 6150 // ElemLVal.reduce_fini = fini; 6151 LValue FiniLVal = CGF.EmitLValueForField(ElemLVal, FiniFD); 6152 llvm::Value *Fini = emitReduceFiniFunction(CGM, Loc, RCG, Cnt); 6153 llvm::Value *FiniAddr = Fini 6154 ? CGF.EmitCastToVoidPtr(Fini) 6155 : llvm::ConstantPointerNull::get(CGM.VoidPtrTy); 6156 CGF.EmitStoreOfScalar(FiniAddr, FiniLVal); 6157 // ElemLVal.reduce_comb = comb; 6158 LValue CombLVal = CGF.EmitLValueForField(ElemLVal, CombFD); 6159 llvm::Value *CombAddr = CGF.EmitCastToVoidPtr(emitReduceCombFunction( 6160 CGM, Loc, RCG, Cnt, Data.ReductionOps[Cnt], LHSExprs[Cnt], 6161 RHSExprs[Cnt], Data.ReductionCopies[Cnt])); 6162 CGF.EmitStoreOfScalar(CombAddr, CombLVal); 6163 // ElemLVal.flags = 0; 6164 LValue FlagsLVal = CGF.EmitLValueForField(ElemLVal, FlagsFD); 6165 if (DelayedCreation) { 6166 CGF.EmitStoreOfScalar( 6167 llvm::ConstantInt::get(CGM.Int32Ty, /*V=*/1, /*IsSigned=*/true), 6168 FlagsLVal); 6169 } else 6170 CGF.EmitNullInitialization(FlagsLVal.getAddress(), FlagsLVal.getType()); 6171 } 6172 // Build call void *__kmpc_task_reduction_init(int gtid, int num_data, void 6173 // *data); 6174 llvm::Value *Args[] = { 6175 CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy, 6176 /*isSigned=*/true), 6177 llvm::ConstantInt::get(CGM.IntTy, Size, /*isSigned=*/true), 6178 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(TaskRedInput.getPointer(), 6179 CGM.VoidPtrTy)}; 6180 return CGF.EmitRuntimeCall( 6181 createRuntimeFunction(OMPRTL__kmpc_task_reduction_init), Args); 6182 } 6183 6184 void CGOpenMPRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 6185 SourceLocation Loc, 6186 ReductionCodeGen &RCG, 6187 unsigned N) { 6188 auto Sizes = RCG.getSizes(N); 6189 // Emit threadprivate global variable if the type is non-constant 6190 // (Sizes.second = nullptr). 6191 if (Sizes.second) { 6192 llvm::Value *SizeVal = CGF.Builder.CreateIntCast(Sizes.second, CGM.SizeTy, 6193 /*isSigned=*/false); 6194 Address SizeAddr = getAddrOfArtificialThreadPrivate( 6195 CGF, CGM.getContext().getSizeType(), 6196 generateUniqueName(CGM, "reduction_size", RCG.getRefExpr(N))); 6197 CGF.Builder.CreateStore(SizeVal, SizeAddr, /*IsVolatile=*/false); 6198 } 6199 // Store address of the original reduction item if custom initializer is used. 6200 if (RCG.usesReductionInitializer(N)) { 6201 Address SharedAddr = getAddrOfArtificialThreadPrivate( 6202 CGF, CGM.getContext().VoidPtrTy, 6203 generateUniqueName(CGM, "reduction", RCG.getRefExpr(N))); 6204 CGF.Builder.CreateStore( 6205 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 6206 RCG.getSharedLValue(N).getPointer(), CGM.VoidPtrTy), 6207 SharedAddr, /*IsVolatile=*/false); 6208 } 6209 } 6210 6211 Address CGOpenMPRuntime::getTaskReductionItem(CodeGenFunction &CGF, 6212 SourceLocation Loc, 6213 llvm::Value *ReductionsPtr, 6214 LValue SharedLVal) { 6215 // Build call void *__kmpc_task_reduction_get_th_data(int gtid, void *tg, void 6216 // *d); 6217 llvm::Value *Args[] = { 6218 CGF.Builder.CreateIntCast(getThreadID(CGF, Loc), CGM.IntTy, 6219 /*isSigned=*/true), 6220 ReductionsPtr, 6221 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(SharedLVal.getPointer(), 6222 CGM.VoidPtrTy)}; 6223 return Address( 6224 CGF.EmitRuntimeCall( 6225 createRuntimeFunction(OMPRTL__kmpc_task_reduction_get_th_data), Args), 6226 SharedLVal.getAlignment()); 6227 } 6228 6229 void CGOpenMPRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 6230 SourceLocation Loc) { 6231 if (!CGF.HaveInsertPoint()) 6232 return; 6233 // Build call kmp_int32 __kmpc_omp_taskwait(ident_t *loc, kmp_int32 6234 // global_tid); 6235 llvm::Value *Args[] = {emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc)}; 6236 // Ignore return result until untied tasks are supported. 6237 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_omp_taskwait), Args); 6238 if (auto *Region = dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) 6239 Region->emitUntiedSwitch(CGF); 6240 } 6241 6242 void CGOpenMPRuntime::emitInlinedDirective(CodeGenFunction &CGF, 6243 OpenMPDirectiveKind InnerKind, 6244 const RegionCodeGenTy &CodeGen, 6245 bool HasCancel) { 6246 if (!CGF.HaveInsertPoint()) 6247 return; 6248 InlinedOpenMPRegionRAII Region(CGF, CodeGen, InnerKind, HasCancel); 6249 CGF.CapturedStmtInfo->EmitBody(CGF, /*S=*/nullptr); 6250 } 6251 6252 namespace { 6253 enum RTCancelKind { 6254 CancelNoreq = 0, 6255 CancelParallel = 1, 6256 CancelLoop = 2, 6257 CancelSections = 3, 6258 CancelTaskgroup = 4 6259 }; 6260 } // anonymous namespace 6261 6262 static RTCancelKind getCancellationKind(OpenMPDirectiveKind CancelRegion) { 6263 RTCancelKind CancelKind = CancelNoreq; 6264 if (CancelRegion == OMPD_parallel) 6265 CancelKind = CancelParallel; 6266 else if (CancelRegion == OMPD_for) 6267 CancelKind = CancelLoop; 6268 else if (CancelRegion == OMPD_sections) 6269 CancelKind = CancelSections; 6270 else { 6271 assert(CancelRegion == OMPD_taskgroup); 6272 CancelKind = CancelTaskgroup; 6273 } 6274 return CancelKind; 6275 } 6276 6277 void CGOpenMPRuntime::emitCancellationPointCall( 6278 CodeGenFunction &CGF, SourceLocation Loc, 6279 OpenMPDirectiveKind CancelRegion) { 6280 if (!CGF.HaveInsertPoint()) 6281 return; 6282 // Build call kmp_int32 __kmpc_cancellationpoint(ident_t *loc, kmp_int32 6283 // global_tid, kmp_int32 cncl_kind); 6284 if (auto *OMPRegionInfo = 6285 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6286 // For 'cancellation point taskgroup', the task region info may not have a 6287 // cancel. This may instead happen in another adjacent task. 6288 if (CancelRegion == OMPD_taskgroup || OMPRegionInfo->hasCancel()) { 6289 llvm::Value *Args[] = { 6290 emitUpdateLocation(CGF, Loc), getThreadID(CGF, Loc), 6291 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6292 // Ignore return result until untied tasks are supported. 6293 llvm::Value *Result = CGF.EmitRuntimeCall( 6294 createRuntimeFunction(OMPRTL__kmpc_cancellationpoint), Args); 6295 // if (__kmpc_cancellationpoint()) { 6296 // exit from construct; 6297 // } 6298 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6299 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6300 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6301 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6302 CGF.EmitBlock(ExitBB); 6303 // exit from construct; 6304 CodeGenFunction::JumpDest CancelDest = 6305 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6306 CGF.EmitBranchThroughCleanup(CancelDest); 6307 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6308 } 6309 } 6310 } 6311 6312 void CGOpenMPRuntime::emitCancelCall(CodeGenFunction &CGF, SourceLocation Loc, 6313 const Expr *IfCond, 6314 OpenMPDirectiveKind CancelRegion) { 6315 if (!CGF.HaveInsertPoint()) 6316 return; 6317 // Build call kmp_int32 __kmpc_cancel(ident_t *loc, kmp_int32 global_tid, 6318 // kmp_int32 cncl_kind); 6319 if (auto *OMPRegionInfo = 6320 dyn_cast_or_null<CGOpenMPRegionInfo>(CGF.CapturedStmtInfo)) { 6321 auto &&ThenGen = [Loc, CancelRegion, OMPRegionInfo](CodeGenFunction &CGF, 6322 PrePostActionTy &) { 6323 CGOpenMPRuntime &RT = CGF.CGM.getOpenMPRuntime(); 6324 llvm::Value *Args[] = { 6325 RT.emitUpdateLocation(CGF, Loc), RT.getThreadID(CGF, Loc), 6326 CGF.Builder.getInt32(getCancellationKind(CancelRegion))}; 6327 // Ignore return result until untied tasks are supported. 6328 llvm::Value *Result = CGF.EmitRuntimeCall( 6329 RT.createRuntimeFunction(OMPRTL__kmpc_cancel), Args); 6330 // if (__kmpc_cancel()) { 6331 // exit from construct; 6332 // } 6333 llvm::BasicBlock *ExitBB = CGF.createBasicBlock(".cancel.exit"); 6334 llvm::BasicBlock *ContBB = CGF.createBasicBlock(".cancel.continue"); 6335 llvm::Value *Cmp = CGF.Builder.CreateIsNotNull(Result); 6336 CGF.Builder.CreateCondBr(Cmp, ExitBB, ContBB); 6337 CGF.EmitBlock(ExitBB); 6338 // exit from construct; 6339 CodeGenFunction::JumpDest CancelDest = 6340 CGF.getOMPCancelDestination(OMPRegionInfo->getDirectiveKind()); 6341 CGF.EmitBranchThroughCleanup(CancelDest); 6342 CGF.EmitBlock(ContBB, /*IsFinished=*/true); 6343 }; 6344 if (IfCond) { 6345 emitOMPIfClause(CGF, IfCond, ThenGen, 6346 [](CodeGenFunction &, PrePostActionTy &) {}); 6347 } else { 6348 RegionCodeGenTy ThenRCG(ThenGen); 6349 ThenRCG(CGF); 6350 } 6351 } 6352 } 6353 6354 void CGOpenMPRuntime::emitTargetOutlinedFunction( 6355 const OMPExecutableDirective &D, StringRef ParentName, 6356 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6357 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6358 assert(!ParentName.empty() && "Invalid target region parent name!"); 6359 emitTargetOutlinedFunctionHelper(D, ParentName, OutlinedFn, OutlinedFnID, 6360 IsOffloadEntry, CodeGen); 6361 } 6362 6363 void CGOpenMPRuntime::emitTargetOutlinedFunctionHelper( 6364 const OMPExecutableDirective &D, StringRef ParentName, 6365 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 6366 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 6367 // Create a unique name for the entry function using the source location 6368 // information of the current target region. The name will be something like: 6369 // 6370 // __omp_offloading_DD_FFFF_PP_lBB 6371 // 6372 // where DD_FFFF is an ID unique to the file (device and file IDs), PP is the 6373 // mangled name of the function that encloses the target region and BB is the 6374 // line number of the target region. 6375 6376 unsigned DeviceID; 6377 unsigned FileID; 6378 unsigned Line; 6379 getTargetEntryUniqueInfo(CGM.getContext(), D.getBeginLoc(), DeviceID, FileID, 6380 Line); 6381 SmallString<64> EntryFnName; 6382 { 6383 llvm::raw_svector_ostream OS(EntryFnName); 6384 OS << "__omp_offloading" << llvm::format("_%x", DeviceID) 6385 << llvm::format("_%x_", FileID) << ParentName << "_l" << Line; 6386 } 6387 6388 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 6389 6390 CodeGenFunction CGF(CGM, true); 6391 CGOpenMPTargetRegionInfo CGInfo(CS, CodeGen, EntryFnName); 6392 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6393 6394 OutlinedFn = CGF.GenerateOpenMPCapturedStmtFunction(CS); 6395 6396 // If this target outline function is not an offload entry, we don't need to 6397 // register it. 6398 if (!IsOffloadEntry) 6399 return; 6400 6401 // The target region ID is used by the runtime library to identify the current 6402 // target region, so it only has to be unique and not necessarily point to 6403 // anything. It could be the pointer to the outlined function that implements 6404 // the target region, but we aren't using that so that the compiler doesn't 6405 // need to keep that, and could therefore inline the host function if proven 6406 // worthwhile during optimization. In the other hand, if emitting code for the 6407 // device, the ID has to be the function address so that it can retrieved from 6408 // the offloading entry and launched by the runtime library. We also mark the 6409 // outlined function to have external linkage in case we are emitting code for 6410 // the device, because these functions will be entry points to the device. 6411 6412 if (CGM.getLangOpts().OpenMPIsDevice) { 6413 OutlinedFnID = llvm::ConstantExpr::getBitCast(OutlinedFn, CGM.Int8PtrTy); 6414 OutlinedFn->setLinkage(llvm::GlobalValue::WeakAnyLinkage); 6415 OutlinedFn->setDSOLocal(false); 6416 } else { 6417 std::string Name = getName({EntryFnName, "region_id"}); 6418 OutlinedFnID = new llvm::GlobalVariable( 6419 CGM.getModule(), CGM.Int8Ty, /*isConstant=*/true, 6420 llvm::GlobalValue::WeakAnyLinkage, 6421 llvm::Constant::getNullValue(CGM.Int8Ty), Name); 6422 } 6423 6424 // Register the information for the entry associated with this target region. 6425 OffloadEntriesInfoManager.registerTargetRegionEntryInfo( 6426 DeviceID, FileID, ParentName, Line, OutlinedFn, OutlinedFnID, 6427 OffloadEntriesInfoManagerTy::OMPTargetRegionEntryTargetRegion); 6428 } 6429 6430 /// discard all CompoundStmts intervening between two constructs 6431 static const Stmt *ignoreCompoundStmts(const Stmt *Body) { 6432 while (const auto *CS = dyn_cast_or_null<CompoundStmt>(Body)) 6433 Body = CS->body_front(); 6434 6435 return Body; 6436 } 6437 6438 /// Emit the number of teams for a target directive. Inspect the num_teams 6439 /// clause associated with a teams construct combined or closely nested 6440 /// with the target directive. 6441 /// 6442 /// Emit a team of size one for directives such as 'target parallel' that 6443 /// have no associated teams construct. 6444 /// 6445 /// Otherwise, return nullptr. 6446 static llvm::Value * 6447 emitNumTeamsForTargetDirective(CGOpenMPRuntime &OMPRuntime, 6448 CodeGenFunction &CGF, 6449 const OMPExecutableDirective &D) { 6450 assert(!CGF.getLangOpts().OpenMPIsDevice && "Clauses associated with the " 6451 "teams directive expected to be " 6452 "emitted only for the host!"); 6453 6454 CGBuilderTy &Bld = CGF.Builder; 6455 6456 // If the target directive is combined with a teams directive: 6457 // Return the value in the num_teams clause, if any. 6458 // Otherwise, return 0 to denote the runtime default. 6459 if (isOpenMPTeamsDirective(D.getDirectiveKind())) { 6460 if (const auto *NumTeamsClause = D.getSingleClause<OMPNumTeamsClause>()) { 6461 CodeGenFunction::RunCleanupsScope NumTeamsScope(CGF); 6462 llvm::Value *NumTeams = CGF.EmitScalarExpr(NumTeamsClause->getNumTeams(), 6463 /*IgnoreResultAssign*/ true); 6464 return Bld.CreateIntCast(NumTeams, CGF.Int32Ty, 6465 /*IsSigned=*/true); 6466 } 6467 6468 // The default value is 0. 6469 return Bld.getInt32(0); 6470 } 6471 6472 // If the target directive is combined with a parallel directive but not a 6473 // teams directive, start one team. 6474 if (isOpenMPParallelDirective(D.getDirectiveKind())) 6475 return Bld.getInt32(1); 6476 6477 // If the current target region has a teams region enclosed, we need to get 6478 // the number of teams to pass to the runtime function call. This is done 6479 // by generating the expression in a inlined region. This is required because 6480 // the expression is captured in the enclosing target environment when the 6481 // teams directive is not combined with target. 6482 6483 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 6484 6485 if (const auto *TeamsDir = dyn_cast_or_null<OMPExecutableDirective>( 6486 ignoreCompoundStmts(CS.getCapturedStmt()))) { 6487 if (isOpenMPTeamsDirective(TeamsDir->getDirectiveKind())) { 6488 if (const auto *NTE = TeamsDir->getSingleClause<OMPNumTeamsClause>()) { 6489 CGOpenMPInnerExprInfo CGInfo(CGF, CS); 6490 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6491 llvm::Value *NumTeams = CGF.EmitScalarExpr(NTE->getNumTeams()); 6492 return Bld.CreateIntCast(NumTeams, CGF.Int32Ty, 6493 /*IsSigned=*/true); 6494 } 6495 6496 // If we have an enclosed teams directive but no num_teams clause we use 6497 // the default value 0. 6498 return Bld.getInt32(0); 6499 } 6500 } 6501 6502 // No teams associated with the directive. 6503 return nullptr; 6504 } 6505 6506 /// Emit the number of threads for a target directive. Inspect the 6507 /// thread_limit clause associated with a teams construct combined or closely 6508 /// nested with the target directive. 6509 /// 6510 /// Emit the num_threads clause for directives such as 'target parallel' that 6511 /// have no associated teams construct. 6512 /// 6513 /// Otherwise, return nullptr. 6514 static llvm::Value * 6515 emitNumThreadsForTargetDirective(CGOpenMPRuntime &OMPRuntime, 6516 CodeGenFunction &CGF, 6517 const OMPExecutableDirective &D) { 6518 assert(!CGF.getLangOpts().OpenMPIsDevice && "Clauses associated with the " 6519 "teams directive expected to be " 6520 "emitted only for the host!"); 6521 6522 CGBuilderTy &Bld = CGF.Builder; 6523 6524 // 6525 // If the target directive is combined with a teams directive: 6526 // Return the value in the thread_limit clause, if any. 6527 // 6528 // If the target directive is combined with a parallel directive: 6529 // Return the value in the num_threads clause, if any. 6530 // 6531 // If both clauses are set, select the minimum of the two. 6532 // 6533 // If neither teams or parallel combined directives set the number of threads 6534 // in a team, return 0 to denote the runtime default. 6535 // 6536 // If this is not a teams directive return nullptr. 6537 6538 if (isOpenMPTeamsDirective(D.getDirectiveKind()) || 6539 isOpenMPParallelDirective(D.getDirectiveKind())) { 6540 llvm::Value *DefaultThreadLimitVal = Bld.getInt32(0); 6541 llvm::Value *NumThreadsVal = nullptr; 6542 llvm::Value *ThreadLimitVal = nullptr; 6543 6544 if (const auto *ThreadLimitClause = 6545 D.getSingleClause<OMPThreadLimitClause>()) { 6546 CodeGenFunction::RunCleanupsScope ThreadLimitScope(CGF); 6547 llvm::Value *ThreadLimit = 6548 CGF.EmitScalarExpr(ThreadLimitClause->getThreadLimit(), 6549 /*IgnoreResultAssign*/ true); 6550 ThreadLimitVal = Bld.CreateIntCast(ThreadLimit, CGF.Int32Ty, 6551 /*IsSigned=*/true); 6552 } 6553 6554 if (const auto *NumThreadsClause = 6555 D.getSingleClause<OMPNumThreadsClause>()) { 6556 CodeGenFunction::RunCleanupsScope NumThreadsScope(CGF); 6557 llvm::Value *NumThreads = 6558 CGF.EmitScalarExpr(NumThreadsClause->getNumThreads(), 6559 /*IgnoreResultAssign*/ true); 6560 NumThreadsVal = 6561 Bld.CreateIntCast(NumThreads, CGF.Int32Ty, /*IsSigned=*/true); 6562 } 6563 6564 // Select the lesser of thread_limit and num_threads. 6565 if (NumThreadsVal) 6566 ThreadLimitVal = ThreadLimitVal 6567 ? Bld.CreateSelect(Bld.CreateICmpSLT(NumThreadsVal, 6568 ThreadLimitVal), 6569 NumThreadsVal, ThreadLimitVal) 6570 : NumThreadsVal; 6571 6572 // Set default value passed to the runtime if either teams or a target 6573 // parallel type directive is found but no clause is specified. 6574 if (!ThreadLimitVal) 6575 ThreadLimitVal = DefaultThreadLimitVal; 6576 6577 return ThreadLimitVal; 6578 } 6579 6580 // If the current target region has a teams region enclosed, we need to get 6581 // the thread limit to pass to the runtime function call. This is done 6582 // by generating the expression in a inlined region. This is required because 6583 // the expression is captured in the enclosing target environment when the 6584 // teams directive is not combined with target. 6585 6586 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 6587 6588 if (const auto *TeamsDir = dyn_cast_or_null<OMPExecutableDirective>( 6589 ignoreCompoundStmts(CS.getCapturedStmt()))) { 6590 if (isOpenMPTeamsDirective(TeamsDir->getDirectiveKind())) { 6591 if (const auto *TLE = TeamsDir->getSingleClause<OMPThreadLimitClause>()) { 6592 CGOpenMPInnerExprInfo CGInfo(CGF, CS); 6593 CodeGenFunction::CGCapturedStmtRAII CapInfoRAII(CGF, &CGInfo); 6594 llvm::Value *ThreadLimit = CGF.EmitScalarExpr(TLE->getThreadLimit()); 6595 return CGF.Builder.CreateIntCast(ThreadLimit, CGF.Int32Ty, 6596 /*IsSigned=*/true); 6597 } 6598 6599 // If we have an enclosed teams directive but no thread_limit clause we 6600 // use the default value 0. 6601 return CGF.Builder.getInt32(0); 6602 } 6603 } 6604 6605 // No teams associated with the directive. 6606 return nullptr; 6607 } 6608 6609 namespace { 6610 LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE(); 6611 6612 // Utility to handle information from clauses associated with a given 6613 // construct that use mappable expressions (e.g. 'map' clause, 'to' clause). 6614 // It provides a convenient interface to obtain the information and generate 6615 // code for that information. 6616 class MappableExprsHandler { 6617 public: 6618 /// Values for bit flags used to specify the mapping type for 6619 /// offloading. 6620 enum OpenMPOffloadMappingFlags : uint64_t { 6621 /// No flags 6622 OMP_MAP_NONE = 0x0, 6623 /// Allocate memory on the device and move data from host to device. 6624 OMP_MAP_TO = 0x01, 6625 /// Allocate memory on the device and move data from device to host. 6626 OMP_MAP_FROM = 0x02, 6627 /// Always perform the requested mapping action on the element, even 6628 /// if it was already mapped before. 6629 OMP_MAP_ALWAYS = 0x04, 6630 /// Delete the element from the device environment, ignoring the 6631 /// current reference count associated with the element. 6632 OMP_MAP_DELETE = 0x08, 6633 /// The element being mapped is a pointer-pointee pair; both the 6634 /// pointer and the pointee should be mapped. 6635 OMP_MAP_PTR_AND_OBJ = 0x10, 6636 /// This flags signals that the base address of an entry should be 6637 /// passed to the target kernel as an argument. 6638 OMP_MAP_TARGET_PARAM = 0x20, 6639 /// Signal that the runtime library has to return the device pointer 6640 /// in the current position for the data being mapped. Used when we have the 6641 /// use_device_ptr clause. 6642 OMP_MAP_RETURN_PARAM = 0x40, 6643 /// This flag signals that the reference being passed is a pointer to 6644 /// private data. 6645 OMP_MAP_PRIVATE = 0x80, 6646 /// Pass the element to the device by value. 6647 OMP_MAP_LITERAL = 0x100, 6648 /// Implicit map 6649 OMP_MAP_IMPLICIT = 0x200, 6650 /// The 16 MSBs of the flags indicate whether the entry is member of some 6651 /// struct/class. 6652 OMP_MAP_MEMBER_OF = 0xffff000000000000, 6653 LLVM_MARK_AS_BITMASK_ENUM(/* LargestFlag = */ OMP_MAP_MEMBER_OF), 6654 }; 6655 6656 /// Class that associates information with a base pointer to be passed to the 6657 /// runtime library. 6658 class BasePointerInfo { 6659 /// The base pointer. 6660 llvm::Value *Ptr = nullptr; 6661 /// The base declaration that refers to this device pointer, or null if 6662 /// there is none. 6663 const ValueDecl *DevPtrDecl = nullptr; 6664 6665 public: 6666 BasePointerInfo(llvm::Value *Ptr, const ValueDecl *DevPtrDecl = nullptr) 6667 : Ptr(Ptr), DevPtrDecl(DevPtrDecl) {} 6668 llvm::Value *operator*() const { return Ptr; } 6669 const ValueDecl *getDevicePtrDecl() const { return DevPtrDecl; } 6670 void setDevicePtrDecl(const ValueDecl *D) { DevPtrDecl = D; } 6671 }; 6672 6673 using MapBaseValuesArrayTy = SmallVector<BasePointerInfo, 4>; 6674 using MapValuesArrayTy = SmallVector<llvm::Value *, 4>; 6675 using MapFlagsArrayTy = SmallVector<OpenMPOffloadMappingFlags, 4>; 6676 6677 /// Map between a struct and the its lowest & highest elements which have been 6678 /// mapped. 6679 /// [ValueDecl *] --> {LE(FieldIndex, Pointer), 6680 /// HE(FieldIndex, Pointer)} 6681 struct StructRangeInfoTy { 6682 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> LowestElem = { 6683 0, Address::invalid()}; 6684 std::pair<unsigned /*FieldIndex*/, Address /*Pointer*/> HighestElem = { 6685 0, Address::invalid()}; 6686 Address Base = Address::invalid(); 6687 }; 6688 6689 private: 6690 /// Kind that defines how a device pointer has to be returned. 6691 struct MapInfo { 6692 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 6693 OpenMPMapClauseKind MapType = OMPC_MAP_unknown; 6694 ArrayRef<OpenMPMapModifierKind> MapModifiers; 6695 bool ReturnDevicePointer = false; 6696 bool IsImplicit = false; 6697 6698 MapInfo() = default; 6699 MapInfo( 6700 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 6701 OpenMPMapClauseKind MapType, 6702 ArrayRef<OpenMPMapModifierKind> MapModifiers, 6703 bool ReturnDevicePointer, bool IsImplicit) 6704 : Components(Components), MapType(MapType), MapModifiers(MapModifiers), 6705 ReturnDevicePointer(ReturnDevicePointer), IsImplicit(IsImplicit) {} 6706 }; 6707 6708 /// If use_device_ptr is used on a pointer which is a struct member and there 6709 /// is no map information about it, then emission of that entry is deferred 6710 /// until the whole struct has been processed. 6711 struct DeferredDevicePtrEntryTy { 6712 const Expr *IE = nullptr; 6713 const ValueDecl *VD = nullptr; 6714 6715 DeferredDevicePtrEntryTy(const Expr *IE, const ValueDecl *VD) 6716 : IE(IE), VD(VD) {} 6717 }; 6718 6719 /// Directive from where the map clauses were extracted. 6720 const OMPExecutableDirective &CurDir; 6721 6722 /// Function the directive is being generated for. 6723 CodeGenFunction &CGF; 6724 6725 /// Set of all first private variables in the current directive. 6726 llvm::SmallPtrSet<const VarDecl *, 8> FirstPrivateDecls; 6727 6728 /// Map between device pointer declarations and their expression components. 6729 /// The key value for declarations in 'this' is null. 6730 llvm::DenseMap< 6731 const ValueDecl *, 6732 SmallVector<OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>> 6733 DevPointersMap; 6734 6735 llvm::Value *getExprTypeSize(const Expr *E) const { 6736 QualType ExprTy = E->getType().getCanonicalType(); 6737 6738 // Reference types are ignored for mapping purposes. 6739 if (const auto *RefTy = ExprTy->getAs<ReferenceType>()) 6740 ExprTy = RefTy->getPointeeType().getCanonicalType(); 6741 6742 // Given that an array section is considered a built-in type, we need to 6743 // do the calculation based on the length of the section instead of relying 6744 // on CGF.getTypeSize(E->getType()). 6745 if (const auto *OAE = dyn_cast<OMPArraySectionExpr>(E)) { 6746 QualType BaseTy = OMPArraySectionExpr::getBaseOriginalType( 6747 OAE->getBase()->IgnoreParenImpCasts()) 6748 .getCanonicalType(); 6749 6750 // If there is no length associated with the expression, that means we 6751 // are using the whole length of the base. 6752 if (!OAE->getLength() && OAE->getColonLoc().isValid()) 6753 return CGF.getTypeSize(BaseTy); 6754 6755 llvm::Value *ElemSize; 6756 if (const auto *PTy = BaseTy->getAs<PointerType>()) { 6757 ElemSize = CGF.getTypeSize(PTy->getPointeeType().getCanonicalType()); 6758 } else { 6759 const auto *ATy = cast<ArrayType>(BaseTy.getTypePtr()); 6760 assert(ATy && "Expecting array type if not a pointer type."); 6761 ElemSize = CGF.getTypeSize(ATy->getElementType().getCanonicalType()); 6762 } 6763 6764 // If we don't have a length at this point, that is because we have an 6765 // array section with a single element. 6766 if (!OAE->getLength()) 6767 return ElemSize; 6768 6769 llvm::Value *LengthVal = CGF.EmitScalarExpr(OAE->getLength()); 6770 LengthVal = 6771 CGF.Builder.CreateIntCast(LengthVal, CGF.SizeTy, /*isSigned=*/false); 6772 return CGF.Builder.CreateNUWMul(LengthVal, ElemSize); 6773 } 6774 return CGF.getTypeSize(ExprTy); 6775 } 6776 6777 /// Return the corresponding bits for a given map clause modifier. Add 6778 /// a flag marking the map as a pointer if requested. Add a flag marking the 6779 /// map as the first one of a series of maps that relate to the same map 6780 /// expression. 6781 OpenMPOffloadMappingFlags getMapTypeBits( 6782 OpenMPMapClauseKind MapType, ArrayRef<OpenMPMapModifierKind> MapModifiers, 6783 bool IsImplicit, bool AddPtrFlag, bool AddIsTargetParamFlag) const { 6784 OpenMPOffloadMappingFlags Bits = 6785 IsImplicit ? OMP_MAP_IMPLICIT : OMP_MAP_NONE; 6786 switch (MapType) { 6787 case OMPC_MAP_alloc: 6788 case OMPC_MAP_release: 6789 // alloc and release is the default behavior in the runtime library, i.e. 6790 // if we don't pass any bits alloc/release that is what the runtime is 6791 // going to do. Therefore, we don't need to signal anything for these two 6792 // type modifiers. 6793 break; 6794 case OMPC_MAP_to: 6795 Bits |= OMP_MAP_TO; 6796 break; 6797 case OMPC_MAP_from: 6798 Bits |= OMP_MAP_FROM; 6799 break; 6800 case OMPC_MAP_tofrom: 6801 Bits |= OMP_MAP_TO | OMP_MAP_FROM; 6802 break; 6803 case OMPC_MAP_delete: 6804 Bits |= OMP_MAP_DELETE; 6805 break; 6806 case OMPC_MAP_unknown: 6807 llvm_unreachable("Unexpected map type!"); 6808 } 6809 if (AddPtrFlag) 6810 Bits |= OMP_MAP_PTR_AND_OBJ; 6811 if (AddIsTargetParamFlag) 6812 Bits |= OMP_MAP_TARGET_PARAM; 6813 if (llvm::find(MapModifiers, OMPC_MAP_MODIFIER_always) 6814 != MapModifiers.end()) 6815 Bits |= OMP_MAP_ALWAYS; 6816 return Bits; 6817 } 6818 6819 /// Return true if the provided expression is a final array section. A 6820 /// final array section, is one whose length can't be proved to be one. 6821 bool isFinalArraySectionExpression(const Expr *E) const { 6822 const auto *OASE = dyn_cast<OMPArraySectionExpr>(E); 6823 6824 // It is not an array section and therefore not a unity-size one. 6825 if (!OASE) 6826 return false; 6827 6828 // An array section with no colon always refer to a single element. 6829 if (OASE->getColonLoc().isInvalid()) 6830 return false; 6831 6832 const Expr *Length = OASE->getLength(); 6833 6834 // If we don't have a length we have to check if the array has size 1 6835 // for this dimension. Also, we should always expect a length if the 6836 // base type is pointer. 6837 if (!Length) { 6838 QualType BaseQTy = OMPArraySectionExpr::getBaseOriginalType( 6839 OASE->getBase()->IgnoreParenImpCasts()) 6840 .getCanonicalType(); 6841 if (const auto *ATy = dyn_cast<ConstantArrayType>(BaseQTy.getTypePtr())) 6842 return ATy->getSize().getSExtValue() != 1; 6843 // If we don't have a constant dimension length, we have to consider 6844 // the current section as having any size, so it is not necessarily 6845 // unitary. If it happen to be unity size, that's user fault. 6846 return true; 6847 } 6848 6849 // Check if the length evaluates to 1. 6850 Expr::EvalResult Result; 6851 if (!Length->EvaluateAsInt(Result, CGF.getContext())) 6852 return true; // Can have more that size 1. 6853 6854 llvm::APSInt ConstLength = Result.Val.getInt(); 6855 return ConstLength.getSExtValue() != 1; 6856 } 6857 6858 /// Generate the base pointers, section pointers, sizes and map type 6859 /// bits for the provided map type, map modifier, and expression components. 6860 /// \a IsFirstComponent should be set to true if the provided set of 6861 /// components is the first associated with a capture. 6862 void generateInfoForComponentList( 6863 OpenMPMapClauseKind MapType, 6864 ArrayRef<OpenMPMapModifierKind> MapModifiers, 6865 OMPClauseMappableExprCommon::MappableExprComponentListRef Components, 6866 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 6867 MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types, 6868 StructRangeInfoTy &PartialStruct, bool IsFirstComponentList, 6869 bool IsImplicit, 6870 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 6871 OverlappedElements = llvm::None) const { 6872 // The following summarizes what has to be generated for each map and the 6873 // types below. The generated information is expressed in this order: 6874 // base pointer, section pointer, size, flags 6875 // (to add to the ones that come from the map type and modifier). 6876 // 6877 // double d; 6878 // int i[100]; 6879 // float *p; 6880 // 6881 // struct S1 { 6882 // int i; 6883 // float f[50]; 6884 // } 6885 // struct S2 { 6886 // int i; 6887 // float f[50]; 6888 // S1 s; 6889 // double *p; 6890 // struct S2 *ps; 6891 // } 6892 // S2 s; 6893 // S2 *ps; 6894 // 6895 // map(d) 6896 // &d, &d, sizeof(double), TARGET_PARAM | TO | FROM 6897 // 6898 // map(i) 6899 // &i, &i, 100*sizeof(int), TARGET_PARAM | TO | FROM 6900 // 6901 // map(i[1:23]) 6902 // &i(=&i[0]), &i[1], 23*sizeof(int), TARGET_PARAM | TO | FROM 6903 // 6904 // map(p) 6905 // &p, &p, sizeof(float*), TARGET_PARAM | TO | FROM 6906 // 6907 // map(p[1:24]) 6908 // p, &p[1], 24*sizeof(float), TARGET_PARAM | TO | FROM 6909 // 6910 // map(s) 6911 // &s, &s, sizeof(S2), TARGET_PARAM | TO | FROM 6912 // 6913 // map(s.i) 6914 // &s, &(s.i), sizeof(int), TARGET_PARAM | TO | FROM 6915 // 6916 // map(s.s.f) 6917 // &s, &(s.s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 6918 // 6919 // map(s.p) 6920 // &s, &(s.p), sizeof(double*), TARGET_PARAM | TO | FROM 6921 // 6922 // map(to: s.p[:22]) 6923 // &s, &(s.p), sizeof(double*), TARGET_PARAM (*) 6924 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) (**) 6925 // &(s.p), &(s.p[0]), 22*sizeof(double), 6926 // MEMBER_OF(1) | PTR_AND_OBJ | TO (***) 6927 // (*) alloc space for struct members, only this is a target parameter 6928 // (**) map the pointer (nothing to be mapped in this example) (the compiler 6929 // optimizes this entry out, same in the examples below) 6930 // (***) map the pointee (map: to) 6931 // 6932 // map(s.ps) 6933 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM | TO | FROM 6934 // 6935 // map(from: s.ps->s.i) 6936 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 6937 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 6938 // &(s.ps), &(s.ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 6939 // 6940 // map(to: s.ps->ps) 6941 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 6942 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 6943 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | TO 6944 // 6945 // map(s.ps->ps->ps) 6946 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 6947 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 6948 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 6949 // &(s.ps->ps), &(s.ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 6950 // 6951 // map(to: s.ps->ps->s.f[:22]) 6952 // &s, &(s.ps), sizeof(S2*), TARGET_PARAM 6953 // &s, &(s.ps), sizeof(S2*), MEMBER_OF(1) 6954 // &(s.ps), &(s.ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 6955 // &(s.ps->ps), &(s.ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 6956 // 6957 // map(ps) 6958 // &ps, &ps, sizeof(S2*), TARGET_PARAM | TO | FROM 6959 // 6960 // map(ps->i) 6961 // ps, &(ps->i), sizeof(int), TARGET_PARAM | TO | FROM 6962 // 6963 // map(ps->s.f) 6964 // ps, &(ps->s.f[0]), 50*sizeof(float), TARGET_PARAM | TO | FROM 6965 // 6966 // map(from: ps->p) 6967 // ps, &(ps->p), sizeof(double*), TARGET_PARAM | FROM 6968 // 6969 // map(to: ps->p[:22]) 6970 // ps, &(ps->p), sizeof(double*), TARGET_PARAM 6971 // ps, &(ps->p), sizeof(double*), MEMBER_OF(1) 6972 // &(ps->p), &(ps->p[0]), 22*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | TO 6973 // 6974 // map(ps->ps) 6975 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM | TO | FROM 6976 // 6977 // map(from: ps->ps->s.i) 6978 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 6979 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 6980 // &(ps->ps), &(ps->ps->s.i), sizeof(int), MEMBER_OF(1) | PTR_AND_OBJ | FROM 6981 // 6982 // map(from: ps->ps->ps) 6983 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 6984 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 6985 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ | FROM 6986 // 6987 // map(ps->ps->ps->ps) 6988 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 6989 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 6990 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 6991 // &(ps->ps->ps), &(ps->ps->ps->ps), sizeof(S2*), PTR_AND_OBJ | TO | FROM 6992 // 6993 // map(to: ps->ps->ps->s.f[:22]) 6994 // ps, &(ps->ps), sizeof(S2*), TARGET_PARAM 6995 // ps, &(ps->ps), sizeof(S2*), MEMBER_OF(1) 6996 // &(ps->ps), &(ps->ps->ps), sizeof(S2*), MEMBER_OF(1) | PTR_AND_OBJ 6997 // &(ps->ps->ps), &(ps->ps->ps->s.f[0]), 22*sizeof(float), PTR_AND_OBJ | TO 6998 // 6999 // map(to: s.f[:22]) map(from: s.p[:33]) 7000 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1) + 7001 // sizeof(double*) (**), TARGET_PARAM 7002 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | TO 7003 // &s, &(s.p), sizeof(double*), MEMBER_OF(1) 7004 // &(s.p), &(s.p[0]), 33*sizeof(double), MEMBER_OF(1) | PTR_AND_OBJ | FROM 7005 // (*) allocate contiguous space needed to fit all mapped members even if 7006 // we allocate space for members not mapped (in this example, 7007 // s.f[22..49] and s.s are not mapped, yet we must allocate space for 7008 // them as well because they fall between &s.f[0] and &s.p) 7009 // 7010 // map(from: s.f[:22]) map(to: ps->p[:33]) 7011 // &s, &(s.f[0]), 22*sizeof(float), TARGET_PARAM | FROM 7012 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7013 // ps, &(ps->p), sizeof(double*), MEMBER_OF(2) (*) 7014 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(2) | PTR_AND_OBJ | TO 7015 // (*) the struct this entry pertains to is the 2nd element in the list of 7016 // arguments, hence MEMBER_OF(2) 7017 // 7018 // map(from: s.f[:22], s.s) map(to: ps->p[:33]) 7019 // &s, &(s.f[0]), 50*sizeof(float) + sizeof(struct S1), TARGET_PARAM 7020 // &s, &(s.f[0]), 22*sizeof(float), MEMBER_OF(1) | FROM 7021 // &s, &(s.s), sizeof(struct S1), MEMBER_OF(1) | FROM 7022 // ps, &(ps->p), sizeof(S2*), TARGET_PARAM 7023 // ps, &(ps->p), sizeof(double*), MEMBER_OF(4) (*) 7024 // &(ps->p), &(ps->p[0]), 33*sizeof(double), MEMBER_OF(4) | PTR_AND_OBJ | TO 7025 // (*) the struct this entry pertains to is the 4th element in the list 7026 // of arguments, hence MEMBER_OF(4) 7027 7028 // Track if the map information being generated is the first for a capture. 7029 bool IsCaptureFirstInfo = IsFirstComponentList; 7030 bool IsLink = false; // Is this variable a "declare target link"? 7031 7032 // Scan the components from the base to the complete expression. 7033 auto CI = Components.rbegin(); 7034 auto CE = Components.rend(); 7035 auto I = CI; 7036 7037 // Track if the map information being generated is the first for a list of 7038 // components. 7039 bool IsExpressionFirstInfo = true; 7040 Address BP = Address::invalid(); 7041 const Expr *AssocExpr = I->getAssociatedExpression(); 7042 const auto *AE = dyn_cast<ArraySubscriptExpr>(AssocExpr); 7043 const auto *OASE = dyn_cast<OMPArraySectionExpr>(AssocExpr); 7044 7045 if (isa<MemberExpr>(AssocExpr)) { 7046 // The base is the 'this' pointer. The content of the pointer is going 7047 // to be the base of the field being mapped. 7048 BP = CGF.LoadCXXThisAddress(); 7049 } else if ((AE && isa<CXXThisExpr>(AE->getBase()->IgnoreParenImpCasts())) || 7050 (OASE && 7051 isa<CXXThisExpr>(OASE->getBase()->IgnoreParenImpCasts()))) { 7052 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(); 7053 } else { 7054 // The base is the reference to the variable. 7055 // BP = &Var. 7056 BP = CGF.EmitOMPSharedLValue(AssocExpr).getAddress(); 7057 if (const auto *VD = 7058 dyn_cast_or_null<VarDecl>(I->getAssociatedDeclaration())) { 7059 if (llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 7060 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD)) 7061 if (*Res == OMPDeclareTargetDeclAttr::MT_Link) { 7062 IsLink = true; 7063 BP = CGF.CGM.getOpenMPRuntime().getAddrOfDeclareTargetLink(VD); 7064 } 7065 } 7066 7067 // If the variable is a pointer and is being dereferenced (i.e. is not 7068 // the last component), the base has to be the pointer itself, not its 7069 // reference. References are ignored for mapping purposes. 7070 QualType Ty = 7071 I->getAssociatedDeclaration()->getType().getNonReferenceType(); 7072 if (Ty->isAnyPointerType() && std::next(I) != CE) { 7073 BP = CGF.EmitLoadOfPointer(BP, Ty->castAs<PointerType>()); 7074 7075 // We do not need to generate individual map information for the 7076 // pointer, it can be associated with the combined storage. 7077 ++I; 7078 } 7079 } 7080 7081 // Track whether a component of the list should be marked as MEMBER_OF some 7082 // combined entry (for partial structs). Only the first PTR_AND_OBJ entry 7083 // in a component list should be marked as MEMBER_OF, all subsequent entries 7084 // do not belong to the base struct. E.g. 7085 // struct S2 s; 7086 // s.ps->ps->ps->f[:] 7087 // (1) (2) (3) (4) 7088 // ps(1) is a member pointer, ps(2) is a pointee of ps(1), so it is a 7089 // PTR_AND_OBJ entry; the PTR is ps(1), so MEMBER_OF the base struct. ps(3) 7090 // is the pointee of ps(2) which is not member of struct s, so it should not 7091 // be marked as such (it is still PTR_AND_OBJ). 7092 // The variable is initialized to false so that PTR_AND_OBJ entries which 7093 // are not struct members are not considered (e.g. array of pointers to 7094 // data). 7095 bool ShouldBeMemberOf = false; 7096 7097 // Variable keeping track of whether or not we have encountered a component 7098 // in the component list which is a member expression. Useful when we have a 7099 // pointer or a final array section, in which case it is the previous 7100 // component in the list which tells us whether we have a member expression. 7101 // E.g. X.f[:] 7102 // While processing the final array section "[:]" it is "f" which tells us 7103 // whether we are dealing with a member of a declared struct. 7104 const MemberExpr *EncounteredME = nullptr; 7105 7106 for (; I != CE; ++I) { 7107 // If the current component is member of a struct (parent struct) mark it. 7108 if (!EncounteredME) { 7109 EncounteredME = dyn_cast<MemberExpr>(I->getAssociatedExpression()); 7110 // If we encounter a PTR_AND_OBJ entry from now on it should be marked 7111 // as MEMBER_OF the parent struct. 7112 if (EncounteredME) 7113 ShouldBeMemberOf = true; 7114 } 7115 7116 auto Next = std::next(I); 7117 7118 // We need to generate the addresses and sizes if this is the last 7119 // component, if the component is a pointer or if it is an array section 7120 // whose length can't be proved to be one. If this is a pointer, it 7121 // becomes the base address for the following components. 7122 7123 // A final array section, is one whose length can't be proved to be one. 7124 bool IsFinalArraySection = 7125 isFinalArraySectionExpression(I->getAssociatedExpression()); 7126 7127 // Get information on whether the element is a pointer. Have to do a 7128 // special treatment for array sections given that they are built-in 7129 // types. 7130 const auto *OASE = 7131 dyn_cast<OMPArraySectionExpr>(I->getAssociatedExpression()); 7132 bool IsPointer = 7133 (OASE && OMPArraySectionExpr::getBaseOriginalType(OASE) 7134 .getCanonicalType() 7135 ->isAnyPointerType()) || 7136 I->getAssociatedExpression()->getType()->isAnyPointerType(); 7137 7138 if (Next == CE || IsPointer || IsFinalArraySection) { 7139 // If this is not the last component, we expect the pointer to be 7140 // associated with an array expression or member expression. 7141 assert((Next == CE || 7142 isa<MemberExpr>(Next->getAssociatedExpression()) || 7143 isa<ArraySubscriptExpr>(Next->getAssociatedExpression()) || 7144 isa<OMPArraySectionExpr>(Next->getAssociatedExpression())) && 7145 "Unexpected expression"); 7146 7147 Address LB = 7148 CGF.EmitOMPSharedLValue(I->getAssociatedExpression()).getAddress(); 7149 7150 // If this component is a pointer inside the base struct then we don't 7151 // need to create any entry for it - it will be combined with the object 7152 // it is pointing to into a single PTR_AND_OBJ entry. 7153 bool IsMemberPointer = 7154 IsPointer && EncounteredME && 7155 (dyn_cast<MemberExpr>(I->getAssociatedExpression()) == 7156 EncounteredME); 7157 if (!OverlappedElements.empty()) { 7158 // Handle base element with the info for overlapped elements. 7159 assert(!PartialStruct.Base.isValid() && "The base element is set."); 7160 assert(Next == CE && 7161 "Expected last element for the overlapped elements."); 7162 assert(!IsPointer && 7163 "Unexpected base element with the pointer type."); 7164 // Mark the whole struct as the struct that requires allocation on the 7165 // device. 7166 PartialStruct.LowestElem = {0, LB}; 7167 CharUnits TypeSize = CGF.getContext().getTypeSizeInChars( 7168 I->getAssociatedExpression()->getType()); 7169 Address HB = CGF.Builder.CreateConstGEP( 7170 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast(LB, 7171 CGF.VoidPtrTy), 7172 TypeSize.getQuantity() - 1, CharUnits::One()); 7173 PartialStruct.HighestElem = { 7174 std::numeric_limits<decltype( 7175 PartialStruct.HighestElem.first)>::max(), 7176 HB}; 7177 PartialStruct.Base = BP; 7178 // Emit data for non-overlapped data. 7179 OpenMPOffloadMappingFlags Flags = 7180 OMP_MAP_MEMBER_OF | 7181 getMapTypeBits(MapType, MapModifiers, IsImplicit, 7182 /*AddPtrFlag=*/false, 7183 /*AddIsTargetParamFlag=*/false); 7184 LB = BP; 7185 llvm::Value *Size = nullptr; 7186 // Do bitcopy of all non-overlapped structure elements. 7187 for (OMPClauseMappableExprCommon::MappableExprComponentListRef 7188 Component : OverlappedElements) { 7189 Address ComponentLB = Address::invalid(); 7190 for (const OMPClauseMappableExprCommon::MappableComponent &MC : 7191 Component) { 7192 if (MC.getAssociatedDeclaration()) { 7193 ComponentLB = 7194 CGF.EmitOMPSharedLValue(MC.getAssociatedExpression()) 7195 .getAddress(); 7196 Size = CGF.Builder.CreatePtrDiff( 7197 CGF.EmitCastToVoidPtr(ComponentLB.getPointer()), 7198 CGF.EmitCastToVoidPtr(LB.getPointer())); 7199 break; 7200 } 7201 } 7202 BasePointers.push_back(BP.getPointer()); 7203 Pointers.push_back(LB.getPointer()); 7204 Sizes.push_back(Size); 7205 Types.push_back(Flags); 7206 LB = CGF.Builder.CreateConstGEP(ComponentLB, 1, 7207 CGF.getPointerSize()); 7208 } 7209 BasePointers.push_back(BP.getPointer()); 7210 Pointers.push_back(LB.getPointer()); 7211 Size = CGF.Builder.CreatePtrDiff( 7212 CGF.EmitCastToVoidPtr( 7213 CGF.Builder.CreateConstGEP(HB, 1, CharUnits::One()) 7214 .getPointer()), 7215 CGF.EmitCastToVoidPtr(LB.getPointer())); 7216 Sizes.push_back(Size); 7217 Types.push_back(Flags); 7218 break; 7219 } 7220 llvm::Value *Size = getExprTypeSize(I->getAssociatedExpression()); 7221 if (!IsMemberPointer) { 7222 BasePointers.push_back(BP.getPointer()); 7223 Pointers.push_back(LB.getPointer()); 7224 Sizes.push_back(Size); 7225 7226 // We need to add a pointer flag for each map that comes from the 7227 // same expression except for the first one. We also need to signal 7228 // this map is the first one that relates with the current capture 7229 // (there is a set of entries for each capture). 7230 OpenMPOffloadMappingFlags Flags = getMapTypeBits( 7231 MapType, MapModifiers, IsImplicit, 7232 !IsExpressionFirstInfo || IsLink, IsCaptureFirstInfo && !IsLink); 7233 7234 if (!IsExpressionFirstInfo) { 7235 // If we have a PTR_AND_OBJ pair where the OBJ is a pointer as well, 7236 // then we reset the TO/FROM/ALWAYS/DELETE flags. 7237 if (IsPointer) 7238 Flags &= ~(OMP_MAP_TO | OMP_MAP_FROM | OMP_MAP_ALWAYS | 7239 OMP_MAP_DELETE); 7240 7241 if (ShouldBeMemberOf) { 7242 // Set placeholder value MEMBER_OF=FFFF to indicate that the flag 7243 // should be later updated with the correct value of MEMBER_OF. 7244 Flags |= OMP_MAP_MEMBER_OF; 7245 // From now on, all subsequent PTR_AND_OBJ entries should not be 7246 // marked as MEMBER_OF. 7247 ShouldBeMemberOf = false; 7248 } 7249 } 7250 7251 Types.push_back(Flags); 7252 } 7253 7254 // If we have encountered a member expression so far, keep track of the 7255 // mapped member. If the parent is "*this", then the value declaration 7256 // is nullptr. 7257 if (EncounteredME) { 7258 const auto *FD = dyn_cast<FieldDecl>(EncounteredME->getMemberDecl()); 7259 unsigned FieldIndex = FD->getFieldIndex(); 7260 7261 // Update info about the lowest and highest elements for this struct 7262 if (!PartialStruct.Base.isValid()) { 7263 PartialStruct.LowestElem = {FieldIndex, LB}; 7264 PartialStruct.HighestElem = {FieldIndex, LB}; 7265 PartialStruct.Base = BP; 7266 } else if (FieldIndex < PartialStruct.LowestElem.first) { 7267 PartialStruct.LowestElem = {FieldIndex, LB}; 7268 } else if (FieldIndex > PartialStruct.HighestElem.first) { 7269 PartialStruct.HighestElem = {FieldIndex, LB}; 7270 } 7271 } 7272 7273 // If we have a final array section, we are done with this expression. 7274 if (IsFinalArraySection) 7275 break; 7276 7277 // The pointer becomes the base for the next element. 7278 if (Next != CE) 7279 BP = LB; 7280 7281 IsExpressionFirstInfo = false; 7282 IsCaptureFirstInfo = false; 7283 } 7284 } 7285 } 7286 7287 /// Return the adjusted map modifiers if the declaration a capture refers to 7288 /// appears in a first-private clause. This is expected to be used only with 7289 /// directives that start with 'target'. 7290 MappableExprsHandler::OpenMPOffloadMappingFlags 7291 getMapModifiersForPrivateClauses(const CapturedStmt::Capture &Cap) const { 7292 assert(Cap.capturesVariable() && "Expected capture by reference only!"); 7293 7294 // A first private variable captured by reference will use only the 7295 // 'private ptr' and 'map to' flag. Return the right flags if the captured 7296 // declaration is known as first-private in this handler. 7297 if (FirstPrivateDecls.count(Cap.getCapturedVar())) 7298 return MappableExprsHandler::OMP_MAP_PRIVATE | 7299 MappableExprsHandler::OMP_MAP_TO; 7300 return MappableExprsHandler::OMP_MAP_TO | 7301 MappableExprsHandler::OMP_MAP_FROM; 7302 } 7303 7304 static OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position) { 7305 // Member of is given by the 16 MSB of the flag, so rotate by 48 bits. 7306 return static_cast<OpenMPOffloadMappingFlags>(((uint64_t)Position + 1) 7307 << 48); 7308 } 7309 7310 static void setCorrectMemberOfFlag(OpenMPOffloadMappingFlags &Flags, 7311 OpenMPOffloadMappingFlags MemberOfFlag) { 7312 // If the entry is PTR_AND_OBJ but has not been marked with the special 7313 // placeholder value 0xFFFF in the MEMBER_OF field, then it should not be 7314 // marked as MEMBER_OF. 7315 if ((Flags & OMP_MAP_PTR_AND_OBJ) && 7316 ((Flags & OMP_MAP_MEMBER_OF) != OMP_MAP_MEMBER_OF)) 7317 return; 7318 7319 // Reset the placeholder value to prepare the flag for the assignment of the 7320 // proper MEMBER_OF value. 7321 Flags &= ~OMP_MAP_MEMBER_OF; 7322 Flags |= MemberOfFlag; 7323 } 7324 7325 void getPlainLayout(const CXXRecordDecl *RD, 7326 llvm::SmallVectorImpl<const FieldDecl *> &Layout, 7327 bool AsBase) const { 7328 const CGRecordLayout &RL = CGF.getTypes().getCGRecordLayout(RD); 7329 7330 llvm::StructType *St = 7331 AsBase ? RL.getBaseSubobjectLLVMType() : RL.getLLVMType(); 7332 7333 unsigned NumElements = St->getNumElements(); 7334 llvm::SmallVector< 7335 llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *>, 4> 7336 RecordLayout(NumElements); 7337 7338 // Fill bases. 7339 for (const auto &I : RD->bases()) { 7340 if (I.isVirtual()) 7341 continue; 7342 const auto *Base = I.getType()->getAsCXXRecordDecl(); 7343 // Ignore empty bases. 7344 if (Base->isEmpty() || CGF.getContext() 7345 .getASTRecordLayout(Base) 7346 .getNonVirtualSize() 7347 .isZero()) 7348 continue; 7349 7350 unsigned FieldIndex = RL.getNonVirtualBaseLLVMFieldNo(Base); 7351 RecordLayout[FieldIndex] = Base; 7352 } 7353 // Fill in virtual bases. 7354 for (const auto &I : RD->vbases()) { 7355 const auto *Base = I.getType()->getAsCXXRecordDecl(); 7356 // Ignore empty bases. 7357 if (Base->isEmpty()) 7358 continue; 7359 unsigned FieldIndex = RL.getVirtualBaseIndex(Base); 7360 if (RecordLayout[FieldIndex]) 7361 continue; 7362 RecordLayout[FieldIndex] = Base; 7363 } 7364 // Fill in all the fields. 7365 assert(!RD->isUnion() && "Unexpected union."); 7366 for (const auto *Field : RD->fields()) { 7367 // Fill in non-bitfields. (Bitfields always use a zero pattern, which we 7368 // will fill in later.) 7369 if (!Field->isBitField()) { 7370 unsigned FieldIndex = RL.getLLVMFieldNo(Field); 7371 RecordLayout[FieldIndex] = Field; 7372 } 7373 } 7374 for (const llvm::PointerUnion<const CXXRecordDecl *, const FieldDecl *> 7375 &Data : RecordLayout) { 7376 if (Data.isNull()) 7377 continue; 7378 if (const auto *Base = Data.dyn_cast<const CXXRecordDecl *>()) 7379 getPlainLayout(Base, Layout, /*AsBase=*/true); 7380 else 7381 Layout.push_back(Data.get<const FieldDecl *>()); 7382 } 7383 } 7384 7385 public: 7386 MappableExprsHandler(const OMPExecutableDirective &Dir, CodeGenFunction &CGF) 7387 : CurDir(Dir), CGF(CGF) { 7388 // Extract firstprivate clause information. 7389 for (const auto *C : Dir.getClausesOfKind<OMPFirstprivateClause>()) 7390 for (const auto *D : C->varlists()) 7391 FirstPrivateDecls.insert( 7392 cast<VarDecl>(cast<DeclRefExpr>(D)->getDecl())->getCanonicalDecl()); 7393 // Extract device pointer clause information. 7394 for (const auto *C : Dir.getClausesOfKind<OMPIsDevicePtrClause>()) 7395 for (auto L : C->component_lists()) 7396 DevPointersMap[L.first].push_back(L.second); 7397 } 7398 7399 /// Generate code for the combined entry if we have a partially mapped struct 7400 /// and take care of the mapping flags of the arguments corresponding to 7401 /// individual struct members. 7402 void emitCombinedEntry(MapBaseValuesArrayTy &BasePointers, 7403 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 7404 MapFlagsArrayTy &Types, MapFlagsArrayTy &CurTypes, 7405 const StructRangeInfoTy &PartialStruct) const { 7406 // Base is the base of the struct 7407 BasePointers.push_back(PartialStruct.Base.getPointer()); 7408 // Pointer is the address of the lowest element 7409 llvm::Value *LB = PartialStruct.LowestElem.second.getPointer(); 7410 Pointers.push_back(LB); 7411 // Size is (addr of {highest+1} element) - (addr of lowest element) 7412 llvm::Value *HB = PartialStruct.HighestElem.second.getPointer(); 7413 llvm::Value *HAddr = CGF.Builder.CreateConstGEP1_32(HB, /*Idx0=*/1); 7414 llvm::Value *CLAddr = CGF.Builder.CreatePointerCast(LB, CGF.VoidPtrTy); 7415 llvm::Value *CHAddr = CGF.Builder.CreatePointerCast(HAddr, CGF.VoidPtrTy); 7416 llvm::Value *Diff = CGF.Builder.CreatePtrDiff(CHAddr, CLAddr); 7417 llvm::Value *Size = CGF.Builder.CreateIntCast(Diff, CGF.SizeTy, 7418 /*isSinged=*/false); 7419 Sizes.push_back(Size); 7420 // Map type is always TARGET_PARAM 7421 Types.push_back(OMP_MAP_TARGET_PARAM); 7422 // Remove TARGET_PARAM flag from the first element 7423 (*CurTypes.begin()) &= ~OMP_MAP_TARGET_PARAM; 7424 7425 // All other current entries will be MEMBER_OF the combined entry 7426 // (except for PTR_AND_OBJ entries which do not have a placeholder value 7427 // 0xFFFF in the MEMBER_OF field). 7428 OpenMPOffloadMappingFlags MemberOfFlag = 7429 getMemberOfFlag(BasePointers.size() - 1); 7430 for (auto &M : CurTypes) 7431 setCorrectMemberOfFlag(M, MemberOfFlag); 7432 } 7433 7434 /// Generate all the base pointers, section pointers, sizes and map 7435 /// types for the extracted mappable expressions. Also, for each item that 7436 /// relates with a device pointer, a pair of the relevant declaration and 7437 /// index where it occurs is appended to the device pointers info array. 7438 void generateAllInfo(MapBaseValuesArrayTy &BasePointers, 7439 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 7440 MapFlagsArrayTy &Types) const { 7441 // We have to process the component lists that relate with the same 7442 // declaration in a single chunk so that we can generate the map flags 7443 // correctly. Therefore, we organize all lists in a map. 7444 llvm::MapVector<const ValueDecl *, SmallVector<MapInfo, 8>> Info; 7445 7446 // Helper function to fill the information map for the different supported 7447 // clauses. 7448 auto &&InfoGen = [&Info]( 7449 const ValueDecl *D, 7450 OMPClauseMappableExprCommon::MappableExprComponentListRef L, 7451 OpenMPMapClauseKind MapType, 7452 ArrayRef<OpenMPMapModifierKind> MapModifiers, 7453 bool ReturnDevicePointer, bool IsImplicit) { 7454 const ValueDecl *VD = 7455 D ? cast<ValueDecl>(D->getCanonicalDecl()) : nullptr; 7456 Info[VD].emplace_back(L, MapType, MapModifiers, ReturnDevicePointer, 7457 IsImplicit); 7458 }; 7459 7460 // FIXME: MSVC 2013 seems to require this-> to find member CurDir. 7461 for (const auto *C : this->CurDir.getClausesOfKind<OMPMapClause>()) 7462 for (const auto &L : C->component_lists()) { 7463 InfoGen(L.first, L.second, C->getMapType(), C->getMapTypeModifiers(), 7464 /*ReturnDevicePointer=*/false, C->isImplicit()); 7465 } 7466 for (const auto *C : this->CurDir.getClausesOfKind<OMPToClause>()) 7467 for (const auto &L : C->component_lists()) { 7468 InfoGen(L.first, L.second, OMPC_MAP_to, llvm::None, 7469 /*ReturnDevicePointer=*/false, C->isImplicit()); 7470 } 7471 for (const auto *C : this->CurDir.getClausesOfKind<OMPFromClause>()) 7472 for (const auto &L : C->component_lists()) { 7473 InfoGen(L.first, L.second, OMPC_MAP_from, llvm::None, 7474 /*ReturnDevicePointer=*/false, C->isImplicit()); 7475 } 7476 7477 // Look at the use_device_ptr clause information and mark the existing map 7478 // entries as such. If there is no map information for an entry in the 7479 // use_device_ptr list, we create one with map type 'alloc' and zero size 7480 // section. It is the user fault if that was not mapped before. If there is 7481 // no map information and the pointer is a struct member, then we defer the 7482 // emission of that entry until the whole struct has been processed. 7483 llvm::MapVector<const ValueDecl *, SmallVector<DeferredDevicePtrEntryTy, 4>> 7484 DeferredInfo; 7485 7486 // FIXME: MSVC 2013 seems to require this-> to find member CurDir. 7487 for (const auto *C : 7488 this->CurDir.getClausesOfKind<OMPUseDevicePtrClause>()) { 7489 for (const auto &L : C->component_lists()) { 7490 assert(!L.second.empty() && "Not expecting empty list of components!"); 7491 const ValueDecl *VD = L.second.back().getAssociatedDeclaration(); 7492 VD = cast<ValueDecl>(VD->getCanonicalDecl()); 7493 const Expr *IE = L.second.back().getAssociatedExpression(); 7494 // If the first component is a member expression, we have to look into 7495 // 'this', which maps to null in the map of map information. Otherwise 7496 // look directly for the information. 7497 auto It = Info.find(isa<MemberExpr>(IE) ? nullptr : VD); 7498 7499 // We potentially have map information for this declaration already. 7500 // Look for the first set of components that refer to it. 7501 if (It != Info.end()) { 7502 auto CI = std::find_if( 7503 It->second.begin(), It->second.end(), [VD](const MapInfo &MI) { 7504 return MI.Components.back().getAssociatedDeclaration() == VD; 7505 }); 7506 // If we found a map entry, signal that the pointer has to be returned 7507 // and move on to the next declaration. 7508 if (CI != It->second.end()) { 7509 CI->ReturnDevicePointer = true; 7510 continue; 7511 } 7512 } 7513 7514 // We didn't find any match in our map information - generate a zero 7515 // size array section - if the pointer is a struct member we defer this 7516 // action until the whole struct has been processed. 7517 // FIXME: MSVC 2013 seems to require this-> to find member CGF. 7518 if (isa<MemberExpr>(IE)) { 7519 // Insert the pointer into Info to be processed by 7520 // generateInfoForComponentList. Because it is a member pointer 7521 // without a pointee, no entry will be generated for it, therefore 7522 // we need to generate one after the whole struct has been processed. 7523 // Nonetheless, generateInfoForComponentList must be called to take 7524 // the pointer into account for the calculation of the range of the 7525 // partial struct. 7526 InfoGen(nullptr, L.second, OMPC_MAP_unknown, llvm::None, 7527 /*ReturnDevicePointer=*/false, C->isImplicit()); 7528 DeferredInfo[nullptr].emplace_back(IE, VD); 7529 } else { 7530 llvm::Value *Ptr = this->CGF.EmitLoadOfScalar( 7531 this->CGF.EmitLValue(IE), IE->getExprLoc()); 7532 BasePointers.emplace_back(Ptr, VD); 7533 Pointers.push_back(Ptr); 7534 Sizes.push_back(llvm::Constant::getNullValue(this->CGF.SizeTy)); 7535 Types.push_back(OMP_MAP_RETURN_PARAM | OMP_MAP_TARGET_PARAM); 7536 } 7537 } 7538 } 7539 7540 for (const auto &M : Info) { 7541 // We need to know when we generate information for the first component 7542 // associated with a capture, because the mapping flags depend on it. 7543 bool IsFirstComponentList = true; 7544 7545 // Temporary versions of arrays 7546 MapBaseValuesArrayTy CurBasePointers; 7547 MapValuesArrayTy CurPointers; 7548 MapValuesArrayTy CurSizes; 7549 MapFlagsArrayTy CurTypes; 7550 StructRangeInfoTy PartialStruct; 7551 7552 for (const MapInfo &L : M.second) { 7553 assert(!L.Components.empty() && 7554 "Not expecting declaration with no component lists."); 7555 7556 // Remember the current base pointer index. 7557 unsigned CurrentBasePointersIdx = CurBasePointers.size(); 7558 // FIXME: MSVC 2013 seems to require this-> to find the member method. 7559 this->generateInfoForComponentList( 7560 L.MapType, L.MapModifiers, L.Components, CurBasePointers, 7561 CurPointers, CurSizes, CurTypes, PartialStruct, 7562 IsFirstComponentList, L.IsImplicit); 7563 7564 // If this entry relates with a device pointer, set the relevant 7565 // declaration and add the 'return pointer' flag. 7566 if (L.ReturnDevicePointer) { 7567 assert(CurBasePointers.size() > CurrentBasePointersIdx && 7568 "Unexpected number of mapped base pointers."); 7569 7570 const ValueDecl *RelevantVD = 7571 L.Components.back().getAssociatedDeclaration(); 7572 assert(RelevantVD && 7573 "No relevant declaration related with device pointer??"); 7574 7575 CurBasePointers[CurrentBasePointersIdx].setDevicePtrDecl(RelevantVD); 7576 CurTypes[CurrentBasePointersIdx] |= OMP_MAP_RETURN_PARAM; 7577 } 7578 IsFirstComponentList = false; 7579 } 7580 7581 // Append any pending zero-length pointers which are struct members and 7582 // used with use_device_ptr. 7583 auto CI = DeferredInfo.find(M.first); 7584 if (CI != DeferredInfo.end()) { 7585 for (const DeferredDevicePtrEntryTy &L : CI->second) { 7586 llvm::Value *BasePtr = this->CGF.EmitLValue(L.IE).getPointer(); 7587 llvm::Value *Ptr = this->CGF.EmitLoadOfScalar( 7588 this->CGF.EmitLValue(L.IE), L.IE->getExprLoc()); 7589 CurBasePointers.emplace_back(BasePtr, L.VD); 7590 CurPointers.push_back(Ptr); 7591 CurSizes.push_back(llvm::Constant::getNullValue(this->CGF.SizeTy)); 7592 // Entry is PTR_AND_OBJ and RETURN_PARAM. Also, set the placeholder 7593 // value MEMBER_OF=FFFF so that the entry is later updated with the 7594 // correct value of MEMBER_OF. 7595 CurTypes.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_RETURN_PARAM | 7596 OMP_MAP_MEMBER_OF); 7597 } 7598 } 7599 7600 // If there is an entry in PartialStruct it means we have a struct with 7601 // individual members mapped. Emit an extra combined entry. 7602 if (PartialStruct.Base.isValid()) 7603 emitCombinedEntry(BasePointers, Pointers, Sizes, Types, CurTypes, 7604 PartialStruct); 7605 7606 // We need to append the results of this capture to what we already have. 7607 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 7608 Pointers.append(CurPointers.begin(), CurPointers.end()); 7609 Sizes.append(CurSizes.begin(), CurSizes.end()); 7610 Types.append(CurTypes.begin(), CurTypes.end()); 7611 } 7612 } 7613 7614 /// Emit capture info for lambdas for variables captured by reference. 7615 void generateInfoForLambdaCaptures( 7616 const ValueDecl *VD, llvm::Value *Arg, MapBaseValuesArrayTy &BasePointers, 7617 MapValuesArrayTy &Pointers, MapValuesArrayTy &Sizes, 7618 MapFlagsArrayTy &Types, 7619 llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers) const { 7620 const auto *RD = VD->getType() 7621 .getCanonicalType() 7622 .getNonReferenceType() 7623 ->getAsCXXRecordDecl(); 7624 if (!RD || !RD->isLambda()) 7625 return; 7626 Address VDAddr = Address(Arg, CGF.getContext().getDeclAlign(VD)); 7627 LValue VDLVal = CGF.MakeAddrLValue( 7628 VDAddr, VD->getType().getCanonicalType().getNonReferenceType()); 7629 llvm::DenseMap<const VarDecl *, FieldDecl *> Captures; 7630 FieldDecl *ThisCapture = nullptr; 7631 RD->getCaptureFields(Captures, ThisCapture); 7632 if (ThisCapture) { 7633 LValue ThisLVal = 7634 CGF.EmitLValueForFieldInitialization(VDLVal, ThisCapture); 7635 LValue ThisLValVal = CGF.EmitLValueForField(VDLVal, ThisCapture); 7636 LambdaPointers.try_emplace(ThisLVal.getPointer(), VDLVal.getPointer()); 7637 BasePointers.push_back(ThisLVal.getPointer()); 7638 Pointers.push_back(ThisLValVal.getPointer()); 7639 Sizes.push_back(CGF.getTypeSize(CGF.getContext().VoidPtrTy)); 7640 Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 7641 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 7642 } 7643 for (const LambdaCapture &LC : RD->captures()) { 7644 if (LC.getCaptureKind() != LCK_ByRef) 7645 continue; 7646 const VarDecl *VD = LC.getCapturedVar(); 7647 auto It = Captures.find(VD); 7648 assert(It != Captures.end() && "Found lambda capture without field."); 7649 LValue VarLVal = CGF.EmitLValueForFieldInitialization(VDLVal, It->second); 7650 LValue VarLValVal = CGF.EmitLValueForField(VDLVal, It->second); 7651 LambdaPointers.try_emplace(VarLVal.getPointer(), VDLVal.getPointer()); 7652 BasePointers.push_back(VarLVal.getPointer()); 7653 Pointers.push_back(VarLValVal.getPointer()); 7654 Sizes.push_back(CGF.getTypeSize( 7655 VD->getType().getCanonicalType().getNonReferenceType())); 7656 Types.push_back(OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 7657 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT); 7658 } 7659 } 7660 7661 /// Set correct indices for lambdas captures. 7662 void adjustMemberOfForLambdaCaptures( 7663 const llvm::DenseMap<llvm::Value *, llvm::Value *> &LambdaPointers, 7664 MapBaseValuesArrayTy &BasePointers, MapValuesArrayTy &Pointers, 7665 MapFlagsArrayTy &Types) const { 7666 for (unsigned I = 0, E = Types.size(); I < E; ++I) { 7667 // Set correct member_of idx for all implicit lambda captures. 7668 if (Types[I] != (OMP_MAP_PTR_AND_OBJ | OMP_MAP_LITERAL | 7669 OMP_MAP_MEMBER_OF | OMP_MAP_IMPLICIT)) 7670 continue; 7671 llvm::Value *BasePtr = LambdaPointers.lookup(*BasePointers[I]); 7672 assert(BasePtr && "Unable to find base lambda address."); 7673 int TgtIdx = -1; 7674 for (unsigned J = I; J > 0; --J) { 7675 unsigned Idx = J - 1; 7676 if (Pointers[Idx] != BasePtr) 7677 continue; 7678 TgtIdx = Idx; 7679 break; 7680 } 7681 assert(TgtIdx != -1 && "Unable to find parent lambda."); 7682 // All other current entries will be MEMBER_OF the combined entry 7683 // (except for PTR_AND_OBJ entries which do not have a placeholder value 7684 // 0xFFFF in the MEMBER_OF field). 7685 OpenMPOffloadMappingFlags MemberOfFlag = getMemberOfFlag(TgtIdx); 7686 setCorrectMemberOfFlag(Types[I], MemberOfFlag); 7687 } 7688 } 7689 7690 /// Generate the base pointers, section pointers, sizes and map types 7691 /// associated to a given capture. 7692 void generateInfoForCapture(const CapturedStmt::Capture *Cap, 7693 llvm::Value *Arg, 7694 MapBaseValuesArrayTy &BasePointers, 7695 MapValuesArrayTy &Pointers, 7696 MapValuesArrayTy &Sizes, MapFlagsArrayTy &Types, 7697 StructRangeInfoTy &PartialStruct) const { 7698 assert(!Cap->capturesVariableArrayType() && 7699 "Not expecting to generate map info for a variable array type!"); 7700 7701 // We need to know when we generating information for the first component 7702 const ValueDecl *VD = Cap->capturesThis() 7703 ? nullptr 7704 : Cap->getCapturedVar()->getCanonicalDecl(); 7705 7706 // If this declaration appears in a is_device_ptr clause we just have to 7707 // pass the pointer by value. If it is a reference to a declaration, we just 7708 // pass its value. 7709 if (DevPointersMap.count(VD)) { 7710 BasePointers.emplace_back(Arg, VD); 7711 Pointers.push_back(Arg); 7712 Sizes.push_back(CGF.getTypeSize(CGF.getContext().VoidPtrTy)); 7713 Types.push_back(OMP_MAP_LITERAL | OMP_MAP_TARGET_PARAM); 7714 return; 7715 } 7716 7717 using MapData = 7718 std::tuple<OMPClauseMappableExprCommon::MappableExprComponentListRef, 7719 OpenMPMapClauseKind, ArrayRef<OpenMPMapModifierKind>, bool>; 7720 SmallVector<MapData, 4> DeclComponentLists; 7721 // FIXME: MSVC 2013 seems to require this-> to find member CurDir. 7722 for (const auto *C : this->CurDir.getClausesOfKind<OMPMapClause>()) { 7723 for (const auto &L : C->decl_component_lists(VD)) { 7724 assert(L.first == VD && 7725 "We got information for the wrong declaration??"); 7726 assert(!L.second.empty() && 7727 "Not expecting declaration with no component lists."); 7728 DeclComponentLists.emplace_back(L.second, C->getMapType(), 7729 C->getMapTypeModifiers(), 7730 C->isImplicit()); 7731 } 7732 } 7733 7734 // Find overlapping elements (including the offset from the base element). 7735 llvm::SmallDenseMap< 7736 const MapData *, 7737 llvm::SmallVector< 7738 OMPClauseMappableExprCommon::MappableExprComponentListRef, 4>, 7739 4> 7740 OverlappedData; 7741 size_t Count = 0; 7742 for (const MapData &L : DeclComponentLists) { 7743 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 7744 OpenMPMapClauseKind MapType; 7745 ArrayRef<OpenMPMapModifierKind> MapModifiers; 7746 bool IsImplicit; 7747 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 7748 ++Count; 7749 for (const MapData &L1 : makeArrayRef(DeclComponentLists).slice(Count)) { 7750 OMPClauseMappableExprCommon::MappableExprComponentListRef Components1; 7751 std::tie(Components1, MapType, MapModifiers, IsImplicit) = L1; 7752 auto CI = Components.rbegin(); 7753 auto CE = Components.rend(); 7754 auto SI = Components1.rbegin(); 7755 auto SE = Components1.rend(); 7756 for (; CI != CE && SI != SE; ++CI, ++SI) { 7757 if (CI->getAssociatedExpression()->getStmtClass() != 7758 SI->getAssociatedExpression()->getStmtClass()) 7759 break; 7760 // Are we dealing with different variables/fields? 7761 if (CI->getAssociatedDeclaration() != SI->getAssociatedDeclaration()) 7762 break; 7763 } 7764 // Found overlapping if, at least for one component, reached the head of 7765 // the components list. 7766 if (CI == CE || SI == SE) { 7767 assert((CI != CE || SI != SE) && 7768 "Unexpected full match of the mapping components."); 7769 const MapData &BaseData = CI == CE ? L : L1; 7770 OMPClauseMappableExprCommon::MappableExprComponentListRef SubData = 7771 SI == SE ? Components : Components1; 7772 auto &OverlappedElements = OverlappedData.FindAndConstruct(&BaseData); 7773 OverlappedElements.getSecond().push_back(SubData); 7774 } 7775 } 7776 } 7777 // Sort the overlapped elements for each item. 7778 llvm::SmallVector<const FieldDecl *, 4> Layout; 7779 if (!OverlappedData.empty()) { 7780 if (const auto *CRD = 7781 VD->getType().getCanonicalType()->getAsCXXRecordDecl()) 7782 getPlainLayout(CRD, Layout, /*AsBase=*/false); 7783 else { 7784 const auto *RD = VD->getType().getCanonicalType()->getAsRecordDecl(); 7785 Layout.append(RD->field_begin(), RD->field_end()); 7786 } 7787 } 7788 for (auto &Pair : OverlappedData) { 7789 llvm::sort( 7790 Pair.getSecond(), 7791 [&Layout]( 7792 OMPClauseMappableExprCommon::MappableExprComponentListRef First, 7793 OMPClauseMappableExprCommon::MappableExprComponentListRef 7794 Second) { 7795 auto CI = First.rbegin(); 7796 auto CE = First.rend(); 7797 auto SI = Second.rbegin(); 7798 auto SE = Second.rend(); 7799 for (; CI != CE && SI != SE; ++CI, ++SI) { 7800 if (CI->getAssociatedExpression()->getStmtClass() != 7801 SI->getAssociatedExpression()->getStmtClass()) 7802 break; 7803 // Are we dealing with different variables/fields? 7804 if (CI->getAssociatedDeclaration() != 7805 SI->getAssociatedDeclaration()) 7806 break; 7807 } 7808 7809 // Lists contain the same elements. 7810 if (CI == CE && SI == SE) 7811 return false; 7812 7813 // List with less elements is less than list with more elements. 7814 if (CI == CE || SI == SE) 7815 return CI == CE; 7816 7817 const auto *FD1 = cast<FieldDecl>(CI->getAssociatedDeclaration()); 7818 const auto *FD2 = cast<FieldDecl>(SI->getAssociatedDeclaration()); 7819 if (FD1->getParent() == FD2->getParent()) 7820 return FD1->getFieldIndex() < FD2->getFieldIndex(); 7821 const auto It = 7822 llvm::find_if(Layout, [FD1, FD2](const FieldDecl *FD) { 7823 return FD == FD1 || FD == FD2; 7824 }); 7825 return *It == FD1; 7826 }); 7827 } 7828 7829 // Associated with a capture, because the mapping flags depend on it. 7830 // Go through all of the elements with the overlapped elements. 7831 for (const auto &Pair : OverlappedData) { 7832 const MapData &L = *Pair.getFirst(); 7833 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 7834 OpenMPMapClauseKind MapType; 7835 ArrayRef<OpenMPMapModifierKind> MapModifiers; 7836 bool IsImplicit; 7837 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 7838 ArrayRef<OMPClauseMappableExprCommon::MappableExprComponentListRef> 7839 OverlappedComponents = Pair.getSecond(); 7840 bool IsFirstComponentList = true; 7841 generateInfoForComponentList(MapType, MapModifiers, Components, 7842 BasePointers, Pointers, Sizes, Types, 7843 PartialStruct, IsFirstComponentList, 7844 IsImplicit, OverlappedComponents); 7845 } 7846 // Go through other elements without overlapped elements. 7847 bool IsFirstComponentList = OverlappedData.empty(); 7848 for (const MapData &L : DeclComponentLists) { 7849 OMPClauseMappableExprCommon::MappableExprComponentListRef Components; 7850 OpenMPMapClauseKind MapType; 7851 ArrayRef<OpenMPMapModifierKind> MapModifiers; 7852 bool IsImplicit; 7853 std::tie(Components, MapType, MapModifiers, IsImplicit) = L; 7854 auto It = OverlappedData.find(&L); 7855 if (It == OverlappedData.end()) 7856 generateInfoForComponentList(MapType, MapModifiers, Components, 7857 BasePointers, Pointers, Sizes, Types, 7858 PartialStruct, IsFirstComponentList, 7859 IsImplicit); 7860 IsFirstComponentList = false; 7861 } 7862 } 7863 7864 /// Generate the base pointers, section pointers, sizes and map types 7865 /// associated with the declare target link variables. 7866 void generateInfoForDeclareTargetLink(MapBaseValuesArrayTy &BasePointers, 7867 MapValuesArrayTy &Pointers, 7868 MapValuesArrayTy &Sizes, 7869 MapFlagsArrayTy &Types) const { 7870 // Map other list items in the map clause which are not captured variables 7871 // but "declare target link" global variables., 7872 for (const auto *C : this->CurDir.getClausesOfKind<OMPMapClause>()) { 7873 for (const auto &L : C->component_lists()) { 7874 if (!L.first) 7875 continue; 7876 const auto *VD = dyn_cast<VarDecl>(L.first); 7877 if (!VD) 7878 continue; 7879 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 7880 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 7881 if (!Res || *Res != OMPDeclareTargetDeclAttr::MT_Link) 7882 continue; 7883 StructRangeInfoTy PartialStruct; 7884 generateInfoForComponentList( 7885 C->getMapType(), C->getMapTypeModifiers(), L.second, BasePointers, 7886 Pointers, Sizes, Types, PartialStruct, 7887 /*IsFirstComponentList=*/true, C->isImplicit()); 7888 assert(!PartialStruct.Base.isValid() && 7889 "No partial structs for declare target link expected."); 7890 } 7891 } 7892 } 7893 7894 /// Generate the default map information for a given capture \a CI, 7895 /// record field declaration \a RI and captured value \a CV. 7896 void generateDefaultMapInfo(const CapturedStmt::Capture &CI, 7897 const FieldDecl &RI, llvm::Value *CV, 7898 MapBaseValuesArrayTy &CurBasePointers, 7899 MapValuesArrayTy &CurPointers, 7900 MapValuesArrayTy &CurSizes, 7901 MapFlagsArrayTy &CurMapTypes) const { 7902 // Do the default mapping. 7903 if (CI.capturesThis()) { 7904 CurBasePointers.push_back(CV); 7905 CurPointers.push_back(CV); 7906 const auto *PtrTy = cast<PointerType>(RI.getType().getTypePtr()); 7907 CurSizes.push_back(CGF.getTypeSize(PtrTy->getPointeeType())); 7908 // Default map type. 7909 CurMapTypes.push_back(OMP_MAP_TO | OMP_MAP_FROM); 7910 } else if (CI.capturesVariableByCopy()) { 7911 CurBasePointers.push_back(CV); 7912 CurPointers.push_back(CV); 7913 if (!RI.getType()->isAnyPointerType()) { 7914 // We have to signal to the runtime captures passed by value that are 7915 // not pointers. 7916 CurMapTypes.push_back(OMP_MAP_LITERAL); 7917 CurSizes.push_back(CGF.getTypeSize(RI.getType())); 7918 } else { 7919 // Pointers are implicitly mapped with a zero size and no flags 7920 // (other than first map that is added for all implicit maps). 7921 CurMapTypes.push_back(OMP_MAP_NONE); 7922 CurSizes.push_back(llvm::Constant::getNullValue(CGF.SizeTy)); 7923 } 7924 } else { 7925 assert(CI.capturesVariable() && "Expected captured reference."); 7926 CurBasePointers.push_back(CV); 7927 CurPointers.push_back(CV); 7928 7929 const auto *PtrTy = cast<ReferenceType>(RI.getType().getTypePtr()); 7930 QualType ElementType = PtrTy->getPointeeType(); 7931 CurSizes.push_back(CGF.getTypeSize(ElementType)); 7932 // The default map type for a scalar/complex type is 'to' because by 7933 // default the value doesn't have to be retrieved. For an aggregate 7934 // type, the default is 'tofrom'. 7935 CurMapTypes.push_back(getMapModifiersForPrivateClauses(CI)); 7936 } 7937 // Every default map produces a single argument which is a target parameter. 7938 CurMapTypes.back() |= OMP_MAP_TARGET_PARAM; 7939 7940 // Add flag stating this is an implicit map. 7941 CurMapTypes.back() |= OMP_MAP_IMPLICIT; 7942 } 7943 }; 7944 7945 enum OpenMPOffloadingReservedDeviceIDs { 7946 /// Device ID if the device was not defined, runtime should get it 7947 /// from environment variables in the spec. 7948 OMP_DEVICEID_UNDEF = -1, 7949 }; 7950 } // anonymous namespace 7951 7952 /// Emit the arrays used to pass the captures and map information to the 7953 /// offloading runtime library. If there is no map or capture information, 7954 /// return nullptr by reference. 7955 static void 7956 emitOffloadingArrays(CodeGenFunction &CGF, 7957 MappableExprsHandler::MapBaseValuesArrayTy &BasePointers, 7958 MappableExprsHandler::MapValuesArrayTy &Pointers, 7959 MappableExprsHandler::MapValuesArrayTy &Sizes, 7960 MappableExprsHandler::MapFlagsArrayTy &MapTypes, 7961 CGOpenMPRuntime::TargetDataInfo &Info) { 7962 CodeGenModule &CGM = CGF.CGM; 7963 ASTContext &Ctx = CGF.getContext(); 7964 7965 // Reset the array information. 7966 Info.clearArrayInfo(); 7967 Info.NumberOfPtrs = BasePointers.size(); 7968 7969 if (Info.NumberOfPtrs) { 7970 // Detect if we have any capture size requiring runtime evaluation of the 7971 // size so that a constant array could be eventually used. 7972 bool hasRuntimeEvaluationCaptureSize = false; 7973 for (llvm::Value *S : Sizes) 7974 if (!isa<llvm::Constant>(S)) { 7975 hasRuntimeEvaluationCaptureSize = true; 7976 break; 7977 } 7978 7979 llvm::APInt PointerNumAP(32, Info.NumberOfPtrs, /*isSigned=*/true); 7980 QualType PointerArrayType = 7981 Ctx.getConstantArrayType(Ctx.VoidPtrTy, PointerNumAP, ArrayType::Normal, 7982 /*IndexTypeQuals=*/0); 7983 7984 Info.BasePointersArray = 7985 CGF.CreateMemTemp(PointerArrayType, ".offload_baseptrs").getPointer(); 7986 Info.PointersArray = 7987 CGF.CreateMemTemp(PointerArrayType, ".offload_ptrs").getPointer(); 7988 7989 // If we don't have any VLA types or other types that require runtime 7990 // evaluation, we can use a constant array for the map sizes, otherwise we 7991 // need to fill up the arrays as we do for the pointers. 7992 if (hasRuntimeEvaluationCaptureSize) { 7993 QualType SizeArrayType = Ctx.getConstantArrayType( 7994 Ctx.getSizeType(), PointerNumAP, ArrayType::Normal, 7995 /*IndexTypeQuals=*/0); 7996 Info.SizesArray = 7997 CGF.CreateMemTemp(SizeArrayType, ".offload_sizes").getPointer(); 7998 } else { 7999 // We expect all the sizes to be constant, so we collect them to create 8000 // a constant array. 8001 SmallVector<llvm::Constant *, 16> ConstSizes; 8002 for (llvm::Value *S : Sizes) 8003 ConstSizes.push_back(cast<llvm::Constant>(S)); 8004 8005 auto *SizesArrayInit = llvm::ConstantArray::get( 8006 llvm::ArrayType::get(CGM.SizeTy, ConstSizes.size()), ConstSizes); 8007 std::string Name = CGM.getOpenMPRuntime().getName({"offload_sizes"}); 8008 auto *SizesArrayGbl = new llvm::GlobalVariable( 8009 CGM.getModule(), SizesArrayInit->getType(), 8010 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 8011 SizesArrayInit, Name); 8012 SizesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 8013 Info.SizesArray = SizesArrayGbl; 8014 } 8015 8016 // The map types are always constant so we don't need to generate code to 8017 // fill arrays. Instead, we create an array constant. 8018 SmallVector<uint64_t, 4> Mapping(MapTypes.size(), 0); 8019 llvm::copy(MapTypes, Mapping.begin()); 8020 llvm::Constant *MapTypesArrayInit = 8021 llvm::ConstantDataArray::get(CGF.Builder.getContext(), Mapping); 8022 std::string MaptypesName = 8023 CGM.getOpenMPRuntime().getName({"offload_maptypes"}); 8024 auto *MapTypesArrayGbl = new llvm::GlobalVariable( 8025 CGM.getModule(), MapTypesArrayInit->getType(), 8026 /*isConstant=*/true, llvm::GlobalValue::PrivateLinkage, 8027 MapTypesArrayInit, MaptypesName); 8028 MapTypesArrayGbl->setUnnamedAddr(llvm::GlobalValue::UnnamedAddr::Global); 8029 Info.MapTypesArray = MapTypesArrayGbl; 8030 8031 for (unsigned I = 0; I < Info.NumberOfPtrs; ++I) { 8032 llvm::Value *BPVal = *BasePointers[I]; 8033 llvm::Value *BP = CGF.Builder.CreateConstInBoundsGEP2_32( 8034 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8035 Info.BasePointersArray, 0, I); 8036 BP = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8037 BP, BPVal->getType()->getPointerTo(/*AddrSpace=*/0)); 8038 Address BPAddr(BP, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 8039 CGF.Builder.CreateStore(BPVal, BPAddr); 8040 8041 if (Info.requiresDevicePointerInfo()) 8042 if (const ValueDecl *DevVD = BasePointers[I].getDevicePtrDecl()) 8043 Info.CaptureDeviceAddrMap.try_emplace(DevVD, BPAddr); 8044 8045 llvm::Value *PVal = Pointers[I]; 8046 llvm::Value *P = CGF.Builder.CreateConstInBoundsGEP2_32( 8047 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8048 Info.PointersArray, 0, I); 8049 P = CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 8050 P, PVal->getType()->getPointerTo(/*AddrSpace=*/0)); 8051 Address PAddr(P, Ctx.getTypeAlignInChars(Ctx.VoidPtrTy)); 8052 CGF.Builder.CreateStore(PVal, PAddr); 8053 8054 if (hasRuntimeEvaluationCaptureSize) { 8055 llvm::Value *S = CGF.Builder.CreateConstInBoundsGEP2_32( 8056 llvm::ArrayType::get(CGM.SizeTy, Info.NumberOfPtrs), 8057 Info.SizesArray, 8058 /*Idx0=*/0, 8059 /*Idx1=*/I); 8060 Address SAddr(S, Ctx.getTypeAlignInChars(Ctx.getSizeType())); 8061 CGF.Builder.CreateStore( 8062 CGF.Builder.CreateIntCast(Sizes[I], CGM.SizeTy, /*isSigned=*/true), 8063 SAddr); 8064 } 8065 } 8066 } 8067 } 8068 /// Emit the arguments to be passed to the runtime library based on the 8069 /// arrays of pointers, sizes and map types. 8070 static void emitOffloadingArraysArgument( 8071 CodeGenFunction &CGF, llvm::Value *&BasePointersArrayArg, 8072 llvm::Value *&PointersArrayArg, llvm::Value *&SizesArrayArg, 8073 llvm::Value *&MapTypesArrayArg, CGOpenMPRuntime::TargetDataInfo &Info) { 8074 CodeGenModule &CGM = CGF.CGM; 8075 if (Info.NumberOfPtrs) { 8076 BasePointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8077 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8078 Info.BasePointersArray, 8079 /*Idx0=*/0, /*Idx1=*/0); 8080 PointersArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8081 llvm::ArrayType::get(CGM.VoidPtrTy, Info.NumberOfPtrs), 8082 Info.PointersArray, 8083 /*Idx0=*/0, 8084 /*Idx1=*/0); 8085 SizesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8086 llvm::ArrayType::get(CGM.SizeTy, Info.NumberOfPtrs), Info.SizesArray, 8087 /*Idx0=*/0, /*Idx1=*/0); 8088 MapTypesArrayArg = CGF.Builder.CreateConstInBoundsGEP2_32( 8089 llvm::ArrayType::get(CGM.Int64Ty, Info.NumberOfPtrs), 8090 Info.MapTypesArray, 8091 /*Idx0=*/0, 8092 /*Idx1=*/0); 8093 } else { 8094 BasePointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 8095 PointersArrayArg = llvm::ConstantPointerNull::get(CGM.VoidPtrPtrTy); 8096 SizesArrayArg = llvm::ConstantPointerNull::get(CGM.SizeTy->getPointerTo()); 8097 MapTypesArrayArg = 8098 llvm::ConstantPointerNull::get(CGM.Int64Ty->getPointerTo()); 8099 } 8100 } 8101 8102 /// Checks if the expression is constant or does not have non-trivial function 8103 /// calls. 8104 static bool isTrivial(ASTContext &Ctx, const Expr * E) { 8105 // We can skip constant expressions. 8106 // We can skip expressions with trivial calls or simple expressions. 8107 return (E->isEvaluatable(Ctx, Expr::SE_AllowUndefinedBehavior) || 8108 !E->hasNonTrivialCall(Ctx)) && 8109 !E->HasSideEffects(Ctx, /*IncludePossibleEffects=*/true); 8110 } 8111 8112 /// Checks if the \p Body is the \a CompoundStmt and returns its child statement 8113 /// iff there is only one that is not evaluatable at the compile time. 8114 static const Stmt *getSingleCompoundChild(ASTContext &Ctx, const Stmt *Body) { 8115 if (const auto *C = dyn_cast<CompoundStmt>(Body)) { 8116 const Stmt *Child = nullptr; 8117 for (const Stmt *S : C->body()) { 8118 if (const auto *E = dyn_cast<Expr>(S)) { 8119 if (isTrivial(Ctx, E)) 8120 continue; 8121 } 8122 // Some of the statements can be ignored. 8123 if (isa<AsmStmt>(S) || isa<NullStmt>(S) || isa<OMPFlushDirective>(S) || 8124 isa<OMPBarrierDirective>(S) || isa<OMPTaskyieldDirective>(S)) 8125 continue; 8126 // Analyze declarations. 8127 if (const auto *DS = dyn_cast<DeclStmt>(S)) { 8128 if (llvm::all_of(DS->decls(), [&Ctx](const Decl *D) { 8129 if (isa<EmptyDecl>(D) || isa<DeclContext>(D) || 8130 isa<TypeDecl>(D) || isa<PragmaCommentDecl>(D) || 8131 isa<PragmaDetectMismatchDecl>(D) || isa<UsingDecl>(D) || 8132 isa<UsingDirectiveDecl>(D) || 8133 isa<OMPDeclareReductionDecl>(D) || 8134 isa<OMPThreadPrivateDecl>(D)) 8135 return true; 8136 const auto *VD = dyn_cast<VarDecl>(D); 8137 if (!VD) 8138 return false; 8139 return VD->isConstexpr() || 8140 ((VD->getType().isTrivialType(Ctx) || 8141 VD->getType()->isReferenceType()) && 8142 (!VD->hasInit() || isTrivial(Ctx, VD->getInit()))); 8143 })) 8144 continue; 8145 } 8146 // Found multiple children - cannot get the one child only. 8147 if (Child) 8148 return Body; 8149 Child = S; 8150 } 8151 if (Child) 8152 return Child; 8153 } 8154 return Body; 8155 } 8156 8157 /// Check for inner distribute directive. 8158 static const OMPExecutableDirective * 8159 getNestedDistributeDirective(ASTContext &Ctx, const OMPExecutableDirective &D) { 8160 const auto *CS = D.getInnermostCapturedStmt(); 8161 const auto *Body = 8162 CS->getCapturedStmt()->IgnoreContainers(/*IgnoreCaptured=*/true); 8163 const Stmt *ChildStmt = getSingleCompoundChild(Ctx, Body); 8164 8165 if (const auto *NestedDir = dyn_cast<OMPExecutableDirective>(ChildStmt)) { 8166 OpenMPDirectiveKind DKind = NestedDir->getDirectiveKind(); 8167 switch (D.getDirectiveKind()) { 8168 case OMPD_target: 8169 if (isOpenMPDistributeDirective(DKind)) 8170 return NestedDir; 8171 if (DKind == OMPD_teams) { 8172 Body = NestedDir->getInnermostCapturedStmt()->IgnoreContainers( 8173 /*IgnoreCaptured=*/true); 8174 if (!Body) 8175 return nullptr; 8176 ChildStmt = getSingleCompoundChild(Ctx, Body); 8177 if (const auto *NND = dyn_cast<OMPExecutableDirective>(ChildStmt)) { 8178 DKind = NND->getDirectiveKind(); 8179 if (isOpenMPDistributeDirective(DKind)) 8180 return NND; 8181 } 8182 } 8183 return nullptr; 8184 case OMPD_target_teams: 8185 if (isOpenMPDistributeDirective(DKind)) 8186 return NestedDir; 8187 return nullptr; 8188 case OMPD_target_parallel: 8189 case OMPD_target_simd: 8190 case OMPD_target_parallel_for: 8191 case OMPD_target_parallel_for_simd: 8192 return nullptr; 8193 case OMPD_target_teams_distribute: 8194 case OMPD_target_teams_distribute_simd: 8195 case OMPD_target_teams_distribute_parallel_for: 8196 case OMPD_target_teams_distribute_parallel_for_simd: 8197 case OMPD_parallel: 8198 case OMPD_for: 8199 case OMPD_parallel_for: 8200 case OMPD_parallel_sections: 8201 case OMPD_for_simd: 8202 case OMPD_parallel_for_simd: 8203 case OMPD_cancel: 8204 case OMPD_cancellation_point: 8205 case OMPD_ordered: 8206 case OMPD_threadprivate: 8207 case OMPD_task: 8208 case OMPD_simd: 8209 case OMPD_sections: 8210 case OMPD_section: 8211 case OMPD_single: 8212 case OMPD_master: 8213 case OMPD_critical: 8214 case OMPD_taskyield: 8215 case OMPD_barrier: 8216 case OMPD_taskwait: 8217 case OMPD_taskgroup: 8218 case OMPD_atomic: 8219 case OMPD_flush: 8220 case OMPD_teams: 8221 case OMPD_target_data: 8222 case OMPD_target_exit_data: 8223 case OMPD_target_enter_data: 8224 case OMPD_distribute: 8225 case OMPD_distribute_simd: 8226 case OMPD_distribute_parallel_for: 8227 case OMPD_distribute_parallel_for_simd: 8228 case OMPD_teams_distribute: 8229 case OMPD_teams_distribute_simd: 8230 case OMPD_teams_distribute_parallel_for: 8231 case OMPD_teams_distribute_parallel_for_simd: 8232 case OMPD_target_update: 8233 case OMPD_declare_simd: 8234 case OMPD_declare_target: 8235 case OMPD_end_declare_target: 8236 case OMPD_declare_reduction: 8237 case OMPD_declare_mapper: 8238 case OMPD_taskloop: 8239 case OMPD_taskloop_simd: 8240 case OMPD_requires: 8241 case OMPD_unknown: 8242 llvm_unreachable("Unexpected directive."); 8243 } 8244 } 8245 8246 return nullptr; 8247 } 8248 8249 void CGOpenMPRuntime::emitTargetNumIterationsCall( 8250 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *Device, 8251 const llvm::function_ref<llvm::Value *( 8252 CodeGenFunction &CGF, const OMPLoopDirective &D)> &SizeEmitter) { 8253 OpenMPDirectiveKind Kind = D.getDirectiveKind(); 8254 const OMPExecutableDirective *TD = &D; 8255 // Get nested teams distribute kind directive, if any. 8256 if (!isOpenMPDistributeDirective(Kind) || !isOpenMPTeamsDirective(Kind)) 8257 TD = getNestedDistributeDirective(CGM.getContext(), D); 8258 if (!TD) 8259 return; 8260 const auto *LD = cast<OMPLoopDirective>(TD); 8261 auto &&CodeGen = [LD, &Device, &SizeEmitter, this](CodeGenFunction &CGF, 8262 PrePostActionTy &) { 8263 llvm::Value *NumIterations = SizeEmitter(CGF, *LD); 8264 8265 // Emit device ID if any. 8266 llvm::Value *DeviceID; 8267 if (Device) 8268 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 8269 CGF.Int64Ty, /*isSigned=*/true); 8270 else 8271 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 8272 8273 llvm::Value *Args[] = {DeviceID, NumIterations}; 8274 CGF.EmitRuntimeCall( 8275 createRuntimeFunction(OMPRTL__kmpc_push_target_tripcount), Args); 8276 }; 8277 emitInlinedDirective(CGF, OMPD_unknown, CodeGen); 8278 } 8279 8280 void CGOpenMPRuntime::emitTargetCall(CodeGenFunction &CGF, 8281 const OMPExecutableDirective &D, 8282 llvm::Function *OutlinedFn, 8283 llvm::Value *OutlinedFnID, 8284 const Expr *IfCond, const Expr *Device) { 8285 if (!CGF.HaveInsertPoint()) 8286 return; 8287 8288 assert(OutlinedFn && "Invalid outlined function!"); 8289 8290 const bool RequiresOuterTask = D.hasClausesOfKind<OMPDependClause>(); 8291 llvm::SmallVector<llvm::Value *, 16> CapturedVars; 8292 const CapturedStmt &CS = *D.getCapturedStmt(OMPD_target); 8293 auto &&ArgsCodegen = [&CS, &CapturedVars](CodeGenFunction &CGF, 8294 PrePostActionTy &) { 8295 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 8296 }; 8297 emitInlinedDirective(CGF, OMPD_unknown, ArgsCodegen); 8298 8299 CodeGenFunction::OMPTargetDataInfo InputInfo; 8300 llvm::Value *MapTypesArray = nullptr; 8301 // Fill up the pointer arrays and transfer execution to the device. 8302 auto &&ThenGen = [this, Device, OutlinedFn, OutlinedFnID, &D, &InputInfo, 8303 &MapTypesArray, &CS, RequiresOuterTask, 8304 &CapturedVars](CodeGenFunction &CGF, PrePostActionTy &) { 8305 // On top of the arrays that were filled up, the target offloading call 8306 // takes as arguments the device id as well as the host pointer. The host 8307 // pointer is used by the runtime library to identify the current target 8308 // region, so it only has to be unique and not necessarily point to 8309 // anything. It could be the pointer to the outlined function that 8310 // implements the target region, but we aren't using that so that the 8311 // compiler doesn't need to keep that, and could therefore inline the host 8312 // function if proven worthwhile during optimization. 8313 8314 // From this point on, we need to have an ID of the target region defined. 8315 assert(OutlinedFnID && "Invalid outlined function ID!"); 8316 8317 // Emit device ID if any. 8318 llvm::Value *DeviceID; 8319 if (Device) { 8320 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 8321 CGF.Int64Ty, /*isSigned=*/true); 8322 } else { 8323 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 8324 } 8325 8326 // Emit the number of elements in the offloading arrays. 8327 llvm::Value *PointerNum = 8328 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 8329 8330 // Return value of the runtime offloading call. 8331 llvm::Value *Return; 8332 8333 llvm::Value *NumTeams = emitNumTeamsForTargetDirective(*this, CGF, D); 8334 llvm::Value *NumThreads = emitNumThreadsForTargetDirective(*this, CGF, D); 8335 8336 bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 8337 // The target region is an outlined function launched by the runtime 8338 // via calls __tgt_target() or __tgt_target_teams(). 8339 // 8340 // __tgt_target() launches a target region with one team and one thread, 8341 // executing a serial region. This master thread may in turn launch 8342 // more threads within its team upon encountering a parallel region, 8343 // however, no additional teams can be launched on the device. 8344 // 8345 // __tgt_target_teams() launches a target region with one or more teams, 8346 // each with one or more threads. This call is required for target 8347 // constructs such as: 8348 // 'target teams' 8349 // 'target' / 'teams' 8350 // 'target teams distribute parallel for' 8351 // 'target parallel' 8352 // and so on. 8353 // 8354 // Note that on the host and CPU targets, the runtime implementation of 8355 // these calls simply call the outlined function without forking threads. 8356 // The outlined functions themselves have runtime calls to 8357 // __kmpc_fork_teams() and __kmpc_fork() for this purpose, codegen'd by 8358 // the compiler in emitTeamsCall() and emitParallelCall(). 8359 // 8360 // In contrast, on the NVPTX target, the implementation of 8361 // __tgt_target_teams() launches a GPU kernel with the requested number 8362 // of teams and threads so no additional calls to the runtime are required. 8363 if (NumTeams) { 8364 // If we have NumTeams defined this means that we have an enclosed teams 8365 // region. Therefore we also expect to have NumThreads defined. These two 8366 // values should be defined in the presence of a teams directive, 8367 // regardless of having any clauses associated. If the user is using teams 8368 // but no clauses, these two values will be the default that should be 8369 // passed to the runtime library - a 32-bit integer with the value zero. 8370 assert(NumThreads && "Thread limit expression should be available along " 8371 "with number of teams."); 8372 llvm::Value *OffloadingArgs[] = {DeviceID, 8373 OutlinedFnID, 8374 PointerNum, 8375 InputInfo.BasePointersArray.getPointer(), 8376 InputInfo.PointersArray.getPointer(), 8377 InputInfo.SizesArray.getPointer(), 8378 MapTypesArray, 8379 NumTeams, 8380 NumThreads}; 8381 Return = CGF.EmitRuntimeCall( 8382 createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_teams_nowait 8383 : OMPRTL__tgt_target_teams), 8384 OffloadingArgs); 8385 } else { 8386 llvm::Value *OffloadingArgs[] = {DeviceID, 8387 OutlinedFnID, 8388 PointerNum, 8389 InputInfo.BasePointersArray.getPointer(), 8390 InputInfo.PointersArray.getPointer(), 8391 InputInfo.SizesArray.getPointer(), 8392 MapTypesArray}; 8393 Return = CGF.EmitRuntimeCall( 8394 createRuntimeFunction(HasNowait ? OMPRTL__tgt_target_nowait 8395 : OMPRTL__tgt_target), 8396 OffloadingArgs); 8397 } 8398 8399 // Check the error code and execute the host version if required. 8400 llvm::BasicBlock *OffloadFailedBlock = 8401 CGF.createBasicBlock("omp_offload.failed"); 8402 llvm::BasicBlock *OffloadContBlock = 8403 CGF.createBasicBlock("omp_offload.cont"); 8404 llvm::Value *Failed = CGF.Builder.CreateIsNotNull(Return); 8405 CGF.Builder.CreateCondBr(Failed, OffloadFailedBlock, OffloadContBlock); 8406 8407 CGF.EmitBlock(OffloadFailedBlock); 8408 if (RequiresOuterTask) { 8409 CapturedVars.clear(); 8410 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 8411 } 8412 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 8413 CGF.EmitBranch(OffloadContBlock); 8414 8415 CGF.EmitBlock(OffloadContBlock, /*IsFinished=*/true); 8416 }; 8417 8418 // Notify that the host version must be executed. 8419 auto &&ElseGen = [this, &D, OutlinedFn, &CS, &CapturedVars, 8420 RequiresOuterTask](CodeGenFunction &CGF, 8421 PrePostActionTy &) { 8422 if (RequiresOuterTask) { 8423 CapturedVars.clear(); 8424 CGF.GenerateOpenMPCapturedVars(CS, CapturedVars); 8425 } 8426 emitOutlinedFunctionCall(CGF, D.getBeginLoc(), OutlinedFn, CapturedVars); 8427 }; 8428 8429 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray, 8430 &CapturedVars, RequiresOuterTask, 8431 &CS](CodeGenFunction &CGF, PrePostActionTy &) { 8432 // Fill up the arrays with all the captured variables. 8433 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 8434 MappableExprsHandler::MapValuesArrayTy Pointers; 8435 MappableExprsHandler::MapValuesArrayTy Sizes; 8436 MappableExprsHandler::MapFlagsArrayTy MapTypes; 8437 8438 // Get mappable expression information. 8439 MappableExprsHandler MEHandler(D, CGF); 8440 llvm::DenseMap<llvm::Value *, llvm::Value *> LambdaPointers; 8441 8442 auto RI = CS.getCapturedRecordDecl()->field_begin(); 8443 auto CV = CapturedVars.begin(); 8444 for (CapturedStmt::const_capture_iterator CI = CS.capture_begin(), 8445 CE = CS.capture_end(); 8446 CI != CE; ++CI, ++RI, ++CV) { 8447 MappableExprsHandler::MapBaseValuesArrayTy CurBasePointers; 8448 MappableExprsHandler::MapValuesArrayTy CurPointers; 8449 MappableExprsHandler::MapValuesArrayTy CurSizes; 8450 MappableExprsHandler::MapFlagsArrayTy CurMapTypes; 8451 MappableExprsHandler::StructRangeInfoTy PartialStruct; 8452 8453 // VLA sizes are passed to the outlined region by copy and do not have map 8454 // information associated. 8455 if (CI->capturesVariableArrayType()) { 8456 CurBasePointers.push_back(*CV); 8457 CurPointers.push_back(*CV); 8458 CurSizes.push_back(CGF.getTypeSize(RI->getType())); 8459 // Copy to the device as an argument. No need to retrieve it. 8460 CurMapTypes.push_back(MappableExprsHandler::OMP_MAP_LITERAL | 8461 MappableExprsHandler::OMP_MAP_TARGET_PARAM); 8462 } else { 8463 // If we have any information in the map clause, we use it, otherwise we 8464 // just do a default mapping. 8465 MEHandler.generateInfoForCapture(CI, *CV, CurBasePointers, CurPointers, 8466 CurSizes, CurMapTypes, PartialStruct); 8467 if (CurBasePointers.empty()) 8468 MEHandler.generateDefaultMapInfo(*CI, **RI, *CV, CurBasePointers, 8469 CurPointers, CurSizes, CurMapTypes); 8470 // Generate correct mapping for variables captured by reference in 8471 // lambdas. 8472 if (CI->capturesVariable()) 8473 MEHandler.generateInfoForLambdaCaptures( 8474 CI->getCapturedVar(), *CV, CurBasePointers, CurPointers, CurSizes, 8475 CurMapTypes, LambdaPointers); 8476 } 8477 // We expect to have at least an element of information for this capture. 8478 assert(!CurBasePointers.empty() && 8479 "Non-existing map pointer for capture!"); 8480 assert(CurBasePointers.size() == CurPointers.size() && 8481 CurBasePointers.size() == CurSizes.size() && 8482 CurBasePointers.size() == CurMapTypes.size() && 8483 "Inconsistent map information sizes!"); 8484 8485 // If there is an entry in PartialStruct it means we have a struct with 8486 // individual members mapped. Emit an extra combined entry. 8487 if (PartialStruct.Base.isValid()) 8488 MEHandler.emitCombinedEntry(BasePointers, Pointers, Sizes, MapTypes, 8489 CurMapTypes, PartialStruct); 8490 8491 // We need to append the results of this capture to what we already have. 8492 BasePointers.append(CurBasePointers.begin(), CurBasePointers.end()); 8493 Pointers.append(CurPointers.begin(), CurPointers.end()); 8494 Sizes.append(CurSizes.begin(), CurSizes.end()); 8495 MapTypes.append(CurMapTypes.begin(), CurMapTypes.end()); 8496 } 8497 // Adjust MEMBER_OF flags for the lambdas captures. 8498 MEHandler.adjustMemberOfForLambdaCaptures(LambdaPointers, BasePointers, 8499 Pointers, MapTypes); 8500 // Map other list items in the map clause which are not captured variables 8501 // but "declare target link" global variables. 8502 MEHandler.generateInfoForDeclareTargetLink(BasePointers, Pointers, Sizes, 8503 MapTypes); 8504 8505 TargetDataInfo Info; 8506 // Fill up the arrays and create the arguments. 8507 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 8508 emitOffloadingArraysArgument(CGF, Info.BasePointersArray, 8509 Info.PointersArray, Info.SizesArray, 8510 Info.MapTypesArray, Info); 8511 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 8512 InputInfo.BasePointersArray = 8513 Address(Info.BasePointersArray, CGM.getPointerAlign()); 8514 InputInfo.PointersArray = 8515 Address(Info.PointersArray, CGM.getPointerAlign()); 8516 InputInfo.SizesArray = Address(Info.SizesArray, CGM.getPointerAlign()); 8517 MapTypesArray = Info.MapTypesArray; 8518 if (RequiresOuterTask) 8519 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 8520 else 8521 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 8522 }; 8523 8524 auto &&TargetElseGen = [this, &ElseGen, &D, RequiresOuterTask]( 8525 CodeGenFunction &CGF, PrePostActionTy &) { 8526 if (RequiresOuterTask) { 8527 CodeGenFunction::OMPTargetDataInfo InputInfo; 8528 CGF.EmitOMPTargetTaskBasedDirective(D, ElseGen, InputInfo); 8529 } else { 8530 emitInlinedDirective(CGF, D.getDirectiveKind(), ElseGen); 8531 } 8532 }; 8533 8534 // If we have a target function ID it means that we need to support 8535 // offloading, otherwise, just execute on the host. We need to execute on host 8536 // regardless of the conditional in the if clause if, e.g., the user do not 8537 // specify target triples. 8538 if (OutlinedFnID) { 8539 if (IfCond) { 8540 emitOMPIfClause(CGF, IfCond, TargetThenGen, TargetElseGen); 8541 } else { 8542 RegionCodeGenTy ThenRCG(TargetThenGen); 8543 ThenRCG(CGF); 8544 } 8545 } else { 8546 RegionCodeGenTy ElseRCG(TargetElseGen); 8547 ElseRCG(CGF); 8548 } 8549 } 8550 8551 void CGOpenMPRuntime::scanForTargetRegionsFunctions(const Stmt *S, 8552 StringRef ParentName) { 8553 if (!S) 8554 return; 8555 8556 // Codegen OMP target directives that offload compute to the device. 8557 bool RequiresDeviceCodegen = 8558 isa<OMPExecutableDirective>(S) && 8559 isOpenMPTargetExecutionDirective( 8560 cast<OMPExecutableDirective>(S)->getDirectiveKind()); 8561 8562 if (RequiresDeviceCodegen) { 8563 const auto &E = *cast<OMPExecutableDirective>(S); 8564 unsigned DeviceID; 8565 unsigned FileID; 8566 unsigned Line; 8567 getTargetEntryUniqueInfo(CGM.getContext(), E.getBeginLoc(), DeviceID, 8568 FileID, Line); 8569 8570 // Is this a target region that should not be emitted as an entry point? If 8571 // so just signal we are done with this target region. 8572 if (!OffloadEntriesInfoManager.hasTargetRegionEntryInfo(DeviceID, FileID, 8573 ParentName, Line)) 8574 return; 8575 8576 switch (E.getDirectiveKind()) { 8577 case OMPD_target: 8578 CodeGenFunction::EmitOMPTargetDeviceFunction(CGM, ParentName, 8579 cast<OMPTargetDirective>(E)); 8580 break; 8581 case OMPD_target_parallel: 8582 CodeGenFunction::EmitOMPTargetParallelDeviceFunction( 8583 CGM, ParentName, cast<OMPTargetParallelDirective>(E)); 8584 break; 8585 case OMPD_target_teams: 8586 CodeGenFunction::EmitOMPTargetTeamsDeviceFunction( 8587 CGM, ParentName, cast<OMPTargetTeamsDirective>(E)); 8588 break; 8589 case OMPD_target_teams_distribute: 8590 CodeGenFunction::EmitOMPTargetTeamsDistributeDeviceFunction( 8591 CGM, ParentName, cast<OMPTargetTeamsDistributeDirective>(E)); 8592 break; 8593 case OMPD_target_teams_distribute_simd: 8594 CodeGenFunction::EmitOMPTargetTeamsDistributeSimdDeviceFunction( 8595 CGM, ParentName, cast<OMPTargetTeamsDistributeSimdDirective>(E)); 8596 break; 8597 case OMPD_target_parallel_for: 8598 CodeGenFunction::EmitOMPTargetParallelForDeviceFunction( 8599 CGM, ParentName, cast<OMPTargetParallelForDirective>(E)); 8600 break; 8601 case OMPD_target_parallel_for_simd: 8602 CodeGenFunction::EmitOMPTargetParallelForSimdDeviceFunction( 8603 CGM, ParentName, cast<OMPTargetParallelForSimdDirective>(E)); 8604 break; 8605 case OMPD_target_simd: 8606 CodeGenFunction::EmitOMPTargetSimdDeviceFunction( 8607 CGM, ParentName, cast<OMPTargetSimdDirective>(E)); 8608 break; 8609 case OMPD_target_teams_distribute_parallel_for: 8610 CodeGenFunction::EmitOMPTargetTeamsDistributeParallelForDeviceFunction( 8611 CGM, ParentName, 8612 cast<OMPTargetTeamsDistributeParallelForDirective>(E)); 8613 break; 8614 case OMPD_target_teams_distribute_parallel_for_simd: 8615 CodeGenFunction:: 8616 EmitOMPTargetTeamsDistributeParallelForSimdDeviceFunction( 8617 CGM, ParentName, 8618 cast<OMPTargetTeamsDistributeParallelForSimdDirective>(E)); 8619 break; 8620 case OMPD_parallel: 8621 case OMPD_for: 8622 case OMPD_parallel_for: 8623 case OMPD_parallel_sections: 8624 case OMPD_for_simd: 8625 case OMPD_parallel_for_simd: 8626 case OMPD_cancel: 8627 case OMPD_cancellation_point: 8628 case OMPD_ordered: 8629 case OMPD_threadprivate: 8630 case OMPD_task: 8631 case OMPD_simd: 8632 case OMPD_sections: 8633 case OMPD_section: 8634 case OMPD_single: 8635 case OMPD_master: 8636 case OMPD_critical: 8637 case OMPD_taskyield: 8638 case OMPD_barrier: 8639 case OMPD_taskwait: 8640 case OMPD_taskgroup: 8641 case OMPD_atomic: 8642 case OMPD_flush: 8643 case OMPD_teams: 8644 case OMPD_target_data: 8645 case OMPD_target_exit_data: 8646 case OMPD_target_enter_data: 8647 case OMPD_distribute: 8648 case OMPD_distribute_simd: 8649 case OMPD_distribute_parallel_for: 8650 case OMPD_distribute_parallel_for_simd: 8651 case OMPD_teams_distribute: 8652 case OMPD_teams_distribute_simd: 8653 case OMPD_teams_distribute_parallel_for: 8654 case OMPD_teams_distribute_parallel_for_simd: 8655 case OMPD_target_update: 8656 case OMPD_declare_simd: 8657 case OMPD_declare_target: 8658 case OMPD_end_declare_target: 8659 case OMPD_declare_reduction: 8660 case OMPD_declare_mapper: 8661 case OMPD_taskloop: 8662 case OMPD_taskloop_simd: 8663 case OMPD_requires: 8664 case OMPD_unknown: 8665 llvm_unreachable("Unknown target directive for OpenMP device codegen."); 8666 } 8667 return; 8668 } 8669 8670 if (const auto *E = dyn_cast<OMPExecutableDirective>(S)) { 8671 if (!E->hasAssociatedStmt() || !E->getAssociatedStmt()) 8672 return; 8673 8674 scanForTargetRegionsFunctions( 8675 E->getInnermostCapturedStmt()->getCapturedStmt(), ParentName); 8676 return; 8677 } 8678 8679 // If this is a lambda function, look into its body. 8680 if (const auto *L = dyn_cast<LambdaExpr>(S)) 8681 S = L->getBody(); 8682 8683 // Keep looking for target regions recursively. 8684 for (const Stmt *II : S->children()) 8685 scanForTargetRegionsFunctions(II, ParentName); 8686 } 8687 8688 bool CGOpenMPRuntime::emitTargetFunctions(GlobalDecl GD) { 8689 // If emitting code for the host, we do not process FD here. Instead we do 8690 // the normal code generation. 8691 if (!CGM.getLangOpts().OpenMPIsDevice) 8692 return false; 8693 8694 const ValueDecl *VD = cast<ValueDecl>(GD.getDecl()); 8695 StringRef Name = CGM.getMangledName(GD); 8696 // Try to detect target regions in the function. 8697 if (const auto *FD = dyn_cast<FunctionDecl>(VD)) 8698 scanForTargetRegionsFunctions(FD->getBody(), Name); 8699 8700 // Do not to emit function if it is not marked as declare target. 8701 return !OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD) && 8702 AlreadyEmittedTargetFunctions.count(Name) == 0; 8703 } 8704 8705 bool CGOpenMPRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 8706 if (!CGM.getLangOpts().OpenMPIsDevice) 8707 return false; 8708 8709 // Check if there are Ctors/Dtors in this declaration and look for target 8710 // regions in it. We use the complete variant to produce the kernel name 8711 // mangling. 8712 QualType RDTy = cast<VarDecl>(GD.getDecl())->getType(); 8713 if (const auto *RD = RDTy->getBaseElementTypeUnsafe()->getAsCXXRecordDecl()) { 8714 for (const CXXConstructorDecl *Ctor : RD->ctors()) { 8715 StringRef ParentName = 8716 CGM.getMangledName(GlobalDecl(Ctor, Ctor_Complete)); 8717 scanForTargetRegionsFunctions(Ctor->getBody(), ParentName); 8718 } 8719 if (const CXXDestructorDecl *Dtor = RD->getDestructor()) { 8720 StringRef ParentName = 8721 CGM.getMangledName(GlobalDecl(Dtor, Dtor_Complete)); 8722 scanForTargetRegionsFunctions(Dtor->getBody(), ParentName); 8723 } 8724 } 8725 8726 // Do not to emit variable if it is not marked as declare target. 8727 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 8728 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration( 8729 cast<VarDecl>(GD.getDecl())); 8730 if (!Res || *Res == OMPDeclareTargetDeclAttr::MT_Link) { 8731 DeferredGlobalVariables.insert(cast<VarDecl>(GD.getDecl())); 8732 return true; 8733 } 8734 return false; 8735 } 8736 8737 void CGOpenMPRuntime::registerTargetGlobalVariable(const VarDecl *VD, 8738 llvm::Constant *Addr) { 8739 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 8740 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 8741 if (!Res) { 8742 if (CGM.getLangOpts().OpenMPIsDevice) { 8743 // Register non-target variables being emitted in device code (debug info 8744 // may cause this). 8745 StringRef VarName = CGM.getMangledName(VD); 8746 EmittedNonTargetVariables.try_emplace(VarName, Addr); 8747 } 8748 return; 8749 } 8750 // Register declare target variables. 8751 OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryKind Flags; 8752 StringRef VarName; 8753 CharUnits VarSize; 8754 llvm::GlobalValue::LinkageTypes Linkage; 8755 switch (*Res) { 8756 case OMPDeclareTargetDeclAttr::MT_To: 8757 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryTo; 8758 VarName = CGM.getMangledName(VD); 8759 if (VD->hasDefinition(CGM.getContext()) != VarDecl::DeclarationOnly) { 8760 VarSize = CGM.getContext().getTypeSizeInChars(VD->getType()); 8761 assert(!VarSize.isZero() && "Expected non-zero size of the variable"); 8762 } else { 8763 VarSize = CharUnits::Zero(); 8764 } 8765 Linkage = CGM.getLLVMLinkageVarDefinition(VD, /*IsConstant=*/false); 8766 // Temp solution to prevent optimizations of the internal variables. 8767 if (CGM.getLangOpts().OpenMPIsDevice && !VD->isExternallyVisible()) { 8768 std::string RefName = getName({VarName, "ref"}); 8769 if (!CGM.GetGlobalValue(RefName)) { 8770 llvm::Constant *AddrRef = 8771 getOrCreateInternalVariable(Addr->getType(), RefName); 8772 auto *GVAddrRef = cast<llvm::GlobalVariable>(AddrRef); 8773 GVAddrRef->setConstant(/*Val=*/true); 8774 GVAddrRef->setLinkage(llvm::GlobalValue::InternalLinkage); 8775 GVAddrRef->setInitializer(Addr); 8776 CGM.addCompilerUsedGlobal(GVAddrRef); 8777 } 8778 } 8779 break; 8780 case OMPDeclareTargetDeclAttr::MT_Link: 8781 Flags = OffloadEntriesInfoManagerTy::OMPTargetGlobalVarEntryLink; 8782 if (CGM.getLangOpts().OpenMPIsDevice) { 8783 VarName = Addr->getName(); 8784 Addr = nullptr; 8785 } else { 8786 VarName = getAddrOfDeclareTargetLink(VD).getName(); 8787 Addr = cast<llvm::Constant>(getAddrOfDeclareTargetLink(VD).getPointer()); 8788 } 8789 VarSize = CGM.getPointerSize(); 8790 Linkage = llvm::GlobalValue::WeakAnyLinkage; 8791 break; 8792 } 8793 OffloadEntriesInfoManager.registerDeviceGlobalVarEntryInfo( 8794 VarName, Addr, VarSize, Flags, Linkage); 8795 } 8796 8797 bool CGOpenMPRuntime::emitTargetGlobal(GlobalDecl GD) { 8798 if (isa<FunctionDecl>(GD.getDecl()) || 8799 isa<OMPDeclareReductionDecl>(GD.getDecl())) 8800 return emitTargetFunctions(GD); 8801 8802 return emitTargetGlobalVariable(GD); 8803 } 8804 8805 void CGOpenMPRuntime::emitDeferredTargetDecls() const { 8806 for (const VarDecl *VD : DeferredGlobalVariables) { 8807 llvm::Optional<OMPDeclareTargetDeclAttr::MapTypeTy> Res = 8808 OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(VD); 8809 if (!Res) 8810 continue; 8811 if (*Res == OMPDeclareTargetDeclAttr::MT_To) { 8812 CGM.EmitGlobal(VD); 8813 } else { 8814 assert(*Res == OMPDeclareTargetDeclAttr::MT_Link && 8815 "Expected to or link clauses."); 8816 (void)CGM.getOpenMPRuntime().getAddrOfDeclareTargetLink(VD); 8817 } 8818 } 8819 } 8820 8821 void CGOpenMPRuntime::adjustTargetSpecificDataForLambdas( 8822 CodeGenFunction &CGF, const OMPExecutableDirective &D) const { 8823 assert(isOpenMPTargetExecutionDirective(D.getDirectiveKind()) && 8824 " Expected target-based directive."); 8825 } 8826 8827 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::DisableAutoDeclareTargetRAII( 8828 CodeGenModule &CGM) 8829 : CGM(CGM) { 8830 if (CGM.getLangOpts().OpenMPIsDevice) { 8831 SavedShouldMarkAsGlobal = CGM.getOpenMPRuntime().ShouldMarkAsGlobal; 8832 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = false; 8833 } 8834 } 8835 8836 CGOpenMPRuntime::DisableAutoDeclareTargetRAII::~DisableAutoDeclareTargetRAII() { 8837 if (CGM.getLangOpts().OpenMPIsDevice) 8838 CGM.getOpenMPRuntime().ShouldMarkAsGlobal = SavedShouldMarkAsGlobal; 8839 } 8840 8841 bool CGOpenMPRuntime::markAsGlobalTarget(GlobalDecl GD) { 8842 if (!CGM.getLangOpts().OpenMPIsDevice || !ShouldMarkAsGlobal) 8843 return true; 8844 8845 StringRef Name = CGM.getMangledName(GD); 8846 const auto *D = cast<FunctionDecl>(GD.getDecl()); 8847 // Do not to emit function if it is marked as declare target as it was already 8848 // emitted. 8849 if (OMPDeclareTargetDeclAttr::isDeclareTargetDeclaration(D)) { 8850 if (D->hasBody() && AlreadyEmittedTargetFunctions.count(Name) == 0) { 8851 if (auto *F = dyn_cast_or_null<llvm::Function>(CGM.GetGlobalValue(Name))) 8852 return !F->isDeclaration(); 8853 return false; 8854 } 8855 return true; 8856 } 8857 8858 return !AlreadyEmittedTargetFunctions.insert(Name).second; 8859 } 8860 8861 llvm::Function *CGOpenMPRuntime::emitRegistrationFunction() { 8862 // If we have offloading in the current module, we need to emit the entries 8863 // now and register the offloading descriptor. 8864 createOffloadEntriesAndInfoMetadata(); 8865 8866 // Create and register the offloading binary descriptors. This is the main 8867 // entity that captures all the information about offloading in the current 8868 // compilation unit. 8869 return createOffloadingBinaryDescriptorRegistration(); 8870 } 8871 8872 void CGOpenMPRuntime::emitTeamsCall(CodeGenFunction &CGF, 8873 const OMPExecutableDirective &D, 8874 SourceLocation Loc, 8875 llvm::Function *OutlinedFn, 8876 ArrayRef<llvm::Value *> CapturedVars) { 8877 if (!CGF.HaveInsertPoint()) 8878 return; 8879 8880 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 8881 CodeGenFunction::RunCleanupsScope Scope(CGF); 8882 8883 // Build call __kmpc_fork_teams(loc, n, microtask, var1, .., varn); 8884 llvm::Value *Args[] = { 8885 RTLoc, 8886 CGF.Builder.getInt32(CapturedVars.size()), // Number of captured vars 8887 CGF.Builder.CreateBitCast(OutlinedFn, getKmpc_MicroPointerTy())}; 8888 llvm::SmallVector<llvm::Value *, 16> RealArgs; 8889 RealArgs.append(std::begin(Args), std::end(Args)); 8890 RealArgs.append(CapturedVars.begin(), CapturedVars.end()); 8891 8892 llvm::FunctionCallee RTLFn = createRuntimeFunction(OMPRTL__kmpc_fork_teams); 8893 CGF.EmitRuntimeCall(RTLFn, RealArgs); 8894 } 8895 8896 void CGOpenMPRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 8897 const Expr *NumTeams, 8898 const Expr *ThreadLimit, 8899 SourceLocation Loc) { 8900 if (!CGF.HaveInsertPoint()) 8901 return; 8902 8903 llvm::Value *RTLoc = emitUpdateLocation(CGF, Loc); 8904 8905 llvm::Value *NumTeamsVal = 8906 NumTeams 8907 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(NumTeams), 8908 CGF.CGM.Int32Ty, /* isSigned = */ true) 8909 : CGF.Builder.getInt32(0); 8910 8911 llvm::Value *ThreadLimitVal = 8912 ThreadLimit 8913 ? CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(ThreadLimit), 8914 CGF.CGM.Int32Ty, /* isSigned = */ true) 8915 : CGF.Builder.getInt32(0); 8916 8917 // Build call __kmpc_push_num_teamss(&loc, global_tid, num_teams, thread_limit) 8918 llvm::Value *PushNumTeamsArgs[] = {RTLoc, getThreadID(CGF, Loc), NumTeamsVal, 8919 ThreadLimitVal}; 8920 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__kmpc_push_num_teams), 8921 PushNumTeamsArgs); 8922 } 8923 8924 void CGOpenMPRuntime::emitTargetDataCalls( 8925 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 8926 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 8927 if (!CGF.HaveInsertPoint()) 8928 return; 8929 8930 // Action used to replace the default codegen action and turn privatization 8931 // off. 8932 PrePostActionTy NoPrivAction; 8933 8934 // Generate the code for the opening of the data environment. Capture all the 8935 // arguments of the runtime call by reference because they are used in the 8936 // closing of the region. 8937 auto &&BeginThenGen = [this, &D, Device, &Info, 8938 &CodeGen](CodeGenFunction &CGF, PrePostActionTy &) { 8939 // Fill up the arrays with all the mapped variables. 8940 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 8941 MappableExprsHandler::MapValuesArrayTy Pointers; 8942 MappableExprsHandler::MapValuesArrayTy Sizes; 8943 MappableExprsHandler::MapFlagsArrayTy MapTypes; 8944 8945 // Get map clause information. 8946 MappableExprsHandler MCHandler(D, CGF); 8947 MCHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes); 8948 8949 // Fill up the arrays and create the arguments. 8950 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 8951 8952 llvm::Value *BasePointersArrayArg = nullptr; 8953 llvm::Value *PointersArrayArg = nullptr; 8954 llvm::Value *SizesArrayArg = nullptr; 8955 llvm::Value *MapTypesArrayArg = nullptr; 8956 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 8957 SizesArrayArg, MapTypesArrayArg, Info); 8958 8959 // Emit device ID if any. 8960 llvm::Value *DeviceID = nullptr; 8961 if (Device) { 8962 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 8963 CGF.Int64Ty, /*isSigned=*/true); 8964 } else { 8965 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 8966 } 8967 8968 // Emit the number of elements in the offloading arrays. 8969 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 8970 8971 llvm::Value *OffloadingArgs[] = { 8972 DeviceID, PointerNum, BasePointersArrayArg, 8973 PointersArrayArg, SizesArrayArg, MapTypesArrayArg}; 8974 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_begin), 8975 OffloadingArgs); 8976 8977 // If device pointer privatization is required, emit the body of the region 8978 // here. It will have to be duplicated: with and without privatization. 8979 if (!Info.CaptureDeviceAddrMap.empty()) 8980 CodeGen(CGF); 8981 }; 8982 8983 // Generate code for the closing of the data region. 8984 auto &&EndThenGen = [this, Device, &Info](CodeGenFunction &CGF, 8985 PrePostActionTy &) { 8986 assert(Info.isValid() && "Invalid data environment closing arguments."); 8987 8988 llvm::Value *BasePointersArrayArg = nullptr; 8989 llvm::Value *PointersArrayArg = nullptr; 8990 llvm::Value *SizesArrayArg = nullptr; 8991 llvm::Value *MapTypesArrayArg = nullptr; 8992 emitOffloadingArraysArgument(CGF, BasePointersArrayArg, PointersArrayArg, 8993 SizesArrayArg, MapTypesArrayArg, Info); 8994 8995 // Emit device ID if any. 8996 llvm::Value *DeviceID = nullptr; 8997 if (Device) { 8998 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 8999 CGF.Int64Ty, /*isSigned=*/true); 9000 } else { 9001 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 9002 } 9003 9004 // Emit the number of elements in the offloading arrays. 9005 llvm::Value *PointerNum = CGF.Builder.getInt32(Info.NumberOfPtrs); 9006 9007 llvm::Value *OffloadingArgs[] = { 9008 DeviceID, PointerNum, BasePointersArrayArg, 9009 PointersArrayArg, SizesArrayArg, MapTypesArrayArg}; 9010 CGF.EmitRuntimeCall(createRuntimeFunction(OMPRTL__tgt_target_data_end), 9011 OffloadingArgs); 9012 }; 9013 9014 // If we need device pointer privatization, we need to emit the body of the 9015 // region with no privatization in the 'else' branch of the conditional. 9016 // Otherwise, we don't have to do anything. 9017 auto &&BeginElseGen = [&Info, &CodeGen, &NoPrivAction](CodeGenFunction &CGF, 9018 PrePostActionTy &) { 9019 if (!Info.CaptureDeviceAddrMap.empty()) { 9020 CodeGen.setAction(NoPrivAction); 9021 CodeGen(CGF); 9022 } 9023 }; 9024 9025 // We don't have to do anything to close the region if the if clause evaluates 9026 // to false. 9027 auto &&EndElseGen = [](CodeGenFunction &CGF, PrePostActionTy &) {}; 9028 9029 if (IfCond) { 9030 emitOMPIfClause(CGF, IfCond, BeginThenGen, BeginElseGen); 9031 } else { 9032 RegionCodeGenTy RCG(BeginThenGen); 9033 RCG(CGF); 9034 } 9035 9036 // If we don't require privatization of device pointers, we emit the body in 9037 // between the runtime calls. This avoids duplicating the body code. 9038 if (Info.CaptureDeviceAddrMap.empty()) { 9039 CodeGen.setAction(NoPrivAction); 9040 CodeGen(CGF); 9041 } 9042 9043 if (IfCond) { 9044 emitOMPIfClause(CGF, IfCond, EndThenGen, EndElseGen); 9045 } else { 9046 RegionCodeGenTy RCG(EndThenGen); 9047 RCG(CGF); 9048 } 9049 } 9050 9051 void CGOpenMPRuntime::emitTargetDataStandAloneCall( 9052 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 9053 const Expr *Device) { 9054 if (!CGF.HaveInsertPoint()) 9055 return; 9056 9057 assert((isa<OMPTargetEnterDataDirective>(D) || 9058 isa<OMPTargetExitDataDirective>(D) || 9059 isa<OMPTargetUpdateDirective>(D)) && 9060 "Expecting either target enter, exit data, or update directives."); 9061 9062 CodeGenFunction::OMPTargetDataInfo InputInfo; 9063 llvm::Value *MapTypesArray = nullptr; 9064 // Generate the code for the opening of the data environment. 9065 auto &&ThenGen = [this, &D, Device, &InputInfo, 9066 &MapTypesArray](CodeGenFunction &CGF, PrePostActionTy &) { 9067 // Emit device ID if any. 9068 llvm::Value *DeviceID = nullptr; 9069 if (Device) { 9070 DeviceID = CGF.Builder.CreateIntCast(CGF.EmitScalarExpr(Device), 9071 CGF.Int64Ty, /*isSigned=*/true); 9072 } else { 9073 DeviceID = CGF.Builder.getInt64(OMP_DEVICEID_UNDEF); 9074 } 9075 9076 // Emit the number of elements in the offloading arrays. 9077 llvm::Constant *PointerNum = 9078 CGF.Builder.getInt32(InputInfo.NumberOfTargetItems); 9079 9080 llvm::Value *OffloadingArgs[] = {DeviceID, 9081 PointerNum, 9082 InputInfo.BasePointersArray.getPointer(), 9083 InputInfo.PointersArray.getPointer(), 9084 InputInfo.SizesArray.getPointer(), 9085 MapTypesArray}; 9086 9087 // Select the right runtime function call for each expected standalone 9088 // directive. 9089 const bool HasNowait = D.hasClausesOfKind<OMPNowaitClause>(); 9090 OpenMPRTLFunction RTLFn; 9091 switch (D.getDirectiveKind()) { 9092 case OMPD_target_enter_data: 9093 RTLFn = HasNowait ? OMPRTL__tgt_target_data_begin_nowait 9094 : OMPRTL__tgt_target_data_begin; 9095 break; 9096 case OMPD_target_exit_data: 9097 RTLFn = HasNowait ? OMPRTL__tgt_target_data_end_nowait 9098 : OMPRTL__tgt_target_data_end; 9099 break; 9100 case OMPD_target_update: 9101 RTLFn = HasNowait ? OMPRTL__tgt_target_data_update_nowait 9102 : OMPRTL__tgt_target_data_update; 9103 break; 9104 case OMPD_parallel: 9105 case OMPD_for: 9106 case OMPD_parallel_for: 9107 case OMPD_parallel_sections: 9108 case OMPD_for_simd: 9109 case OMPD_parallel_for_simd: 9110 case OMPD_cancel: 9111 case OMPD_cancellation_point: 9112 case OMPD_ordered: 9113 case OMPD_threadprivate: 9114 case OMPD_task: 9115 case OMPD_simd: 9116 case OMPD_sections: 9117 case OMPD_section: 9118 case OMPD_single: 9119 case OMPD_master: 9120 case OMPD_critical: 9121 case OMPD_taskyield: 9122 case OMPD_barrier: 9123 case OMPD_taskwait: 9124 case OMPD_taskgroup: 9125 case OMPD_atomic: 9126 case OMPD_flush: 9127 case OMPD_teams: 9128 case OMPD_target_data: 9129 case OMPD_distribute: 9130 case OMPD_distribute_simd: 9131 case OMPD_distribute_parallel_for: 9132 case OMPD_distribute_parallel_for_simd: 9133 case OMPD_teams_distribute: 9134 case OMPD_teams_distribute_simd: 9135 case OMPD_teams_distribute_parallel_for: 9136 case OMPD_teams_distribute_parallel_for_simd: 9137 case OMPD_declare_simd: 9138 case OMPD_declare_target: 9139 case OMPD_end_declare_target: 9140 case OMPD_declare_reduction: 9141 case OMPD_declare_mapper: 9142 case OMPD_taskloop: 9143 case OMPD_taskloop_simd: 9144 case OMPD_target: 9145 case OMPD_target_simd: 9146 case OMPD_target_teams_distribute: 9147 case OMPD_target_teams_distribute_simd: 9148 case OMPD_target_teams_distribute_parallel_for: 9149 case OMPD_target_teams_distribute_parallel_for_simd: 9150 case OMPD_target_teams: 9151 case OMPD_target_parallel: 9152 case OMPD_target_parallel_for: 9153 case OMPD_target_parallel_for_simd: 9154 case OMPD_requires: 9155 case OMPD_unknown: 9156 llvm_unreachable("Unexpected standalone target data directive."); 9157 break; 9158 } 9159 CGF.EmitRuntimeCall(createRuntimeFunction(RTLFn), OffloadingArgs); 9160 }; 9161 9162 auto &&TargetThenGen = [this, &ThenGen, &D, &InputInfo, &MapTypesArray]( 9163 CodeGenFunction &CGF, PrePostActionTy &) { 9164 // Fill up the arrays with all the mapped variables. 9165 MappableExprsHandler::MapBaseValuesArrayTy BasePointers; 9166 MappableExprsHandler::MapValuesArrayTy Pointers; 9167 MappableExprsHandler::MapValuesArrayTy Sizes; 9168 MappableExprsHandler::MapFlagsArrayTy MapTypes; 9169 9170 // Get map clause information. 9171 MappableExprsHandler MEHandler(D, CGF); 9172 MEHandler.generateAllInfo(BasePointers, Pointers, Sizes, MapTypes); 9173 9174 TargetDataInfo Info; 9175 // Fill up the arrays and create the arguments. 9176 emitOffloadingArrays(CGF, BasePointers, Pointers, Sizes, MapTypes, Info); 9177 emitOffloadingArraysArgument(CGF, Info.BasePointersArray, 9178 Info.PointersArray, Info.SizesArray, 9179 Info.MapTypesArray, Info); 9180 InputInfo.NumberOfTargetItems = Info.NumberOfPtrs; 9181 InputInfo.BasePointersArray = 9182 Address(Info.BasePointersArray, CGM.getPointerAlign()); 9183 InputInfo.PointersArray = 9184 Address(Info.PointersArray, CGM.getPointerAlign()); 9185 InputInfo.SizesArray = 9186 Address(Info.SizesArray, CGM.getPointerAlign()); 9187 MapTypesArray = Info.MapTypesArray; 9188 if (D.hasClausesOfKind<OMPDependClause>()) 9189 CGF.EmitOMPTargetTaskBasedDirective(D, ThenGen, InputInfo); 9190 else 9191 emitInlinedDirective(CGF, D.getDirectiveKind(), ThenGen); 9192 }; 9193 9194 if (IfCond) { 9195 emitOMPIfClause(CGF, IfCond, TargetThenGen, 9196 [](CodeGenFunction &CGF, PrePostActionTy &) {}); 9197 } else { 9198 RegionCodeGenTy ThenRCG(TargetThenGen); 9199 ThenRCG(CGF); 9200 } 9201 } 9202 9203 namespace { 9204 /// Kind of parameter in a function with 'declare simd' directive. 9205 enum ParamKindTy { LinearWithVarStride, Linear, Uniform, Vector }; 9206 /// Attribute set of the parameter. 9207 struct ParamAttrTy { 9208 ParamKindTy Kind = Vector; 9209 llvm::APSInt StrideOrArg; 9210 llvm::APSInt Alignment; 9211 }; 9212 } // namespace 9213 9214 static unsigned evaluateCDTSize(const FunctionDecl *FD, 9215 ArrayRef<ParamAttrTy> ParamAttrs) { 9216 // Every vector variant of a SIMD-enabled function has a vector length (VLEN). 9217 // If OpenMP clause "simdlen" is used, the VLEN is the value of the argument 9218 // of that clause. The VLEN value must be power of 2. 9219 // In other case the notion of the function`s "characteristic data type" (CDT) 9220 // is used to compute the vector length. 9221 // CDT is defined in the following order: 9222 // a) For non-void function, the CDT is the return type. 9223 // b) If the function has any non-uniform, non-linear parameters, then the 9224 // CDT is the type of the first such parameter. 9225 // c) If the CDT determined by a) or b) above is struct, union, or class 9226 // type which is pass-by-value (except for the type that maps to the 9227 // built-in complex data type), the characteristic data type is int. 9228 // d) If none of the above three cases is applicable, the CDT is int. 9229 // The VLEN is then determined based on the CDT and the size of vector 9230 // register of that ISA for which current vector version is generated. The 9231 // VLEN is computed using the formula below: 9232 // VLEN = sizeof(vector_register) / sizeof(CDT), 9233 // where vector register size specified in section 3.2.1 Registers and the 9234 // Stack Frame of original AMD64 ABI document. 9235 QualType RetType = FD->getReturnType(); 9236 if (RetType.isNull()) 9237 return 0; 9238 ASTContext &C = FD->getASTContext(); 9239 QualType CDT; 9240 if (!RetType.isNull() && !RetType->isVoidType()) { 9241 CDT = RetType; 9242 } else { 9243 unsigned Offset = 0; 9244 if (const auto *MD = dyn_cast<CXXMethodDecl>(FD)) { 9245 if (ParamAttrs[Offset].Kind == Vector) 9246 CDT = C.getPointerType(C.getRecordType(MD->getParent())); 9247 ++Offset; 9248 } 9249 if (CDT.isNull()) { 9250 for (unsigned I = 0, E = FD->getNumParams(); I < E; ++I) { 9251 if (ParamAttrs[I + Offset].Kind == Vector) { 9252 CDT = FD->getParamDecl(I)->getType(); 9253 break; 9254 } 9255 } 9256 } 9257 } 9258 if (CDT.isNull()) 9259 CDT = C.IntTy; 9260 CDT = CDT->getCanonicalTypeUnqualified(); 9261 if (CDT->isRecordType() || CDT->isUnionType()) 9262 CDT = C.IntTy; 9263 return C.getTypeSize(CDT); 9264 } 9265 9266 static void 9267 emitX86DeclareSimdFunction(const FunctionDecl *FD, llvm::Function *Fn, 9268 const llvm::APSInt &VLENVal, 9269 ArrayRef<ParamAttrTy> ParamAttrs, 9270 OMPDeclareSimdDeclAttr::BranchStateTy State) { 9271 struct ISADataTy { 9272 char ISA; 9273 unsigned VecRegSize; 9274 }; 9275 ISADataTy ISAData[] = { 9276 { 9277 'b', 128 9278 }, // SSE 9279 { 9280 'c', 256 9281 }, // AVX 9282 { 9283 'd', 256 9284 }, // AVX2 9285 { 9286 'e', 512 9287 }, // AVX512 9288 }; 9289 llvm::SmallVector<char, 2> Masked; 9290 switch (State) { 9291 case OMPDeclareSimdDeclAttr::BS_Undefined: 9292 Masked.push_back('N'); 9293 Masked.push_back('M'); 9294 break; 9295 case OMPDeclareSimdDeclAttr::BS_Notinbranch: 9296 Masked.push_back('N'); 9297 break; 9298 case OMPDeclareSimdDeclAttr::BS_Inbranch: 9299 Masked.push_back('M'); 9300 break; 9301 } 9302 for (char Mask : Masked) { 9303 for (const ISADataTy &Data : ISAData) { 9304 SmallString<256> Buffer; 9305 llvm::raw_svector_ostream Out(Buffer); 9306 Out << "_ZGV" << Data.ISA << Mask; 9307 if (!VLENVal) { 9308 Out << llvm::APSInt::getUnsigned(Data.VecRegSize / 9309 evaluateCDTSize(FD, ParamAttrs)); 9310 } else { 9311 Out << VLENVal; 9312 } 9313 for (const ParamAttrTy &ParamAttr : ParamAttrs) { 9314 switch (ParamAttr.Kind){ 9315 case LinearWithVarStride: 9316 Out << 's' << ParamAttr.StrideOrArg; 9317 break; 9318 case Linear: 9319 Out << 'l'; 9320 if (!!ParamAttr.StrideOrArg) 9321 Out << ParamAttr.StrideOrArg; 9322 break; 9323 case Uniform: 9324 Out << 'u'; 9325 break; 9326 case Vector: 9327 Out << 'v'; 9328 break; 9329 } 9330 if (!!ParamAttr.Alignment) 9331 Out << 'a' << ParamAttr.Alignment; 9332 } 9333 Out << '_' << Fn->getName(); 9334 Fn->addFnAttr(Out.str()); 9335 } 9336 } 9337 } 9338 9339 void CGOpenMPRuntime::emitDeclareSimdFunction(const FunctionDecl *FD, 9340 llvm::Function *Fn) { 9341 ASTContext &C = CGM.getContext(); 9342 FD = FD->getMostRecentDecl(); 9343 // Map params to their positions in function decl. 9344 llvm::DenseMap<const Decl *, unsigned> ParamPositions; 9345 if (isa<CXXMethodDecl>(FD)) 9346 ParamPositions.try_emplace(FD, 0); 9347 unsigned ParamPos = ParamPositions.size(); 9348 for (const ParmVarDecl *P : FD->parameters()) { 9349 ParamPositions.try_emplace(P->getCanonicalDecl(), ParamPos); 9350 ++ParamPos; 9351 } 9352 while (FD) { 9353 for (const auto *Attr : FD->specific_attrs<OMPDeclareSimdDeclAttr>()) { 9354 llvm::SmallVector<ParamAttrTy, 8> ParamAttrs(ParamPositions.size()); 9355 // Mark uniform parameters. 9356 for (const Expr *E : Attr->uniforms()) { 9357 E = E->IgnoreParenImpCasts(); 9358 unsigned Pos; 9359 if (isa<CXXThisExpr>(E)) { 9360 Pos = ParamPositions[FD]; 9361 } else { 9362 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 9363 ->getCanonicalDecl(); 9364 Pos = ParamPositions[PVD]; 9365 } 9366 ParamAttrs[Pos].Kind = Uniform; 9367 } 9368 // Get alignment info. 9369 auto NI = Attr->alignments_begin(); 9370 for (const Expr *E : Attr->aligneds()) { 9371 E = E->IgnoreParenImpCasts(); 9372 unsigned Pos; 9373 QualType ParmTy; 9374 if (isa<CXXThisExpr>(E)) { 9375 Pos = ParamPositions[FD]; 9376 ParmTy = E->getType(); 9377 } else { 9378 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 9379 ->getCanonicalDecl(); 9380 Pos = ParamPositions[PVD]; 9381 ParmTy = PVD->getType(); 9382 } 9383 ParamAttrs[Pos].Alignment = 9384 (*NI) 9385 ? (*NI)->EvaluateKnownConstInt(C) 9386 : llvm::APSInt::getUnsigned( 9387 C.toCharUnitsFromBits(C.getOpenMPDefaultSimdAlign(ParmTy)) 9388 .getQuantity()); 9389 ++NI; 9390 } 9391 // Mark linear parameters. 9392 auto SI = Attr->steps_begin(); 9393 auto MI = Attr->modifiers_begin(); 9394 for (const Expr *E : Attr->linears()) { 9395 E = E->IgnoreParenImpCasts(); 9396 unsigned Pos; 9397 if (isa<CXXThisExpr>(E)) { 9398 Pos = ParamPositions[FD]; 9399 } else { 9400 const auto *PVD = cast<ParmVarDecl>(cast<DeclRefExpr>(E)->getDecl()) 9401 ->getCanonicalDecl(); 9402 Pos = ParamPositions[PVD]; 9403 } 9404 ParamAttrTy &ParamAttr = ParamAttrs[Pos]; 9405 ParamAttr.Kind = Linear; 9406 if (*SI) { 9407 Expr::EvalResult Result; 9408 if (!(*SI)->EvaluateAsInt(Result, C, Expr::SE_AllowSideEffects)) { 9409 if (const auto *DRE = 9410 cast<DeclRefExpr>((*SI)->IgnoreParenImpCasts())) { 9411 if (const auto *StridePVD = cast<ParmVarDecl>(DRE->getDecl())) { 9412 ParamAttr.Kind = LinearWithVarStride; 9413 ParamAttr.StrideOrArg = llvm::APSInt::getUnsigned( 9414 ParamPositions[StridePVD->getCanonicalDecl()]); 9415 } 9416 } 9417 } else { 9418 ParamAttr.StrideOrArg = Result.Val.getInt(); 9419 } 9420 } 9421 ++SI; 9422 ++MI; 9423 } 9424 llvm::APSInt VLENVal; 9425 if (const Expr *VLEN = Attr->getSimdlen()) 9426 VLENVal = VLEN->EvaluateKnownConstInt(C); 9427 OMPDeclareSimdDeclAttr::BranchStateTy State = Attr->getBranchState(); 9428 if (CGM.getTriple().getArch() == llvm::Triple::x86 || 9429 CGM.getTriple().getArch() == llvm::Triple::x86_64) 9430 emitX86DeclareSimdFunction(FD, Fn, VLENVal, ParamAttrs, State); 9431 } 9432 FD = FD->getPreviousDecl(); 9433 } 9434 } 9435 9436 namespace { 9437 /// Cleanup action for doacross support. 9438 class DoacrossCleanupTy final : public EHScopeStack::Cleanup { 9439 public: 9440 static const int DoacrossFinArgs = 2; 9441 9442 private: 9443 llvm::FunctionCallee RTLFn; 9444 llvm::Value *Args[DoacrossFinArgs]; 9445 9446 public: 9447 DoacrossCleanupTy(llvm::FunctionCallee RTLFn, 9448 ArrayRef<llvm::Value *> CallArgs) 9449 : RTLFn(RTLFn) { 9450 assert(CallArgs.size() == DoacrossFinArgs); 9451 std::copy(CallArgs.begin(), CallArgs.end(), std::begin(Args)); 9452 } 9453 void Emit(CodeGenFunction &CGF, Flags /*flags*/) override { 9454 if (!CGF.HaveInsertPoint()) 9455 return; 9456 CGF.EmitRuntimeCall(RTLFn, Args); 9457 } 9458 }; 9459 } // namespace 9460 9461 void CGOpenMPRuntime::emitDoacrossInit(CodeGenFunction &CGF, 9462 const OMPLoopDirective &D, 9463 ArrayRef<Expr *> NumIterations) { 9464 if (!CGF.HaveInsertPoint()) 9465 return; 9466 9467 ASTContext &C = CGM.getContext(); 9468 QualType Int64Ty = C.getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/true); 9469 RecordDecl *RD; 9470 if (KmpDimTy.isNull()) { 9471 // Build struct kmp_dim { // loop bounds info casted to kmp_int64 9472 // kmp_int64 lo; // lower 9473 // kmp_int64 up; // upper 9474 // kmp_int64 st; // stride 9475 // }; 9476 RD = C.buildImplicitRecord("kmp_dim"); 9477 RD->startDefinition(); 9478 addFieldToRecordDecl(C, RD, Int64Ty); 9479 addFieldToRecordDecl(C, RD, Int64Ty); 9480 addFieldToRecordDecl(C, RD, Int64Ty); 9481 RD->completeDefinition(); 9482 KmpDimTy = C.getRecordType(RD); 9483 } else { 9484 RD = cast<RecordDecl>(KmpDimTy->getAsTagDecl()); 9485 } 9486 llvm::APInt Size(/*numBits=*/32, NumIterations.size()); 9487 QualType ArrayTy = 9488 C.getConstantArrayType(KmpDimTy, Size, ArrayType::Normal, 0); 9489 9490 Address DimsAddr = CGF.CreateMemTemp(ArrayTy, "dims"); 9491 CGF.EmitNullInitialization(DimsAddr, ArrayTy); 9492 enum { LowerFD = 0, UpperFD, StrideFD }; 9493 // Fill dims with data. 9494 for (unsigned I = 0, E = NumIterations.size(); I < E; ++I) { 9495 LValue DimsLVal = 9496 CGF.MakeAddrLValue(CGF.Builder.CreateConstArrayGEP( 9497 DimsAddr, I, C.getTypeSizeInChars(KmpDimTy)), 9498 KmpDimTy); 9499 // dims.upper = num_iterations; 9500 LValue UpperLVal = CGF.EmitLValueForField( 9501 DimsLVal, *std::next(RD->field_begin(), UpperFD)); 9502 llvm::Value *NumIterVal = 9503 CGF.EmitScalarConversion(CGF.EmitScalarExpr(NumIterations[I]), 9504 D.getNumIterations()->getType(), Int64Ty, 9505 D.getNumIterations()->getExprLoc()); 9506 CGF.EmitStoreOfScalar(NumIterVal, UpperLVal); 9507 // dims.stride = 1; 9508 LValue StrideLVal = CGF.EmitLValueForField( 9509 DimsLVal, *std::next(RD->field_begin(), StrideFD)); 9510 CGF.EmitStoreOfScalar(llvm::ConstantInt::getSigned(CGM.Int64Ty, /*V=*/1), 9511 StrideLVal); 9512 } 9513 9514 // Build call void __kmpc_doacross_init(ident_t *loc, kmp_int32 gtid, 9515 // kmp_int32 num_dims, struct kmp_dim * dims); 9516 llvm::Value *Args[] = { 9517 emitUpdateLocation(CGF, D.getBeginLoc()), 9518 getThreadID(CGF, D.getBeginLoc()), 9519 llvm::ConstantInt::getSigned(CGM.Int32Ty, NumIterations.size()), 9520 CGF.Builder.CreatePointerBitCastOrAddrSpaceCast( 9521 CGF.Builder 9522 .CreateConstArrayGEP(DimsAddr, 0, C.getTypeSizeInChars(KmpDimTy)) 9523 .getPointer(), 9524 CGM.VoidPtrTy)}; 9525 9526 llvm::FunctionCallee RTLFn = 9527 createRuntimeFunction(OMPRTL__kmpc_doacross_init); 9528 CGF.EmitRuntimeCall(RTLFn, Args); 9529 llvm::Value *FiniArgs[DoacrossCleanupTy::DoacrossFinArgs] = { 9530 emitUpdateLocation(CGF, D.getEndLoc()), getThreadID(CGF, D.getEndLoc())}; 9531 llvm::FunctionCallee FiniRTLFn = 9532 createRuntimeFunction(OMPRTL__kmpc_doacross_fini); 9533 CGF.EHStack.pushCleanup<DoacrossCleanupTy>(NormalAndEHCleanup, FiniRTLFn, 9534 llvm::makeArrayRef(FiniArgs)); 9535 } 9536 9537 void CGOpenMPRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 9538 const OMPDependClause *C) { 9539 QualType Int64Ty = 9540 CGM.getContext().getIntTypeForBitwidth(/*DestWidth=*/64, /*Signed=*/1); 9541 llvm::APInt Size(/*numBits=*/32, C->getNumLoops()); 9542 QualType ArrayTy = CGM.getContext().getConstantArrayType( 9543 Int64Ty, Size, ArrayType::Normal, 0); 9544 Address CntAddr = CGF.CreateMemTemp(ArrayTy, ".cnt.addr"); 9545 for (unsigned I = 0, E = C->getNumLoops(); I < E; ++I) { 9546 const Expr *CounterVal = C->getLoopData(I); 9547 assert(CounterVal); 9548 llvm::Value *CntVal = CGF.EmitScalarConversion( 9549 CGF.EmitScalarExpr(CounterVal), CounterVal->getType(), Int64Ty, 9550 CounterVal->getExprLoc()); 9551 CGF.EmitStoreOfScalar( 9552 CntVal, 9553 CGF.Builder.CreateConstArrayGEP( 9554 CntAddr, I, CGM.getContext().getTypeSizeInChars(Int64Ty)), 9555 /*Volatile=*/false, Int64Ty); 9556 } 9557 llvm::Value *Args[] = { 9558 emitUpdateLocation(CGF, C->getBeginLoc()), 9559 getThreadID(CGF, C->getBeginLoc()), 9560 CGF.Builder 9561 .CreateConstArrayGEP(CntAddr, 0, 9562 CGM.getContext().getTypeSizeInChars(Int64Ty)) 9563 .getPointer()}; 9564 llvm::FunctionCallee RTLFn; 9565 if (C->getDependencyKind() == OMPC_DEPEND_source) { 9566 RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_post); 9567 } else { 9568 assert(C->getDependencyKind() == OMPC_DEPEND_sink); 9569 RTLFn = createRuntimeFunction(OMPRTL__kmpc_doacross_wait); 9570 } 9571 CGF.EmitRuntimeCall(RTLFn, Args); 9572 } 9573 9574 void CGOpenMPRuntime::emitCall(CodeGenFunction &CGF, SourceLocation Loc, 9575 llvm::FunctionCallee Callee, 9576 ArrayRef<llvm::Value *> Args) const { 9577 assert(Loc.isValid() && "Outlined function call location must be valid."); 9578 auto DL = ApplyDebugLocation::CreateDefaultArtificial(CGF, Loc); 9579 9580 if (auto *Fn = dyn_cast<llvm::Function>(Callee.getCallee())) { 9581 if (Fn->doesNotThrow()) { 9582 CGF.EmitNounwindRuntimeCall(Fn, Args); 9583 return; 9584 } 9585 } 9586 CGF.EmitRuntimeCall(Callee, Args); 9587 } 9588 9589 void CGOpenMPRuntime::emitOutlinedFunctionCall( 9590 CodeGenFunction &CGF, SourceLocation Loc, llvm::FunctionCallee OutlinedFn, 9591 ArrayRef<llvm::Value *> Args) const { 9592 emitCall(CGF, Loc, OutlinedFn, Args); 9593 } 9594 9595 Address CGOpenMPRuntime::getParameterAddress(CodeGenFunction &CGF, 9596 const VarDecl *NativeParam, 9597 const VarDecl *TargetParam) const { 9598 return CGF.GetAddrOfLocalVar(NativeParam); 9599 } 9600 9601 Address CGOpenMPRuntime::getAddressOfLocalVariable(CodeGenFunction &CGF, 9602 const VarDecl *VD) { 9603 return Address::invalid(); 9604 } 9605 9606 llvm::Function *CGOpenMPSIMDRuntime::emitParallelOutlinedFunction( 9607 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 9608 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 9609 llvm_unreachable("Not supported in SIMD-only mode"); 9610 } 9611 9612 llvm::Function *CGOpenMPSIMDRuntime::emitTeamsOutlinedFunction( 9613 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 9614 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen) { 9615 llvm_unreachable("Not supported in SIMD-only mode"); 9616 } 9617 9618 llvm::Function *CGOpenMPSIMDRuntime::emitTaskOutlinedFunction( 9619 const OMPExecutableDirective &D, const VarDecl *ThreadIDVar, 9620 const VarDecl *PartIDVar, const VarDecl *TaskTVar, 9621 OpenMPDirectiveKind InnermostKind, const RegionCodeGenTy &CodeGen, 9622 bool Tied, unsigned &NumberOfParts) { 9623 llvm_unreachable("Not supported in SIMD-only mode"); 9624 } 9625 9626 void CGOpenMPSIMDRuntime::emitParallelCall(CodeGenFunction &CGF, 9627 SourceLocation Loc, 9628 llvm::Function *OutlinedFn, 9629 ArrayRef<llvm::Value *> CapturedVars, 9630 const Expr *IfCond) { 9631 llvm_unreachable("Not supported in SIMD-only mode"); 9632 } 9633 9634 void CGOpenMPSIMDRuntime::emitCriticalRegion( 9635 CodeGenFunction &CGF, StringRef CriticalName, 9636 const RegionCodeGenTy &CriticalOpGen, SourceLocation Loc, 9637 const Expr *Hint) { 9638 llvm_unreachable("Not supported in SIMD-only mode"); 9639 } 9640 9641 void CGOpenMPSIMDRuntime::emitMasterRegion(CodeGenFunction &CGF, 9642 const RegionCodeGenTy &MasterOpGen, 9643 SourceLocation Loc) { 9644 llvm_unreachable("Not supported in SIMD-only mode"); 9645 } 9646 9647 void CGOpenMPSIMDRuntime::emitTaskyieldCall(CodeGenFunction &CGF, 9648 SourceLocation Loc) { 9649 llvm_unreachable("Not supported in SIMD-only mode"); 9650 } 9651 9652 void CGOpenMPSIMDRuntime::emitTaskgroupRegion( 9653 CodeGenFunction &CGF, const RegionCodeGenTy &TaskgroupOpGen, 9654 SourceLocation Loc) { 9655 llvm_unreachable("Not supported in SIMD-only mode"); 9656 } 9657 9658 void CGOpenMPSIMDRuntime::emitSingleRegion( 9659 CodeGenFunction &CGF, const RegionCodeGenTy &SingleOpGen, 9660 SourceLocation Loc, ArrayRef<const Expr *> CopyprivateVars, 9661 ArrayRef<const Expr *> DestExprs, ArrayRef<const Expr *> SrcExprs, 9662 ArrayRef<const Expr *> AssignmentOps) { 9663 llvm_unreachable("Not supported in SIMD-only mode"); 9664 } 9665 9666 void CGOpenMPSIMDRuntime::emitOrderedRegion(CodeGenFunction &CGF, 9667 const RegionCodeGenTy &OrderedOpGen, 9668 SourceLocation Loc, 9669 bool IsThreads) { 9670 llvm_unreachable("Not supported in SIMD-only mode"); 9671 } 9672 9673 void CGOpenMPSIMDRuntime::emitBarrierCall(CodeGenFunction &CGF, 9674 SourceLocation Loc, 9675 OpenMPDirectiveKind Kind, 9676 bool EmitChecks, 9677 bool ForceSimpleCall) { 9678 llvm_unreachable("Not supported in SIMD-only mode"); 9679 } 9680 9681 void CGOpenMPSIMDRuntime::emitForDispatchInit( 9682 CodeGenFunction &CGF, SourceLocation Loc, 9683 const OpenMPScheduleTy &ScheduleKind, unsigned IVSize, bool IVSigned, 9684 bool Ordered, const DispatchRTInput &DispatchValues) { 9685 llvm_unreachable("Not supported in SIMD-only mode"); 9686 } 9687 9688 void CGOpenMPSIMDRuntime::emitForStaticInit( 9689 CodeGenFunction &CGF, SourceLocation Loc, OpenMPDirectiveKind DKind, 9690 const OpenMPScheduleTy &ScheduleKind, const StaticRTInput &Values) { 9691 llvm_unreachable("Not supported in SIMD-only mode"); 9692 } 9693 9694 void CGOpenMPSIMDRuntime::emitDistributeStaticInit( 9695 CodeGenFunction &CGF, SourceLocation Loc, 9696 OpenMPDistScheduleClauseKind SchedKind, const StaticRTInput &Values) { 9697 llvm_unreachable("Not supported in SIMD-only mode"); 9698 } 9699 9700 void CGOpenMPSIMDRuntime::emitForOrderedIterationEnd(CodeGenFunction &CGF, 9701 SourceLocation Loc, 9702 unsigned IVSize, 9703 bool IVSigned) { 9704 llvm_unreachable("Not supported in SIMD-only mode"); 9705 } 9706 9707 void CGOpenMPSIMDRuntime::emitForStaticFinish(CodeGenFunction &CGF, 9708 SourceLocation Loc, 9709 OpenMPDirectiveKind DKind) { 9710 llvm_unreachable("Not supported in SIMD-only mode"); 9711 } 9712 9713 llvm::Value *CGOpenMPSIMDRuntime::emitForNext(CodeGenFunction &CGF, 9714 SourceLocation Loc, 9715 unsigned IVSize, bool IVSigned, 9716 Address IL, Address LB, 9717 Address UB, Address ST) { 9718 llvm_unreachable("Not supported in SIMD-only mode"); 9719 } 9720 9721 void CGOpenMPSIMDRuntime::emitNumThreadsClause(CodeGenFunction &CGF, 9722 llvm::Value *NumThreads, 9723 SourceLocation Loc) { 9724 llvm_unreachable("Not supported in SIMD-only mode"); 9725 } 9726 9727 void CGOpenMPSIMDRuntime::emitProcBindClause(CodeGenFunction &CGF, 9728 OpenMPProcBindClauseKind ProcBind, 9729 SourceLocation Loc) { 9730 llvm_unreachable("Not supported in SIMD-only mode"); 9731 } 9732 9733 Address CGOpenMPSIMDRuntime::getAddrOfThreadPrivate(CodeGenFunction &CGF, 9734 const VarDecl *VD, 9735 Address VDAddr, 9736 SourceLocation Loc) { 9737 llvm_unreachable("Not supported in SIMD-only mode"); 9738 } 9739 9740 llvm::Function *CGOpenMPSIMDRuntime::emitThreadPrivateVarDefinition( 9741 const VarDecl *VD, Address VDAddr, SourceLocation Loc, bool PerformInit, 9742 CodeGenFunction *CGF) { 9743 llvm_unreachable("Not supported in SIMD-only mode"); 9744 } 9745 9746 Address CGOpenMPSIMDRuntime::getAddrOfArtificialThreadPrivate( 9747 CodeGenFunction &CGF, QualType VarType, StringRef Name) { 9748 llvm_unreachable("Not supported in SIMD-only mode"); 9749 } 9750 9751 void CGOpenMPSIMDRuntime::emitFlush(CodeGenFunction &CGF, 9752 ArrayRef<const Expr *> Vars, 9753 SourceLocation Loc) { 9754 llvm_unreachable("Not supported in SIMD-only mode"); 9755 } 9756 9757 void CGOpenMPSIMDRuntime::emitTaskCall(CodeGenFunction &CGF, SourceLocation Loc, 9758 const OMPExecutableDirective &D, 9759 llvm::Function *TaskFunction, 9760 QualType SharedsTy, Address Shareds, 9761 const Expr *IfCond, 9762 const OMPTaskDataTy &Data) { 9763 llvm_unreachable("Not supported in SIMD-only mode"); 9764 } 9765 9766 void CGOpenMPSIMDRuntime::emitTaskLoopCall( 9767 CodeGenFunction &CGF, SourceLocation Loc, const OMPLoopDirective &D, 9768 llvm::Function *TaskFunction, QualType SharedsTy, Address Shareds, 9769 const Expr *IfCond, const OMPTaskDataTy &Data) { 9770 llvm_unreachable("Not supported in SIMD-only mode"); 9771 } 9772 9773 void CGOpenMPSIMDRuntime::emitReduction( 9774 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> Privates, 9775 ArrayRef<const Expr *> LHSExprs, ArrayRef<const Expr *> RHSExprs, 9776 ArrayRef<const Expr *> ReductionOps, ReductionOptionsTy Options) { 9777 assert(Options.SimpleReduction && "Only simple reduction is expected."); 9778 CGOpenMPRuntime::emitReduction(CGF, Loc, Privates, LHSExprs, RHSExprs, 9779 ReductionOps, Options); 9780 } 9781 9782 llvm::Value *CGOpenMPSIMDRuntime::emitTaskReductionInit( 9783 CodeGenFunction &CGF, SourceLocation Loc, ArrayRef<const Expr *> LHSExprs, 9784 ArrayRef<const Expr *> RHSExprs, const OMPTaskDataTy &Data) { 9785 llvm_unreachable("Not supported in SIMD-only mode"); 9786 } 9787 9788 void CGOpenMPSIMDRuntime::emitTaskReductionFixups(CodeGenFunction &CGF, 9789 SourceLocation Loc, 9790 ReductionCodeGen &RCG, 9791 unsigned N) { 9792 llvm_unreachable("Not supported in SIMD-only mode"); 9793 } 9794 9795 Address CGOpenMPSIMDRuntime::getTaskReductionItem(CodeGenFunction &CGF, 9796 SourceLocation Loc, 9797 llvm::Value *ReductionsPtr, 9798 LValue SharedLVal) { 9799 llvm_unreachable("Not supported in SIMD-only mode"); 9800 } 9801 9802 void CGOpenMPSIMDRuntime::emitTaskwaitCall(CodeGenFunction &CGF, 9803 SourceLocation Loc) { 9804 llvm_unreachable("Not supported in SIMD-only mode"); 9805 } 9806 9807 void CGOpenMPSIMDRuntime::emitCancellationPointCall( 9808 CodeGenFunction &CGF, SourceLocation Loc, 9809 OpenMPDirectiveKind CancelRegion) { 9810 llvm_unreachable("Not supported in SIMD-only mode"); 9811 } 9812 9813 void CGOpenMPSIMDRuntime::emitCancelCall(CodeGenFunction &CGF, 9814 SourceLocation Loc, const Expr *IfCond, 9815 OpenMPDirectiveKind CancelRegion) { 9816 llvm_unreachable("Not supported in SIMD-only mode"); 9817 } 9818 9819 void CGOpenMPSIMDRuntime::emitTargetOutlinedFunction( 9820 const OMPExecutableDirective &D, StringRef ParentName, 9821 llvm::Function *&OutlinedFn, llvm::Constant *&OutlinedFnID, 9822 bool IsOffloadEntry, const RegionCodeGenTy &CodeGen) { 9823 llvm_unreachable("Not supported in SIMD-only mode"); 9824 } 9825 9826 void CGOpenMPSIMDRuntime::emitTargetCall(CodeGenFunction &CGF, 9827 const OMPExecutableDirective &D, 9828 llvm::Function *OutlinedFn, 9829 llvm::Value *OutlinedFnID, 9830 const Expr *IfCond, 9831 const Expr *Device) { 9832 llvm_unreachable("Not supported in SIMD-only mode"); 9833 } 9834 9835 bool CGOpenMPSIMDRuntime::emitTargetFunctions(GlobalDecl GD) { 9836 llvm_unreachable("Not supported in SIMD-only mode"); 9837 } 9838 9839 bool CGOpenMPSIMDRuntime::emitTargetGlobalVariable(GlobalDecl GD) { 9840 llvm_unreachable("Not supported in SIMD-only mode"); 9841 } 9842 9843 bool CGOpenMPSIMDRuntime::emitTargetGlobal(GlobalDecl GD) { 9844 return false; 9845 } 9846 9847 llvm::Function *CGOpenMPSIMDRuntime::emitRegistrationFunction() { 9848 return nullptr; 9849 } 9850 9851 void CGOpenMPSIMDRuntime::emitTeamsCall(CodeGenFunction &CGF, 9852 const OMPExecutableDirective &D, 9853 SourceLocation Loc, 9854 llvm::Function *OutlinedFn, 9855 ArrayRef<llvm::Value *> CapturedVars) { 9856 llvm_unreachable("Not supported in SIMD-only mode"); 9857 } 9858 9859 void CGOpenMPSIMDRuntime::emitNumTeamsClause(CodeGenFunction &CGF, 9860 const Expr *NumTeams, 9861 const Expr *ThreadLimit, 9862 SourceLocation Loc) { 9863 llvm_unreachable("Not supported in SIMD-only mode"); 9864 } 9865 9866 void CGOpenMPSIMDRuntime::emitTargetDataCalls( 9867 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 9868 const Expr *Device, const RegionCodeGenTy &CodeGen, TargetDataInfo &Info) { 9869 llvm_unreachable("Not supported in SIMD-only mode"); 9870 } 9871 9872 void CGOpenMPSIMDRuntime::emitTargetDataStandAloneCall( 9873 CodeGenFunction &CGF, const OMPExecutableDirective &D, const Expr *IfCond, 9874 const Expr *Device) { 9875 llvm_unreachable("Not supported in SIMD-only mode"); 9876 } 9877 9878 void CGOpenMPSIMDRuntime::emitDoacrossInit(CodeGenFunction &CGF, 9879 const OMPLoopDirective &D, 9880 ArrayRef<Expr *> NumIterations) { 9881 llvm_unreachable("Not supported in SIMD-only mode"); 9882 } 9883 9884 void CGOpenMPSIMDRuntime::emitDoacrossOrdered(CodeGenFunction &CGF, 9885 const OMPDependClause *C) { 9886 llvm_unreachable("Not supported in SIMD-only mode"); 9887 } 9888 9889 const VarDecl * 9890 CGOpenMPSIMDRuntime::translateParameter(const FieldDecl *FD, 9891 const VarDecl *NativeParam) const { 9892 llvm_unreachable("Not supported in SIMD-only mode"); 9893 } 9894 9895 Address 9896 CGOpenMPSIMDRuntime::getParameterAddress(CodeGenFunction &CGF, 9897 const VarDecl *NativeParam, 9898 const VarDecl *TargetParam) const { 9899 llvm_unreachable("Not supported in SIMD-only mode"); 9900 } 9901