1 //===------ omptarget.cpp - Target independent OpenMP target RTL -- C++ -*-===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // Implementation of the interface to be used by Clang during the codegen of a 10 // target region. 11 // 12 //===----------------------------------------------------------------------===// 13 14 #include "omptarget.h" 15 #include "device.h" 16 #include "private.h" 17 #include "rtl.h" 18 19 #include <cassert> 20 #include <vector> 21 22 int AsyncInfoTy::synchronize() { 23 int Result = OFFLOAD_SUCCESS; 24 if (AsyncInfo.Queue) { 25 // If we have a queue we need to synchronize it now. 26 Result = Device.synchronize(*this); 27 assert(AsyncInfo.Queue == nullptr && 28 "The device plugin should have nulled the queue to indicate there " 29 "are no outstanding actions!"); 30 } 31 return Result; 32 } 33 34 void *&AsyncInfoTy::getVoidPtrLocation() { 35 BufferLocations.push_back(nullptr); 36 return BufferLocations.back(); 37 } 38 39 /* All begin addresses for partially mapped structs must be 8-aligned in order 40 * to ensure proper alignment of members. E.g. 41 * 42 * struct S { 43 * int a; // 4-aligned 44 * int b; // 4-aligned 45 * int *p; // 8-aligned 46 * } s1; 47 * ... 48 * #pragma omp target map(tofrom: s1.b, s1.p[0:N]) 49 * { 50 * s1.b = 5; 51 * for (int i...) s1.p[i] = ...; 52 * } 53 * 54 * Here we are mapping s1 starting from member b, so BaseAddress=&s1=&s1.a and 55 * BeginAddress=&s1.b. Let's assume that the struct begins at address 0x100, 56 * then &s1.a=0x100, &s1.b=0x104, &s1.p=0x108. Each member obeys the alignment 57 * requirements for its type. Now, when we allocate memory on the device, in 58 * CUDA's case cuMemAlloc() returns an address which is at least 256-aligned. 59 * This means that the chunk of the struct on the device will start at a 60 * 256-aligned address, let's say 0x200. Then the address of b will be 0x200 and 61 * address of p will be a misaligned 0x204 (on the host there was no need to add 62 * padding between b and p, so p comes exactly 4 bytes after b). If the device 63 * kernel tries to access s1.p, a misaligned address error occurs (as reported 64 * by the CUDA plugin). By padding the begin address down to a multiple of 8 and 65 * extending the size of the allocated chuck accordingly, the chuck on the 66 * device will start at 0x200 with the padding (4 bytes), then &s1.b=0x204 and 67 * &s1.p=0x208, as they should be to satisfy the alignment requirements. 68 */ 69 static const int64_t Alignment = 8; 70 71 /// Map global data and execute pending ctors 72 static int InitLibrary(DeviceTy &Device) { 73 /* 74 * Map global data 75 */ 76 int32_t device_id = Device.DeviceID; 77 int rc = OFFLOAD_SUCCESS; 78 bool supportsEmptyImages = Device.RTL->supports_empty_images && 79 Device.RTL->supports_empty_images() > 0; 80 81 Device.PendingGlobalsMtx.lock(); 82 PM->TrlTblMtx.lock(); 83 for (auto *HostEntriesBegin : PM->HostEntriesBeginRegistrationOrder) { 84 TranslationTable *TransTable = 85 &PM->HostEntriesBeginToTransTable[HostEntriesBegin]; 86 if (TransTable->HostTable.EntriesBegin == 87 TransTable->HostTable.EntriesEnd && 88 !supportsEmptyImages) { 89 // No host entry so no need to proceed 90 continue; 91 } 92 93 if (TransTable->TargetsTable[device_id] != 0) { 94 // Library entries have already been processed 95 continue; 96 } 97 98 // 1) get image. 99 assert(TransTable->TargetsImages.size() > (size_t)device_id && 100 "Not expecting a device ID outside the table's bounds!"); 101 __tgt_device_image *img = TransTable->TargetsImages[device_id]; 102 if (!img) { 103 REPORT("No image loaded for device id %d.\n", device_id); 104 rc = OFFLOAD_FAIL; 105 break; 106 } 107 // 2) load image into the target table. 108 __tgt_target_table *TargetTable = TransTable->TargetsTable[device_id] = 109 Device.load_binary(img); 110 // Unable to get table for this image: invalidate image and fail. 111 if (!TargetTable) { 112 REPORT("Unable to generate entries table for device id %d.\n", device_id); 113 TransTable->TargetsImages[device_id] = 0; 114 rc = OFFLOAD_FAIL; 115 break; 116 } 117 118 // Verify whether the two table sizes match. 119 size_t hsize = 120 TransTable->HostTable.EntriesEnd - TransTable->HostTable.EntriesBegin; 121 size_t tsize = TargetTable->EntriesEnd - TargetTable->EntriesBegin; 122 123 // Invalid image for these host entries! 124 if (hsize != tsize) { 125 REPORT("Host and Target tables mismatch for device id %d [%zx != %zx].\n", 126 device_id, hsize, tsize); 127 TransTable->TargetsImages[device_id] = 0; 128 TransTable->TargetsTable[device_id] = 0; 129 rc = OFFLOAD_FAIL; 130 break; 131 } 132 133 // process global data that needs to be mapped. 134 Device.DataMapMtx.lock(); 135 __tgt_target_table *HostTable = &TransTable->HostTable; 136 for (__tgt_offload_entry *CurrDeviceEntry = TargetTable->EntriesBegin, 137 *CurrHostEntry = HostTable->EntriesBegin, 138 *EntryDeviceEnd = TargetTable->EntriesEnd; 139 CurrDeviceEntry != EntryDeviceEnd; 140 CurrDeviceEntry++, CurrHostEntry++) { 141 if (CurrDeviceEntry->size != 0) { 142 // has data. 143 assert(CurrDeviceEntry->size == CurrHostEntry->size && 144 "data size mismatch"); 145 146 // Fortran may use multiple weak declarations for the same symbol, 147 // therefore we must allow for multiple weak symbols to be loaded from 148 // the fat binary. Treat these mappings as any other "regular" mapping. 149 // Add entry to map. 150 if (Device.getTgtPtrBegin(CurrHostEntry->addr, CurrHostEntry->size)) 151 continue; 152 DP("Add mapping from host " DPxMOD " to device " DPxMOD " with size %zu" 153 "\n", 154 DPxPTR(CurrHostEntry->addr), DPxPTR(CurrDeviceEntry->addr), 155 CurrDeviceEntry->size); 156 Device.HostDataToTargetMap.emplace( 157 (uintptr_t)CurrHostEntry->addr /*HstPtrBase*/, 158 (uintptr_t)CurrHostEntry->addr /*HstPtrBegin*/, 159 (uintptr_t)CurrHostEntry->addr + CurrHostEntry->size /*HstPtrEnd*/, 160 (uintptr_t)CurrDeviceEntry->addr /*TgtPtrBegin*/, nullptr, 161 true /*IsRefCountINF*/); 162 } 163 } 164 Device.DataMapMtx.unlock(); 165 } 166 PM->TrlTblMtx.unlock(); 167 168 if (rc != OFFLOAD_SUCCESS) { 169 Device.PendingGlobalsMtx.unlock(); 170 return rc; 171 } 172 173 /* 174 * Run ctors for static objects 175 */ 176 if (!Device.PendingCtorsDtors.empty()) { 177 AsyncInfoTy AsyncInfo(Device); 178 // Call all ctors for all libraries registered so far 179 for (auto &lib : Device.PendingCtorsDtors) { 180 if (!lib.second.PendingCtors.empty()) { 181 DP("Has pending ctors... call now\n"); 182 for (auto &entry : lib.second.PendingCtors) { 183 void *ctor = entry; 184 int rc = 185 target(nullptr, Device, ctor, 0, nullptr, nullptr, nullptr, 186 nullptr, nullptr, nullptr, 1, 1, true /*team*/, AsyncInfo); 187 if (rc != OFFLOAD_SUCCESS) { 188 REPORT("Running ctor " DPxMOD " failed.\n", DPxPTR(ctor)); 189 Device.PendingGlobalsMtx.unlock(); 190 return OFFLOAD_FAIL; 191 } 192 } 193 // Clear the list to indicate that this device has been used 194 lib.second.PendingCtors.clear(); 195 DP("Done with pending ctors for lib " DPxMOD "\n", DPxPTR(lib.first)); 196 } 197 } 198 // All constructors have been issued, wait for them now. 199 if (AsyncInfo.synchronize() != OFFLOAD_SUCCESS) 200 return OFFLOAD_FAIL; 201 } 202 Device.HasPendingGlobals = false; 203 Device.PendingGlobalsMtx.unlock(); 204 205 return OFFLOAD_SUCCESS; 206 } 207 208 void handleTargetOutcome(bool Success, ident_t *Loc) { 209 switch (PM->TargetOffloadPolicy) { 210 case tgt_disabled: 211 if (Success) { 212 FATAL_MESSAGE0(1, "expected no offloading while offloading is disabled"); 213 } 214 break; 215 case tgt_default: 216 FATAL_MESSAGE0(1, "default offloading policy must be switched to " 217 "mandatory or disabled"); 218 break; 219 case tgt_mandatory: 220 if (!Success) { 221 if (getInfoLevel() & OMP_INFOTYPE_DUMP_TABLE) 222 for (auto &Device : PM->Devices) 223 dumpTargetPointerMappings(Loc, Device); 224 else 225 FAILURE_MESSAGE("Run with LIBOMPTARGET_INFO=%d to dump host-target " 226 "pointer mappings.\n", 227 OMP_INFOTYPE_DUMP_TABLE); 228 229 SourceInfo info(Loc); 230 if (info.isAvailible()) 231 fprintf(stderr, "%s:%d:%d: ", info.getFilename(), info.getLine(), 232 info.getColumn()); 233 else 234 FAILURE_MESSAGE("Source location information not present. Compile with " 235 "-g or -gline-tables-only.\n"); 236 FATAL_MESSAGE0( 237 1, "failure of target construct while offloading is mandatory"); 238 } else { 239 if (getInfoLevel() & OMP_INFOTYPE_DUMP_TABLE) 240 for (auto &Device : PM->Devices) 241 dumpTargetPointerMappings(Loc, Device); 242 } 243 break; 244 } 245 } 246 247 static void handleDefaultTargetOffload() { 248 PM->TargetOffloadMtx.lock(); 249 if (PM->TargetOffloadPolicy == tgt_default) { 250 if (omp_get_num_devices() > 0) { 251 DP("Default TARGET OFFLOAD policy is now mandatory " 252 "(devices were found)\n"); 253 PM->TargetOffloadPolicy = tgt_mandatory; 254 } else { 255 DP("Default TARGET OFFLOAD policy is now disabled " 256 "(no devices were found)\n"); 257 PM->TargetOffloadPolicy = tgt_disabled; 258 } 259 } 260 PM->TargetOffloadMtx.unlock(); 261 } 262 263 static bool isOffloadDisabled() { 264 if (PM->TargetOffloadPolicy == tgt_default) 265 handleDefaultTargetOffload(); 266 return PM->TargetOffloadPolicy == tgt_disabled; 267 } 268 269 // If offload is enabled, ensure that device DeviceID has been initialized, 270 // global ctors have been executed, and global data has been mapped. 271 // 272 // There are three possible results: 273 // - Return OFFLOAD_SUCCESS if the device is ready for offload. 274 // - Return OFFLOAD_FAIL without reporting a runtime error if offload is 275 // disabled, perhaps because the initial device was specified. 276 // - Report a runtime error and return OFFLOAD_FAIL. 277 // 278 // If DeviceID == OFFLOAD_DEVICE_DEFAULT, set DeviceID to the default device. 279 // This step might be skipped if offload is disabled. 280 int checkDeviceAndCtors(int64_t &DeviceID, ident_t *Loc) { 281 if (isOffloadDisabled()) { 282 DP("Offload is disabled\n"); 283 return OFFLOAD_FAIL; 284 } 285 286 if (DeviceID == OFFLOAD_DEVICE_DEFAULT) { 287 DeviceID = omp_get_default_device(); 288 DP("Use default device id %" PRId64 "\n", DeviceID); 289 } 290 291 // Proposed behavior for OpenMP 5.2 in OpenMP spec github issue 2669. 292 if (omp_get_num_devices() == 0) { 293 DP("omp_get_num_devices() == 0 but offload is manadatory\n"); 294 handleTargetOutcome(false, Loc); 295 return OFFLOAD_FAIL; 296 } 297 298 if (DeviceID == omp_get_initial_device()) { 299 DP("Device is host (%" PRId64 "), returning as if offload is disabled\n", 300 DeviceID); 301 return OFFLOAD_FAIL; 302 } 303 304 // Is device ready? 305 if (!device_is_ready(DeviceID)) { 306 REPORT("Device %" PRId64 " is not ready.\n", DeviceID); 307 handleTargetOutcome(false, Loc); 308 return OFFLOAD_FAIL; 309 } 310 311 // Get device info. 312 DeviceTy &Device = PM->Devices[DeviceID]; 313 314 // Check whether global data has been mapped for this device 315 Device.PendingGlobalsMtx.lock(); 316 bool hasPendingGlobals = Device.HasPendingGlobals; 317 Device.PendingGlobalsMtx.unlock(); 318 if (hasPendingGlobals && InitLibrary(Device) != OFFLOAD_SUCCESS) { 319 REPORT("Failed to init globals on device %" PRId64 "\n", DeviceID); 320 handleTargetOutcome(false, Loc); 321 return OFFLOAD_FAIL; 322 } 323 324 return OFFLOAD_SUCCESS; 325 } 326 327 static int32_t getParentIndex(int64_t type) { 328 return ((type & OMP_TGT_MAPTYPE_MEMBER_OF) >> 48) - 1; 329 } 330 331 void *targetAllocExplicit(size_t size, int device_num, int kind, 332 const char *name) { 333 TIMESCOPE(); 334 DP("Call to %s for device %d requesting %zu bytes\n", name, device_num, size); 335 336 if (size <= 0) { 337 DP("Call to %s with non-positive length\n", name); 338 return NULL; 339 } 340 341 void *rc = NULL; 342 343 if (device_num == omp_get_initial_device()) { 344 rc = malloc(size); 345 DP("%s returns host ptr " DPxMOD "\n", name, DPxPTR(rc)); 346 return rc; 347 } 348 349 if (!device_is_ready(device_num)) { 350 DP("%s returns NULL ptr\n", name); 351 return NULL; 352 } 353 354 DeviceTy &Device = PM->Devices[device_num]; 355 rc = Device.allocData(size, nullptr, kind); 356 DP("%s returns device ptr " DPxMOD "\n", name, DPxPTR(rc)); 357 return rc; 358 } 359 360 /// Call the user-defined mapper function followed by the appropriate 361 // targetData* function (targetData{Begin,End,Update}). 362 int targetDataMapper(ident_t *loc, DeviceTy &Device, void *arg_base, void *arg, 363 int64_t arg_size, int64_t arg_type, 364 map_var_info_t arg_names, void *arg_mapper, 365 AsyncInfoTy &AsyncInfo, 366 TargetDataFuncPtrTy target_data_function) { 367 TIMESCOPE_WITH_IDENT(loc); 368 DP("Calling the mapper function " DPxMOD "\n", DPxPTR(arg_mapper)); 369 370 // The mapper function fills up Components. 371 MapperComponentsTy MapperComponents; 372 MapperFuncPtrTy MapperFuncPtr = (MapperFuncPtrTy)(arg_mapper); 373 (*MapperFuncPtr)((void *)&MapperComponents, arg_base, arg, arg_size, arg_type, 374 arg_names); 375 376 // Construct new arrays for args_base, args, arg_sizes and arg_types 377 // using the information in MapperComponents and call the corresponding 378 // targetData* function using these new arrays. 379 std::vector<void *> MapperArgsBase(MapperComponents.Components.size()); 380 std::vector<void *> MapperArgs(MapperComponents.Components.size()); 381 std::vector<int64_t> MapperArgSizes(MapperComponents.Components.size()); 382 std::vector<int64_t> MapperArgTypes(MapperComponents.Components.size()); 383 std::vector<void *> MapperArgNames(MapperComponents.Components.size()); 384 385 for (unsigned I = 0, E = MapperComponents.Components.size(); I < E; ++I) { 386 auto &C = MapperComponents.Components[I]; 387 MapperArgsBase[I] = C.Base; 388 MapperArgs[I] = C.Begin; 389 MapperArgSizes[I] = C.Size; 390 MapperArgTypes[I] = C.Type; 391 MapperArgNames[I] = C.Name; 392 } 393 394 int rc = target_data_function(loc, Device, MapperComponents.Components.size(), 395 MapperArgsBase.data(), MapperArgs.data(), 396 MapperArgSizes.data(), MapperArgTypes.data(), 397 MapperArgNames.data(), /*arg_mappers*/ nullptr, 398 AsyncInfo, /*FromMapper=*/true); 399 400 return rc; 401 } 402 403 /// Internal function to do the mapping and transfer the data to the device 404 int targetDataBegin(ident_t *loc, DeviceTy &Device, int32_t arg_num, 405 void **args_base, void **args, int64_t *arg_sizes, 406 int64_t *arg_types, map_var_info_t *arg_names, 407 void **arg_mappers, AsyncInfoTy &AsyncInfo, 408 bool FromMapper) { 409 // process each input. 410 for (int32_t i = 0; i < arg_num; ++i) { 411 // Ignore private variables and arrays - there is no mapping for them. 412 if ((arg_types[i] & OMP_TGT_MAPTYPE_LITERAL) || 413 (arg_types[i] & OMP_TGT_MAPTYPE_PRIVATE)) 414 continue; 415 416 if (arg_mappers && arg_mappers[i]) { 417 // Instead of executing the regular path of targetDataBegin, call the 418 // targetDataMapper variant which will call targetDataBegin again 419 // with new arguments. 420 DP("Calling targetDataMapper for the %dth argument\n", i); 421 422 map_var_info_t arg_name = (!arg_names) ? nullptr : arg_names[i]; 423 int rc = targetDataMapper(loc, Device, args_base[i], args[i], 424 arg_sizes[i], arg_types[i], arg_name, 425 arg_mappers[i], AsyncInfo, targetDataBegin); 426 427 if (rc != OFFLOAD_SUCCESS) { 428 REPORT("Call to targetDataBegin via targetDataMapper for custom mapper" 429 " failed.\n"); 430 return OFFLOAD_FAIL; 431 } 432 433 // Skip the rest of this function, continue to the next argument. 434 continue; 435 } 436 437 void *HstPtrBegin = args[i]; 438 void *HstPtrBase = args_base[i]; 439 int64_t data_size = arg_sizes[i]; 440 map_var_info_t HstPtrName = (!arg_names) ? nullptr : arg_names[i]; 441 442 // Adjust for proper alignment if this is a combined entry (for structs). 443 // Look at the next argument - if that is MEMBER_OF this one, then this one 444 // is a combined entry. 445 int64_t padding = 0; 446 const int next_i = i + 1; 447 if (getParentIndex(arg_types[i]) < 0 && next_i < arg_num && 448 getParentIndex(arg_types[next_i]) == i) { 449 padding = (int64_t)HstPtrBegin % Alignment; 450 if (padding) { 451 DP("Using a padding of %" PRId64 " bytes for begin address " DPxMOD 452 "\n", 453 padding, DPxPTR(HstPtrBegin)); 454 HstPtrBegin = (char *)HstPtrBegin - padding; 455 data_size += padding; 456 } 457 } 458 459 // Address of pointer on the host and device, respectively. 460 void *Pointer_HstPtrBegin, *PointerTgtPtrBegin; 461 TargetPointerResultTy Pointer_TPR; 462 bool IsHostPtr = false; 463 bool IsImplicit = arg_types[i] & OMP_TGT_MAPTYPE_IMPLICIT; 464 // Force the creation of a device side copy of the data when: 465 // a close map modifier was associated with a map that contained a to. 466 bool HasCloseModifier = arg_types[i] & OMP_TGT_MAPTYPE_CLOSE; 467 bool HasPresentModifier = arg_types[i] & OMP_TGT_MAPTYPE_PRESENT; 468 // UpdateRef is based on MEMBER_OF instead of TARGET_PARAM because if we 469 // have reached this point via __tgt_target_data_begin and not __tgt_target 470 // then no argument is marked as TARGET_PARAM ("omp target data map" is not 471 // associated with a target region, so there are no target parameters). This 472 // may be considered a hack, we could revise the scheme in the future. 473 bool UpdateRef = 474 !(arg_types[i] & OMP_TGT_MAPTYPE_MEMBER_OF) && !(FromMapper && i == 0); 475 if (arg_types[i] & OMP_TGT_MAPTYPE_PTR_AND_OBJ) { 476 DP("Has a pointer entry: \n"); 477 // Base is address of pointer. 478 // 479 // Usually, the pointer is already allocated by this time. For example: 480 // 481 // #pragma omp target map(s.p[0:N]) 482 // 483 // The map entry for s comes first, and the PTR_AND_OBJ entry comes 484 // afterward, so the pointer is already allocated by the time the 485 // PTR_AND_OBJ entry is handled below, and PointerTgtPtrBegin is thus 486 // non-null. However, "declare target link" can produce a PTR_AND_OBJ 487 // entry for a global that might not already be allocated by the time the 488 // PTR_AND_OBJ entry is handled below, and so the allocation might fail 489 // when HasPresentModifier. 490 Pointer_TPR = Device.getOrAllocTgtPtr( 491 HstPtrBase, HstPtrBase, sizeof(void *), nullptr, IsImplicit, 492 UpdateRef, HasCloseModifier, HasPresentModifier); 493 PointerTgtPtrBegin = Pointer_TPR.TargetPointer; 494 IsHostPtr = Pointer_TPR.Flags.IsHostPointer; 495 if (!PointerTgtPtrBegin) { 496 REPORT("Call to getOrAllocTgtPtr returned null pointer (%s).\n", 497 HasPresentModifier ? "'present' map type modifier" 498 : "device failure or illegal mapping"); 499 return OFFLOAD_FAIL; 500 } 501 DP("There are %zu bytes allocated at target address " DPxMOD " - is%s new" 502 "\n", 503 sizeof(void *), DPxPTR(PointerTgtPtrBegin), 504 (Pointer_TPR.Flags.IsNewEntry ? "" : " not")); 505 Pointer_HstPtrBegin = HstPtrBase; 506 // modify current entry. 507 HstPtrBase = *(void **)HstPtrBase; 508 // No need to update pointee ref count for the first element of the 509 // subelement that comes from mapper. 510 UpdateRef = 511 (!FromMapper || i != 0); // subsequently update ref count of pointee 512 } 513 514 auto TPR = Device.getOrAllocTgtPtr(HstPtrBegin, HstPtrBase, data_size, 515 HstPtrName, IsImplicit, UpdateRef, 516 HasCloseModifier, HasPresentModifier); 517 void *TgtPtrBegin = TPR.TargetPointer; 518 IsHostPtr = TPR.Flags.IsHostPointer; 519 // If data_size==0, then the argument could be a zero-length pointer to 520 // NULL, so getOrAlloc() returning NULL is not an error. 521 if (!TgtPtrBegin && (data_size || HasPresentModifier)) { 522 REPORT("Call to getOrAllocTgtPtr returned null pointer (%s).\n", 523 HasPresentModifier ? "'present' map type modifier" 524 : "device failure or illegal mapping"); 525 return OFFLOAD_FAIL; 526 } 527 DP("There are %" PRId64 " bytes allocated at target address " DPxMOD 528 " - is%s new\n", 529 data_size, DPxPTR(TgtPtrBegin), (TPR.Flags.IsNewEntry ? "" : " not")); 530 531 if (arg_types[i] & OMP_TGT_MAPTYPE_RETURN_PARAM) { 532 uintptr_t Delta = (uintptr_t)HstPtrBegin - (uintptr_t)HstPtrBase; 533 void *TgtPtrBase = (void *)((uintptr_t)TgtPtrBegin - Delta); 534 DP("Returning device pointer " DPxMOD "\n", DPxPTR(TgtPtrBase)); 535 args_base[i] = TgtPtrBase; 536 } 537 538 if (arg_types[i] & OMP_TGT_MAPTYPE_TO) { 539 bool copy = false; 540 if (!(PM->RTLs.RequiresFlags & OMP_REQ_UNIFIED_SHARED_MEMORY) || 541 HasCloseModifier) { 542 if (TPR.Flags.IsNewEntry || (arg_types[i] & OMP_TGT_MAPTYPE_ALWAYS)) { 543 copy = true; 544 } else if ((arg_types[i] & OMP_TGT_MAPTYPE_MEMBER_OF) && 545 !(arg_types[i] & OMP_TGT_MAPTYPE_PTR_AND_OBJ)) { 546 // Copy data only if the "parent" struct has RefCount==1. 547 // If this is a PTR_AND_OBJ entry, the OBJ is not part of the struct, 548 // so exclude it from this check. 549 int32_t parent_idx = getParentIndex(arg_types[i]); 550 uint64_t parent_rc = Device.getMapEntryRefCnt(args[parent_idx]); 551 assert(parent_rc > 0 && "parent struct not found"); 552 if (parent_rc == 1) { 553 copy = true; 554 } 555 } 556 } 557 558 if (copy && !IsHostPtr) { 559 DP("Moving %" PRId64 " bytes (hst:" DPxMOD ") -> (tgt:" DPxMOD ")\n", 560 data_size, DPxPTR(HstPtrBegin), DPxPTR(TgtPtrBegin)); 561 int rt = 562 Device.submitData(TgtPtrBegin, HstPtrBegin, data_size, AsyncInfo); 563 if (rt != OFFLOAD_SUCCESS) { 564 REPORT("Copying data to device failed.\n"); 565 return OFFLOAD_FAIL; 566 } 567 } 568 } 569 570 if (arg_types[i] & OMP_TGT_MAPTYPE_PTR_AND_OBJ && !IsHostPtr) { 571 DP("Update pointer (" DPxMOD ") -> [" DPxMOD "]\n", 572 DPxPTR(PointerTgtPtrBegin), DPxPTR(TgtPtrBegin)); 573 uint64_t Delta = (uint64_t)HstPtrBegin - (uint64_t)HstPtrBase; 574 void *&TgtPtrBase = AsyncInfo.getVoidPtrLocation(); 575 TgtPtrBase = (void *)((uint64_t)TgtPtrBegin - Delta); 576 int rt = Device.submitData(PointerTgtPtrBegin, &TgtPtrBase, 577 sizeof(void *), AsyncInfo); 578 if (rt != OFFLOAD_SUCCESS) { 579 REPORT("Copying data to device failed.\n"); 580 return OFFLOAD_FAIL; 581 } 582 // create shadow pointers for this entry 583 Device.ShadowMtx.lock(); 584 Device.ShadowPtrMap[Pointer_HstPtrBegin] = { 585 HstPtrBase, PointerTgtPtrBegin, TgtPtrBase}; 586 Device.ShadowMtx.unlock(); 587 } 588 } 589 590 return OFFLOAD_SUCCESS; 591 } 592 593 namespace { 594 /// This structure contains information to deallocate a target pointer, aka. 595 /// used to call the function \p DeviceTy::deallocTgtPtr. 596 struct DeallocTgtPtrInfo { 597 /// Host pointer used to look up into the map table 598 void *HstPtrBegin; 599 /// Size of the data 600 int64_t DataSize; 601 /// Whether it has \p close modifier 602 bool HasCloseModifier; 603 604 DeallocTgtPtrInfo(void *HstPtr, int64_t Size, bool HasCloseModifier) 605 : HstPtrBegin(HstPtr), DataSize(Size), 606 HasCloseModifier(HasCloseModifier) {} 607 }; 608 } // namespace 609 610 /// Internal function to undo the mapping and retrieve the data from the device. 611 int targetDataEnd(ident_t *loc, DeviceTy &Device, int32_t ArgNum, 612 void **ArgBases, void **Args, int64_t *ArgSizes, 613 int64_t *ArgTypes, map_var_info_t *ArgNames, 614 void **ArgMappers, AsyncInfoTy &AsyncInfo, bool FromMapper) { 615 int Ret; 616 std::vector<DeallocTgtPtrInfo> DeallocTgtPtrs; 617 void *FromMapperBase = nullptr; 618 // process each input. 619 for (int32_t I = ArgNum - 1; I >= 0; --I) { 620 // Ignore private variables and arrays - there is no mapping for them. 621 // Also, ignore the use_device_ptr directive, it has no effect here. 622 if ((ArgTypes[I] & OMP_TGT_MAPTYPE_LITERAL) || 623 (ArgTypes[I] & OMP_TGT_MAPTYPE_PRIVATE)) 624 continue; 625 626 if (ArgMappers && ArgMappers[I]) { 627 // Instead of executing the regular path of targetDataEnd, call the 628 // targetDataMapper variant which will call targetDataEnd again 629 // with new arguments. 630 DP("Calling targetDataMapper for the %dth argument\n", I); 631 632 map_var_info_t ArgName = (!ArgNames) ? nullptr : ArgNames[I]; 633 Ret = targetDataMapper(loc, Device, ArgBases[I], Args[I], ArgSizes[I], 634 ArgTypes[I], ArgName, ArgMappers[I], AsyncInfo, 635 targetDataEnd); 636 637 if (Ret != OFFLOAD_SUCCESS) { 638 REPORT("Call to targetDataEnd via targetDataMapper for custom mapper" 639 " failed.\n"); 640 return OFFLOAD_FAIL; 641 } 642 643 // Skip the rest of this function, continue to the next argument. 644 continue; 645 } 646 647 void *HstPtrBegin = Args[I]; 648 int64_t DataSize = ArgSizes[I]; 649 // Adjust for proper alignment if this is a combined entry (for structs). 650 // Look at the next argument - if that is MEMBER_OF this one, then this one 651 // is a combined entry. 652 const int NextI = I + 1; 653 if (getParentIndex(ArgTypes[I]) < 0 && NextI < ArgNum && 654 getParentIndex(ArgTypes[NextI]) == I) { 655 int64_t Padding = (int64_t)HstPtrBegin % Alignment; 656 if (Padding) { 657 DP("Using a Padding of %" PRId64 " bytes for begin address " DPxMOD 658 "\n", 659 Padding, DPxPTR(HstPtrBegin)); 660 HstPtrBegin = (char *)HstPtrBegin - Padding; 661 DataSize += Padding; 662 } 663 } 664 665 bool IsLast, IsHostPtr; 666 bool IsImplicit = ArgTypes[I] & OMP_TGT_MAPTYPE_IMPLICIT; 667 bool UpdateRef = (!(ArgTypes[I] & OMP_TGT_MAPTYPE_MEMBER_OF) || 668 (ArgTypes[I] & OMP_TGT_MAPTYPE_PTR_AND_OBJ)) && 669 !(FromMapper && I == 0); 670 bool ForceDelete = ArgTypes[I] & OMP_TGT_MAPTYPE_DELETE; 671 bool HasCloseModifier = ArgTypes[I] & OMP_TGT_MAPTYPE_CLOSE; 672 bool HasPresentModifier = ArgTypes[I] & OMP_TGT_MAPTYPE_PRESENT; 673 674 // If PTR_AND_OBJ, HstPtrBegin is address of pointee 675 void *TgtPtrBegin = 676 Device.getTgtPtrBegin(HstPtrBegin, DataSize, IsLast, UpdateRef, 677 IsHostPtr, !IsImplicit, ForceDelete); 678 if (!TgtPtrBegin && (DataSize || HasPresentModifier)) { 679 DP("Mapping does not exist (%s)\n", 680 (HasPresentModifier ? "'present' map type modifier" : "ignored")); 681 if (HasPresentModifier) { 682 // OpenMP 5.1, sec. 2.21.7.1 "map Clause", p. 350 L10-13: 683 // "If a map clause appears on a target, target data, target enter data 684 // or target exit data construct with a present map-type-modifier then 685 // on entry to the region if the corresponding list item does not appear 686 // in the device data environment then an error occurs and the program 687 // terminates." 688 // 689 // This should be an error upon entering an "omp target exit data". It 690 // should not be an error upon exiting an "omp target data" or "omp 691 // target". For "omp target data", Clang thus doesn't include present 692 // modifiers for end calls. For "omp target", we have not found a valid 693 // OpenMP program for which the error matters: it appears that, if a 694 // program can guarantee that data is present at the beginning of an 695 // "omp target" region so that there's no error there, that data is also 696 // guaranteed to be present at the end. 697 MESSAGE("device mapping required by 'present' map type modifier does " 698 "not exist for host address " DPxMOD " (%" PRId64 " bytes)", 699 DPxPTR(HstPtrBegin), DataSize); 700 return OFFLOAD_FAIL; 701 } 702 } else { 703 DP("There are %" PRId64 " bytes allocated at target address " DPxMOD 704 " - is%s last\n", 705 DataSize, DPxPTR(TgtPtrBegin), (IsLast ? "" : " not")); 706 } 707 708 // OpenMP 5.1, sec. 2.21.7.1 "map Clause", p. 351 L14-16: 709 // "If the map clause appears on a target, target data, or target exit data 710 // construct and a corresponding list item of the original list item is not 711 // present in the device data environment on exit from the region then the 712 // list item is ignored." 713 if (!TgtPtrBegin) 714 continue; 715 716 bool DelEntry = IsLast; 717 718 // If the last element from the mapper (for end transfer args comes in 719 // reverse order), do not remove the partial entry, the parent struct still 720 // exists. 721 if ((ArgTypes[I] & OMP_TGT_MAPTYPE_MEMBER_OF) && 722 !(ArgTypes[I] & OMP_TGT_MAPTYPE_PTR_AND_OBJ)) { 723 DelEntry = false; // protect parent struct from being deallocated 724 } 725 726 if ((ArgTypes[I] & OMP_TGT_MAPTYPE_FROM) || DelEntry) { 727 // Move data back to the host 728 if (ArgTypes[I] & OMP_TGT_MAPTYPE_FROM) { 729 bool Always = ArgTypes[I] & OMP_TGT_MAPTYPE_ALWAYS; 730 bool CopyMember = false; 731 if (!(PM->RTLs.RequiresFlags & OMP_REQ_UNIFIED_SHARED_MEMORY) || 732 HasCloseModifier) { 733 if ((ArgTypes[I] & OMP_TGT_MAPTYPE_MEMBER_OF) && 734 !(ArgTypes[I] & OMP_TGT_MAPTYPE_PTR_AND_OBJ)) { 735 // Copy data only if the "parent" struct has RefCount==1. 736 int32_t ParentIdx = getParentIndex(ArgTypes[I]); 737 uint64_t ParentRC = Device.getMapEntryRefCnt(Args[ParentIdx]); 738 assert(ParentRC > 0 && "parent struct not found"); 739 if (ParentRC == 1) 740 CopyMember = true; 741 } 742 } 743 744 if ((DelEntry || Always || CopyMember) && 745 !(PM->RTLs.RequiresFlags & OMP_REQ_UNIFIED_SHARED_MEMORY && 746 TgtPtrBegin == HstPtrBegin)) { 747 DP("Moving %" PRId64 " bytes (tgt:" DPxMOD ") -> (hst:" DPxMOD ")\n", 748 DataSize, DPxPTR(TgtPtrBegin), DPxPTR(HstPtrBegin)); 749 Ret = Device.retrieveData(HstPtrBegin, TgtPtrBegin, DataSize, 750 AsyncInfo); 751 if (Ret != OFFLOAD_SUCCESS) { 752 REPORT("Copying data from device failed.\n"); 753 return OFFLOAD_FAIL; 754 } 755 } 756 } 757 if (DelEntry && FromMapper && I == 0) { 758 DelEntry = false; 759 FromMapperBase = HstPtrBegin; 760 } 761 762 // If we copied back to the host a struct/array containing pointers, we 763 // need to restore the original host pointer values from their shadow 764 // copies. If the struct is going to be deallocated, remove any remaining 765 // shadow pointer entries for this struct. 766 uintptr_t LB = (uintptr_t)HstPtrBegin; 767 uintptr_t UB = (uintptr_t)HstPtrBegin + DataSize; 768 Device.ShadowMtx.lock(); 769 for (ShadowPtrListTy::iterator Itr = Device.ShadowPtrMap.begin(); 770 Itr != Device.ShadowPtrMap.end();) { 771 void **ShadowHstPtrAddr = (void **)Itr->first; 772 773 // An STL map is sorted on its keys; use this property 774 // to quickly determine when to break out of the loop. 775 if ((uintptr_t)ShadowHstPtrAddr < LB) { 776 ++Itr; 777 continue; 778 } 779 if ((uintptr_t)ShadowHstPtrAddr >= UB) 780 break; 781 782 // If we copied the struct to the host, we need to restore the pointer. 783 if (ArgTypes[I] & OMP_TGT_MAPTYPE_FROM) { 784 DP("Restoring original host pointer value " DPxMOD " for host " 785 "pointer " DPxMOD "\n", 786 DPxPTR(Itr->second.HstPtrVal), DPxPTR(ShadowHstPtrAddr)); 787 *ShadowHstPtrAddr = Itr->second.HstPtrVal; 788 } 789 // If the struct is to be deallocated, remove the shadow entry. 790 if (DelEntry) { 791 DP("Removing shadow pointer " DPxMOD "\n", DPxPTR(ShadowHstPtrAddr)); 792 Itr = Device.ShadowPtrMap.erase(Itr); 793 } else { 794 ++Itr; 795 } 796 } 797 Device.ShadowMtx.unlock(); 798 799 // Add pointer to the buffer for later deallocation 800 if (DelEntry) 801 DeallocTgtPtrs.emplace_back(HstPtrBegin, DataSize, HasCloseModifier); 802 } 803 } 804 805 // TODO: We should not synchronize here but pass the AsyncInfo object to the 806 // allocate/deallocate device APIs. 807 // 808 // We need to synchronize before deallocating data. 809 Ret = AsyncInfo.synchronize(); 810 if (Ret != OFFLOAD_SUCCESS) 811 return OFFLOAD_FAIL; 812 813 // Deallocate target pointer 814 for (DeallocTgtPtrInfo &Info : DeallocTgtPtrs) { 815 if (FromMapperBase && FromMapperBase == Info.HstPtrBegin) 816 continue; 817 Ret = Device.deallocTgtPtr(Info.HstPtrBegin, Info.DataSize, 818 Info.HasCloseModifier); 819 if (Ret != OFFLOAD_SUCCESS) { 820 REPORT("Deallocating data from device failed.\n"); 821 return OFFLOAD_FAIL; 822 } 823 } 824 825 return OFFLOAD_SUCCESS; 826 } 827 828 static int targetDataContiguous(ident_t *loc, DeviceTy &Device, void *ArgsBase, 829 void *HstPtrBegin, int64_t ArgSize, 830 int64_t ArgType, AsyncInfoTy &AsyncInfo) { 831 TIMESCOPE_WITH_IDENT(loc); 832 bool IsLast, IsHostPtr; 833 void *TgtPtrBegin = Device.getTgtPtrBegin(HstPtrBegin, ArgSize, IsLast, false, 834 IsHostPtr, /*MustContain=*/true); 835 if (!TgtPtrBegin) { 836 DP("hst data:" DPxMOD " not found, becomes a noop\n", DPxPTR(HstPtrBegin)); 837 if (ArgType & OMP_TGT_MAPTYPE_PRESENT) { 838 MESSAGE("device mapping required by 'present' motion modifier does not " 839 "exist for host address " DPxMOD " (%" PRId64 " bytes)", 840 DPxPTR(HstPtrBegin), ArgSize); 841 return OFFLOAD_FAIL; 842 } 843 return OFFLOAD_SUCCESS; 844 } 845 846 if (PM->RTLs.RequiresFlags & OMP_REQ_UNIFIED_SHARED_MEMORY && 847 TgtPtrBegin == HstPtrBegin) { 848 DP("hst data:" DPxMOD " unified and shared, becomes a noop\n", 849 DPxPTR(HstPtrBegin)); 850 return OFFLOAD_SUCCESS; 851 } 852 853 if (ArgType & OMP_TGT_MAPTYPE_FROM) { 854 DP("Moving %" PRId64 " bytes (tgt:" DPxMOD ") -> (hst:" DPxMOD ")\n", 855 ArgSize, DPxPTR(TgtPtrBegin), DPxPTR(HstPtrBegin)); 856 int Ret = Device.retrieveData(HstPtrBegin, TgtPtrBegin, ArgSize, AsyncInfo); 857 if (Ret != OFFLOAD_SUCCESS) { 858 REPORT("Copying data from device failed.\n"); 859 return OFFLOAD_FAIL; 860 } 861 862 uintptr_t LB = (uintptr_t)HstPtrBegin; 863 uintptr_t UB = (uintptr_t)HstPtrBegin + ArgSize; 864 Device.ShadowMtx.lock(); 865 for (ShadowPtrListTy::iterator IT = Device.ShadowPtrMap.begin(); 866 IT != Device.ShadowPtrMap.end(); ++IT) { 867 void **ShadowHstPtrAddr = (void **)IT->first; 868 if ((uintptr_t)ShadowHstPtrAddr < LB) 869 continue; 870 if ((uintptr_t)ShadowHstPtrAddr >= UB) 871 break; 872 DP("Restoring original host pointer value " DPxMOD 873 " for host pointer " DPxMOD "\n", 874 DPxPTR(IT->second.HstPtrVal), DPxPTR(ShadowHstPtrAddr)); 875 *ShadowHstPtrAddr = IT->second.HstPtrVal; 876 } 877 Device.ShadowMtx.unlock(); 878 } 879 880 if (ArgType & OMP_TGT_MAPTYPE_TO) { 881 DP("Moving %" PRId64 " bytes (hst:" DPxMOD ") -> (tgt:" DPxMOD ")\n", 882 ArgSize, DPxPTR(HstPtrBegin), DPxPTR(TgtPtrBegin)); 883 int Ret = Device.submitData(TgtPtrBegin, HstPtrBegin, ArgSize, AsyncInfo); 884 if (Ret != OFFLOAD_SUCCESS) { 885 REPORT("Copying data to device failed.\n"); 886 return OFFLOAD_FAIL; 887 } 888 889 uintptr_t LB = (uintptr_t)HstPtrBegin; 890 uintptr_t UB = (uintptr_t)HstPtrBegin + ArgSize; 891 Device.ShadowMtx.lock(); 892 for (ShadowPtrListTy::iterator IT = Device.ShadowPtrMap.begin(); 893 IT != Device.ShadowPtrMap.end(); ++IT) { 894 void **ShadowHstPtrAddr = (void **)IT->first; 895 if ((uintptr_t)ShadowHstPtrAddr < LB) 896 continue; 897 if ((uintptr_t)ShadowHstPtrAddr >= UB) 898 break; 899 DP("Restoring original target pointer value " DPxMOD " for target " 900 "pointer " DPxMOD "\n", 901 DPxPTR(IT->second.TgtPtrVal), DPxPTR(IT->second.TgtPtrAddr)); 902 Ret = Device.submitData(IT->second.TgtPtrAddr, &IT->second.TgtPtrVal, 903 sizeof(void *), AsyncInfo); 904 if (Ret != OFFLOAD_SUCCESS) { 905 REPORT("Copying data to device failed.\n"); 906 Device.ShadowMtx.unlock(); 907 return OFFLOAD_FAIL; 908 } 909 } 910 Device.ShadowMtx.unlock(); 911 } 912 return OFFLOAD_SUCCESS; 913 } 914 915 static int targetDataNonContiguous(ident_t *loc, DeviceTy &Device, 916 void *ArgsBase, 917 __tgt_target_non_contig *NonContig, 918 uint64_t Size, int64_t ArgType, 919 int CurrentDim, int DimSize, uint64_t Offset, 920 AsyncInfoTy &AsyncInfo) { 921 TIMESCOPE_WITH_IDENT(loc); 922 int Ret = OFFLOAD_SUCCESS; 923 if (CurrentDim < DimSize) { 924 for (unsigned int I = 0; I < NonContig[CurrentDim].Count; ++I) { 925 uint64_t CurOffset = 926 (NonContig[CurrentDim].Offset + I) * NonContig[CurrentDim].Stride; 927 // we only need to transfer the first element for the last dimension 928 // since we've already got a contiguous piece. 929 if (CurrentDim != DimSize - 1 || I == 0) { 930 Ret = targetDataNonContiguous(loc, Device, ArgsBase, NonContig, Size, 931 ArgType, CurrentDim + 1, DimSize, 932 Offset + CurOffset, AsyncInfo); 933 // Stop the whole process if any contiguous piece returns anything 934 // other than OFFLOAD_SUCCESS. 935 if (Ret != OFFLOAD_SUCCESS) 936 return Ret; 937 } 938 } 939 } else { 940 char *Ptr = (char *)ArgsBase + Offset; 941 DP("Transfer of non-contiguous : host ptr " DPxMOD " offset %" PRIu64 942 " len %" PRIu64 "\n", 943 DPxPTR(Ptr), Offset, Size); 944 Ret = targetDataContiguous(loc, Device, ArgsBase, Ptr, Size, ArgType, 945 AsyncInfo); 946 } 947 return Ret; 948 } 949 950 static int getNonContigMergedDimension(__tgt_target_non_contig *NonContig, 951 int32_t DimSize) { 952 int RemovedDim = 0; 953 for (int I = DimSize - 1; I > 0; --I) { 954 if (NonContig[I].Count * NonContig[I].Stride == NonContig[I - 1].Stride) 955 RemovedDim++; 956 } 957 return RemovedDim; 958 } 959 960 /// Internal function to pass data to/from the target. 961 int targetDataUpdate(ident_t *loc, DeviceTy &Device, int32_t ArgNum, 962 void **ArgsBase, void **Args, int64_t *ArgSizes, 963 int64_t *ArgTypes, map_var_info_t *ArgNames, 964 void **ArgMappers, AsyncInfoTy &AsyncInfo, bool) { 965 // process each input. 966 for (int32_t I = 0; I < ArgNum; ++I) { 967 if ((ArgTypes[I] & OMP_TGT_MAPTYPE_LITERAL) || 968 (ArgTypes[I] & OMP_TGT_MAPTYPE_PRIVATE)) 969 continue; 970 971 if (ArgMappers && ArgMappers[I]) { 972 // Instead of executing the regular path of targetDataUpdate, call the 973 // targetDataMapper variant which will call targetDataUpdate again 974 // with new arguments. 975 DP("Calling targetDataMapper for the %dth argument\n", I); 976 977 map_var_info_t ArgName = (!ArgNames) ? nullptr : ArgNames[I]; 978 int Ret = targetDataMapper(loc, Device, ArgsBase[I], Args[I], ArgSizes[I], 979 ArgTypes[I], ArgName, ArgMappers[I], AsyncInfo, 980 targetDataUpdate); 981 982 if (Ret != OFFLOAD_SUCCESS) { 983 REPORT("Call to targetDataUpdate via targetDataMapper for custom mapper" 984 " failed.\n"); 985 return OFFLOAD_FAIL; 986 } 987 988 // Skip the rest of this function, continue to the next argument. 989 continue; 990 } 991 992 int Ret = OFFLOAD_SUCCESS; 993 994 if (ArgTypes[I] & OMP_TGT_MAPTYPE_NON_CONTIG) { 995 __tgt_target_non_contig *NonContig = (__tgt_target_non_contig *)Args[I]; 996 int32_t DimSize = ArgSizes[I]; 997 uint64_t Size = 998 NonContig[DimSize - 1].Count * NonContig[DimSize - 1].Stride; 999 int32_t MergedDim = getNonContigMergedDimension(NonContig, DimSize); 1000 Ret = targetDataNonContiguous( 1001 loc, Device, ArgsBase[I], NonContig, Size, ArgTypes[I], 1002 /*current_dim=*/0, DimSize - MergedDim, /*offset=*/0, AsyncInfo); 1003 } else { 1004 Ret = targetDataContiguous(loc, Device, ArgsBase[I], Args[I], ArgSizes[I], 1005 ArgTypes[I], AsyncInfo); 1006 } 1007 if (Ret == OFFLOAD_FAIL) 1008 return OFFLOAD_FAIL; 1009 } 1010 return OFFLOAD_SUCCESS; 1011 } 1012 1013 static const unsigned LambdaMapping = OMP_TGT_MAPTYPE_PTR_AND_OBJ | 1014 OMP_TGT_MAPTYPE_LITERAL | 1015 OMP_TGT_MAPTYPE_IMPLICIT; 1016 static bool isLambdaMapping(int64_t Mapping) { 1017 return (Mapping & LambdaMapping) == LambdaMapping; 1018 } 1019 1020 namespace { 1021 /// Find the table information in the map or look it up in the translation 1022 /// tables. 1023 TableMap *getTableMap(void *HostPtr) { 1024 std::lock_guard<std::mutex> TblMapLock(PM->TblMapMtx); 1025 HostPtrToTableMapTy::iterator TableMapIt = 1026 PM->HostPtrToTableMap.find(HostPtr); 1027 1028 if (TableMapIt != PM->HostPtrToTableMap.end()) 1029 return &TableMapIt->second; 1030 1031 // We don't have a map. So search all the registered libraries. 1032 TableMap *TM = nullptr; 1033 std::lock_guard<std::mutex> TrlTblLock(PM->TrlTblMtx); 1034 for (HostEntriesBeginToTransTableTy::iterator Itr = 1035 PM->HostEntriesBeginToTransTable.begin(); 1036 Itr != PM->HostEntriesBeginToTransTable.end(); ++Itr) { 1037 // get the translation table (which contains all the good info). 1038 TranslationTable *TransTable = &Itr->second; 1039 // iterate over all the host table entries to see if we can locate the 1040 // host_ptr. 1041 __tgt_offload_entry *Cur = TransTable->HostTable.EntriesBegin; 1042 for (uint32_t I = 0; Cur < TransTable->HostTable.EntriesEnd; ++Cur, ++I) { 1043 if (Cur->addr != HostPtr) 1044 continue; 1045 // we got a match, now fill the HostPtrToTableMap so that we 1046 // may avoid this search next time. 1047 TM = &(PM->HostPtrToTableMap)[HostPtr]; 1048 TM->Table = TransTable; 1049 TM->Index = I; 1050 return TM; 1051 } 1052 } 1053 1054 return nullptr; 1055 } 1056 1057 /// Get loop trip count 1058 /// FIXME: This function will not work right if calling 1059 /// __kmpc_push_target_tripcount_mapper in one thread but doing offloading in 1060 /// another thread, which might occur when we call task yield. 1061 uint64_t getLoopTripCount(int64_t DeviceId) { 1062 DeviceTy &Device = PM->Devices[DeviceId]; 1063 uint64_t LoopTripCount = 0; 1064 1065 { 1066 std::lock_guard<std::mutex> TblMapLock(PM->TblMapMtx); 1067 auto I = Device.LoopTripCnt.find(__kmpc_global_thread_num(NULL)); 1068 if (I != Device.LoopTripCnt.end()) { 1069 LoopTripCount = I->second; 1070 Device.LoopTripCnt.erase(I); 1071 DP("loop trip count is %" PRIu64 ".\n", LoopTripCount); 1072 } 1073 } 1074 1075 return LoopTripCount; 1076 } 1077 1078 /// A class manages private arguments in a target region. 1079 class PrivateArgumentManagerTy { 1080 /// A data structure for the information of first-private arguments. We can 1081 /// use this information to optimize data transfer by packing all 1082 /// first-private arguments and transfer them all at once. 1083 struct FirstPrivateArgInfoTy { 1084 /// The index of the element in \p TgtArgs corresponding to the argument 1085 const int Index; 1086 /// Host pointer begin 1087 const char *HstPtrBegin; 1088 /// Host pointer end 1089 const char *HstPtrEnd; 1090 /// Aligned size 1091 const int64_t AlignedSize; 1092 /// Host pointer name 1093 const map_var_info_t HstPtrName = nullptr; 1094 1095 FirstPrivateArgInfoTy(int Index, const void *HstPtr, int64_t Size, 1096 const map_var_info_t HstPtrName = nullptr) 1097 : Index(Index), HstPtrBegin(reinterpret_cast<const char *>(HstPtr)), 1098 HstPtrEnd(HstPtrBegin + Size), AlignedSize(Size + Size % Alignment), 1099 HstPtrName(HstPtrName) {} 1100 }; 1101 1102 /// A vector of target pointers for all private arguments 1103 std::vector<void *> TgtPtrs; 1104 1105 /// A vector of information of all first-private arguments to be packed 1106 std::vector<FirstPrivateArgInfoTy> FirstPrivateArgInfo; 1107 /// Host buffer for all arguments to be packed 1108 std::vector<char> FirstPrivateArgBuffer; 1109 /// The total size of all arguments to be packed 1110 int64_t FirstPrivateArgSize = 0; 1111 1112 /// A reference to the \p DeviceTy object 1113 DeviceTy &Device; 1114 /// A pointer to a \p AsyncInfoTy object 1115 AsyncInfoTy &AsyncInfo; 1116 1117 // TODO: What would be the best value here? Should we make it configurable? 1118 // If the size is larger than this threshold, we will allocate and transfer it 1119 // immediately instead of packing it. 1120 static constexpr const int64_t FirstPrivateArgSizeThreshold = 1024; 1121 1122 public: 1123 /// Constructor 1124 PrivateArgumentManagerTy(DeviceTy &Dev, AsyncInfoTy &AsyncInfo) 1125 : Device(Dev), AsyncInfo(AsyncInfo) {} 1126 1127 /// Add a private argument 1128 int addArg(void *HstPtr, int64_t ArgSize, int64_t ArgOffset, 1129 bool IsFirstPrivate, void *&TgtPtr, int TgtArgsIndex, 1130 const map_var_info_t HstPtrName = nullptr, 1131 const bool AllocImmediately = false) { 1132 // If the argument is not first-private, or its size is greater than a 1133 // predefined threshold, we will allocate memory and issue the transfer 1134 // immediately. 1135 if (ArgSize > FirstPrivateArgSizeThreshold || !IsFirstPrivate || 1136 AllocImmediately) { 1137 TgtPtr = Device.allocData(ArgSize, HstPtr); 1138 if (!TgtPtr) { 1139 DP("Data allocation for %sprivate array " DPxMOD " failed.\n", 1140 (IsFirstPrivate ? "first-" : ""), DPxPTR(HstPtr)); 1141 return OFFLOAD_FAIL; 1142 } 1143 #ifdef OMPTARGET_DEBUG 1144 void *TgtPtrBase = (void *)((intptr_t)TgtPtr + ArgOffset); 1145 DP("Allocated %" PRId64 " bytes of target memory at " DPxMOD 1146 " for %sprivate array " DPxMOD " - pushing target argument " DPxMOD 1147 "\n", 1148 ArgSize, DPxPTR(TgtPtr), (IsFirstPrivate ? "first-" : ""), 1149 DPxPTR(HstPtr), DPxPTR(TgtPtrBase)); 1150 #endif 1151 // If first-private, copy data from host 1152 if (IsFirstPrivate) { 1153 DP("Submitting firstprivate data to the device.\n"); 1154 int Ret = Device.submitData(TgtPtr, HstPtr, ArgSize, AsyncInfo); 1155 if (Ret != OFFLOAD_SUCCESS) { 1156 DP("Copying data to device failed, failed.\n"); 1157 return OFFLOAD_FAIL; 1158 } 1159 } 1160 TgtPtrs.push_back(TgtPtr); 1161 } else { 1162 DP("Firstprivate array " DPxMOD " of size %" PRId64 " will be packed\n", 1163 DPxPTR(HstPtr), ArgSize); 1164 // When reach this point, the argument must meet all following 1165 // requirements: 1166 // 1. Its size does not exceed the threshold (see the comment for 1167 // FirstPrivateArgSizeThreshold); 1168 // 2. It must be first-private (needs to be mapped to target device). 1169 // We will pack all this kind of arguments to transfer them all at once 1170 // to reduce the number of data transfer. We will not take 1171 // non-first-private arguments, aka. private arguments that doesn't need 1172 // to be mapped to target device, into account because data allocation 1173 // can be very efficient with memory manager. 1174 1175 // Placeholder value 1176 TgtPtr = nullptr; 1177 FirstPrivateArgInfo.emplace_back(TgtArgsIndex, HstPtr, ArgSize, 1178 HstPtrName); 1179 FirstPrivateArgSize += FirstPrivateArgInfo.back().AlignedSize; 1180 } 1181 1182 return OFFLOAD_SUCCESS; 1183 } 1184 1185 /// Pack first-private arguments, replace place holder pointers in \p TgtArgs, 1186 /// and start the transfer. 1187 int packAndTransfer(std::vector<void *> &TgtArgs) { 1188 if (!FirstPrivateArgInfo.empty()) { 1189 assert(FirstPrivateArgSize != 0 && 1190 "FirstPrivateArgSize is 0 but FirstPrivateArgInfo is empty"); 1191 FirstPrivateArgBuffer.resize(FirstPrivateArgSize, 0); 1192 auto Itr = FirstPrivateArgBuffer.begin(); 1193 // Copy all host data to this buffer 1194 for (FirstPrivateArgInfoTy &Info : FirstPrivateArgInfo) { 1195 std::copy(Info.HstPtrBegin, Info.HstPtrEnd, Itr); 1196 Itr = std::next(Itr, Info.AlignedSize); 1197 } 1198 // Allocate target memory 1199 void *TgtPtr = 1200 Device.allocData(FirstPrivateArgSize, FirstPrivateArgBuffer.data()); 1201 if (TgtPtr == nullptr) { 1202 DP("Failed to allocate target memory for private arguments.\n"); 1203 return OFFLOAD_FAIL; 1204 } 1205 TgtPtrs.push_back(TgtPtr); 1206 DP("Allocated %" PRId64 " bytes of target memory at " DPxMOD "\n", 1207 FirstPrivateArgSize, DPxPTR(TgtPtr)); 1208 // Transfer data to target device 1209 int Ret = Device.submitData(TgtPtr, FirstPrivateArgBuffer.data(), 1210 FirstPrivateArgSize, AsyncInfo); 1211 if (Ret != OFFLOAD_SUCCESS) { 1212 DP("Failed to submit data of private arguments.\n"); 1213 return OFFLOAD_FAIL; 1214 } 1215 // Fill in all placeholder pointers 1216 auto TP = reinterpret_cast<uintptr_t>(TgtPtr); 1217 for (FirstPrivateArgInfoTy &Info : FirstPrivateArgInfo) { 1218 void *&Ptr = TgtArgs[Info.Index]; 1219 assert(Ptr == nullptr && "Target pointer is already set by mistaken"); 1220 Ptr = reinterpret_cast<void *>(TP); 1221 TP += Info.AlignedSize; 1222 DP("Firstprivate array " DPxMOD " of size %" PRId64 " mapped to " DPxMOD 1223 "\n", 1224 DPxPTR(Info.HstPtrBegin), Info.HstPtrEnd - Info.HstPtrBegin, 1225 DPxPTR(Ptr)); 1226 } 1227 } 1228 1229 return OFFLOAD_SUCCESS; 1230 } 1231 1232 /// Free all target memory allocated for private arguments 1233 int free() { 1234 for (void *P : TgtPtrs) { 1235 int Ret = Device.deleteData(P); 1236 if (Ret != OFFLOAD_SUCCESS) { 1237 DP("Deallocation of (first-)private arrays failed.\n"); 1238 return OFFLOAD_FAIL; 1239 } 1240 } 1241 1242 TgtPtrs.clear(); 1243 1244 return OFFLOAD_SUCCESS; 1245 } 1246 }; 1247 1248 /// Process data before launching the kernel, including calling targetDataBegin 1249 /// to map and transfer data to target device, transferring (first-)private 1250 /// variables. 1251 static int processDataBefore(ident_t *loc, int64_t DeviceId, void *HostPtr, 1252 int32_t ArgNum, void **ArgBases, void **Args, 1253 int64_t *ArgSizes, int64_t *ArgTypes, 1254 map_var_info_t *ArgNames, void **ArgMappers, 1255 std::vector<void *> &TgtArgs, 1256 std::vector<ptrdiff_t> &TgtOffsets, 1257 PrivateArgumentManagerTy &PrivateArgumentManager, 1258 AsyncInfoTy &AsyncInfo) { 1259 TIMESCOPE_WITH_NAME_AND_IDENT("mappingBeforeTargetRegion", loc); 1260 DeviceTy &Device = PM->Devices[DeviceId]; 1261 int Ret = targetDataBegin(loc, Device, ArgNum, ArgBases, Args, ArgSizes, 1262 ArgTypes, ArgNames, ArgMappers, AsyncInfo); 1263 if (Ret != OFFLOAD_SUCCESS) { 1264 REPORT("Call to targetDataBegin failed, abort target.\n"); 1265 return OFFLOAD_FAIL; 1266 } 1267 1268 // List of (first-)private arrays allocated for this target region 1269 std::vector<int> TgtArgsPositions(ArgNum, -1); 1270 1271 for (int32_t I = 0; I < ArgNum; ++I) { 1272 if (!(ArgTypes[I] & OMP_TGT_MAPTYPE_TARGET_PARAM)) { 1273 // This is not a target parameter, do not push it into TgtArgs. 1274 // Check for lambda mapping. 1275 if (isLambdaMapping(ArgTypes[I])) { 1276 assert((ArgTypes[I] & OMP_TGT_MAPTYPE_MEMBER_OF) && 1277 "PTR_AND_OBJ must be also MEMBER_OF."); 1278 unsigned Idx = getParentIndex(ArgTypes[I]); 1279 int TgtIdx = TgtArgsPositions[Idx]; 1280 assert(TgtIdx != -1 && "Base address must be translated already."); 1281 // The parent lambda must be processed already and it must be the last 1282 // in TgtArgs and TgtOffsets arrays. 1283 void *HstPtrVal = Args[I]; 1284 void *HstPtrBegin = ArgBases[I]; 1285 void *HstPtrBase = Args[Idx]; 1286 bool IsLast, IsHostPtr; // unused. 1287 void *TgtPtrBase = 1288 (void *)((intptr_t)TgtArgs[TgtIdx] + TgtOffsets[TgtIdx]); 1289 DP("Parent lambda base " DPxMOD "\n", DPxPTR(TgtPtrBase)); 1290 uint64_t Delta = (uint64_t)HstPtrBegin - (uint64_t)HstPtrBase; 1291 void *TgtPtrBegin = (void *)((uintptr_t)TgtPtrBase + Delta); 1292 void *&PointerTgtPtrBegin = AsyncInfo.getVoidPtrLocation(); 1293 PointerTgtPtrBegin = Device.getTgtPtrBegin(HstPtrVal, ArgSizes[I], 1294 IsLast, false, IsHostPtr); 1295 if (!PointerTgtPtrBegin) { 1296 DP("No lambda captured variable mapped (" DPxMOD ") - ignored\n", 1297 DPxPTR(HstPtrVal)); 1298 continue; 1299 } 1300 if (PM->RTLs.RequiresFlags & OMP_REQ_UNIFIED_SHARED_MEMORY && 1301 TgtPtrBegin == HstPtrBegin) { 1302 DP("Unified memory is active, no need to map lambda captured" 1303 "variable (" DPxMOD ")\n", 1304 DPxPTR(HstPtrVal)); 1305 continue; 1306 } 1307 DP("Update lambda reference (" DPxMOD ") -> [" DPxMOD "]\n", 1308 DPxPTR(PointerTgtPtrBegin), DPxPTR(TgtPtrBegin)); 1309 Ret = Device.submitData(TgtPtrBegin, &PointerTgtPtrBegin, 1310 sizeof(void *), AsyncInfo); 1311 if (Ret != OFFLOAD_SUCCESS) { 1312 REPORT("Copying data to device failed.\n"); 1313 return OFFLOAD_FAIL; 1314 } 1315 } 1316 continue; 1317 } 1318 void *HstPtrBegin = Args[I]; 1319 void *HstPtrBase = ArgBases[I]; 1320 void *TgtPtrBegin; 1321 map_var_info_t HstPtrName = (!ArgNames) ? nullptr : ArgNames[I]; 1322 ptrdiff_t TgtBaseOffset; 1323 bool IsLast, IsHostPtr; // unused. 1324 if (ArgTypes[I] & OMP_TGT_MAPTYPE_LITERAL) { 1325 DP("Forwarding first-private value " DPxMOD " to the target construct\n", 1326 DPxPTR(HstPtrBase)); 1327 TgtPtrBegin = HstPtrBase; 1328 TgtBaseOffset = 0; 1329 } else if (ArgTypes[I] & OMP_TGT_MAPTYPE_PRIVATE) { 1330 TgtBaseOffset = (intptr_t)HstPtrBase - (intptr_t)HstPtrBegin; 1331 const bool IsFirstPrivate = (ArgTypes[I] & OMP_TGT_MAPTYPE_TO); 1332 // If there is a next argument and it depends on the current one, we need 1333 // to allocate the private memory immediately. If this is not the case, 1334 // then the argument can be marked for optimization and packed with the 1335 // other privates. 1336 const bool AllocImmediately = 1337 (I < ArgNum - 1 && (ArgTypes[I + 1] & OMP_TGT_MAPTYPE_MEMBER_OF)); 1338 Ret = PrivateArgumentManager.addArg( 1339 HstPtrBegin, ArgSizes[I], TgtBaseOffset, IsFirstPrivate, TgtPtrBegin, 1340 TgtArgs.size(), HstPtrName, AllocImmediately); 1341 if (Ret != OFFLOAD_SUCCESS) { 1342 REPORT("Failed to process %sprivate argument " DPxMOD "\n", 1343 (IsFirstPrivate ? "first-" : ""), DPxPTR(HstPtrBegin)); 1344 return OFFLOAD_FAIL; 1345 } 1346 } else { 1347 if (ArgTypes[I] & OMP_TGT_MAPTYPE_PTR_AND_OBJ) 1348 HstPtrBase = *reinterpret_cast<void **>(HstPtrBase); 1349 TgtPtrBegin = Device.getTgtPtrBegin(HstPtrBegin, ArgSizes[I], IsLast, 1350 false, IsHostPtr); 1351 TgtBaseOffset = (intptr_t)HstPtrBase - (intptr_t)HstPtrBegin; 1352 #ifdef OMPTARGET_DEBUG 1353 void *TgtPtrBase = (void *)((intptr_t)TgtPtrBegin + TgtBaseOffset); 1354 DP("Obtained target argument " DPxMOD " from host pointer " DPxMOD "\n", 1355 DPxPTR(TgtPtrBase), DPxPTR(HstPtrBegin)); 1356 #endif 1357 } 1358 TgtArgsPositions[I] = TgtArgs.size(); 1359 TgtArgs.push_back(TgtPtrBegin); 1360 TgtOffsets.push_back(TgtBaseOffset); 1361 } 1362 1363 assert(TgtArgs.size() == TgtOffsets.size() && 1364 "Size mismatch in arguments and offsets"); 1365 1366 // Pack and transfer first-private arguments 1367 Ret = PrivateArgumentManager.packAndTransfer(TgtArgs); 1368 if (Ret != OFFLOAD_SUCCESS) { 1369 DP("Failed to pack and transfer first private arguments\n"); 1370 return OFFLOAD_FAIL; 1371 } 1372 1373 return OFFLOAD_SUCCESS; 1374 } 1375 1376 /// Process data after launching the kernel, including transferring data back to 1377 /// host if needed and deallocating target memory of (first-)private variables. 1378 static int processDataAfter(ident_t *loc, int64_t DeviceId, void *HostPtr, 1379 int32_t ArgNum, void **ArgBases, void **Args, 1380 int64_t *ArgSizes, int64_t *ArgTypes, 1381 map_var_info_t *ArgNames, void **ArgMappers, 1382 PrivateArgumentManagerTy &PrivateArgumentManager, 1383 AsyncInfoTy &AsyncInfo) { 1384 TIMESCOPE_WITH_NAME_AND_IDENT("mappingAfterTargetRegion", loc); 1385 DeviceTy &Device = PM->Devices[DeviceId]; 1386 1387 // Move data from device. 1388 int Ret = targetDataEnd(loc, Device, ArgNum, ArgBases, Args, ArgSizes, 1389 ArgTypes, ArgNames, ArgMappers, AsyncInfo); 1390 if (Ret != OFFLOAD_SUCCESS) { 1391 REPORT("Call to targetDataEnd failed, abort target.\n"); 1392 return OFFLOAD_FAIL; 1393 } 1394 1395 // Free target memory for private arguments 1396 Ret = PrivateArgumentManager.free(); 1397 if (Ret != OFFLOAD_SUCCESS) { 1398 REPORT("Failed to deallocate target memory for private args\n"); 1399 return OFFLOAD_FAIL; 1400 } 1401 1402 return OFFLOAD_SUCCESS; 1403 } 1404 } // namespace 1405 1406 /// performs the same actions as data_begin in case arg_num is 1407 /// non-zero and initiates run of the offloaded region on the target platform; 1408 /// if arg_num is non-zero after the region execution is done it also 1409 /// performs the same action as data_update and data_end above. This function 1410 /// returns 0 if it was able to transfer the execution to a target and an 1411 /// integer different from zero otherwise. 1412 int target(ident_t *loc, DeviceTy &Device, void *HostPtr, int32_t ArgNum, 1413 void **ArgBases, void **Args, int64_t *ArgSizes, int64_t *ArgTypes, 1414 map_var_info_t *ArgNames, void **ArgMappers, int32_t TeamNum, 1415 int32_t ThreadLimit, int IsTeamConstruct, AsyncInfoTy &AsyncInfo) { 1416 int32_t DeviceId = Device.DeviceID; 1417 1418 TableMap *TM = getTableMap(HostPtr); 1419 // No map for this host pointer found! 1420 if (!TM) { 1421 REPORT("Host ptr " DPxMOD " does not have a matching target pointer.\n", 1422 DPxPTR(HostPtr)); 1423 return OFFLOAD_FAIL; 1424 } 1425 1426 // get target table. 1427 __tgt_target_table *TargetTable = nullptr; 1428 { 1429 std::lock_guard<std::mutex> TrlTblLock(PM->TrlTblMtx); 1430 assert(TM->Table->TargetsTable.size() > (size_t)DeviceId && 1431 "Not expecting a device ID outside the table's bounds!"); 1432 TargetTable = TM->Table->TargetsTable[DeviceId]; 1433 } 1434 assert(TargetTable && "Global data has not been mapped\n"); 1435 1436 std::vector<void *> TgtArgs; 1437 std::vector<ptrdiff_t> TgtOffsets; 1438 1439 PrivateArgumentManagerTy PrivateArgumentManager(Device, AsyncInfo); 1440 1441 int Ret; 1442 if (ArgNum) { 1443 // Process data, such as data mapping, before launching the kernel 1444 Ret = processDataBefore(loc, DeviceId, HostPtr, ArgNum, ArgBases, Args, 1445 ArgSizes, ArgTypes, ArgNames, ArgMappers, TgtArgs, 1446 TgtOffsets, PrivateArgumentManager, AsyncInfo); 1447 if (Ret != OFFLOAD_SUCCESS) { 1448 REPORT("Failed to process data before launching the kernel.\n"); 1449 return OFFLOAD_FAIL; 1450 } 1451 } 1452 1453 // Get loop trip count 1454 uint64_t LoopTripCount = getLoopTripCount(DeviceId); 1455 1456 // Launch device execution. 1457 void *TgtEntryPtr = TargetTable->EntriesBegin[TM->Index].addr; 1458 DP("Launching target execution %s with pointer " DPxMOD " (index=%d).\n", 1459 TargetTable->EntriesBegin[TM->Index].name, DPxPTR(TgtEntryPtr), TM->Index); 1460 1461 { 1462 TIMESCOPE_WITH_NAME_AND_IDENT( 1463 IsTeamConstruct ? "runTargetTeamRegion" : "runTargetRegion", loc); 1464 if (IsTeamConstruct) 1465 Ret = Device.runTeamRegion(TgtEntryPtr, &TgtArgs[0], &TgtOffsets[0], 1466 TgtArgs.size(), TeamNum, ThreadLimit, 1467 LoopTripCount, AsyncInfo); 1468 else 1469 Ret = Device.runRegion(TgtEntryPtr, &TgtArgs[0], &TgtOffsets[0], 1470 TgtArgs.size(), AsyncInfo); 1471 } 1472 1473 if (Ret != OFFLOAD_SUCCESS) { 1474 REPORT("Executing target region abort target.\n"); 1475 return OFFLOAD_FAIL; 1476 } 1477 1478 if (ArgNum) { 1479 // Transfer data back and deallocate target memory for (first-)private 1480 // variables 1481 Ret = processDataAfter(loc, DeviceId, HostPtr, ArgNum, ArgBases, Args, 1482 ArgSizes, ArgTypes, ArgNames, ArgMappers, 1483 PrivateArgumentManager, AsyncInfo); 1484 if (Ret != OFFLOAD_SUCCESS) { 1485 REPORT("Failed to process data after launching the kernel.\n"); 1486 return OFFLOAD_FAIL; 1487 } 1488 } 1489 1490 return OFFLOAD_SUCCESS; 1491 } 1492