1 /* 2 * kmp_tasking.cpp -- OpenMP 3.0 tasking support. 3 */ 4 5 //===----------------------------------------------------------------------===// 6 // 7 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 8 // See https://llvm.org/LICENSE.txt for license information. 9 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "kmp.h" 14 #include "kmp_i18n.h" 15 #include "kmp_itt.h" 16 #include "kmp_stats.h" 17 #include "kmp_wait_release.h" 18 #include "kmp_taskdeps.h" 19 20 #if OMPT_SUPPORT 21 #include "ompt-specific.h" 22 #endif 23 24 #include "tsan_annotations.h" 25 26 /* forward declaration */ 27 static void __kmp_enable_tasking(kmp_task_team_t *task_team, 28 kmp_info_t *this_thr); 29 static void __kmp_alloc_task_deque(kmp_info_t *thread, 30 kmp_thread_data_t *thread_data); 31 static int __kmp_realloc_task_threads_data(kmp_info_t *thread, 32 kmp_task_team_t *task_team); 33 34 #if OMP_45_ENABLED 35 static void __kmp_bottom_half_finish_proxy(kmp_int32 gtid, kmp_task_t *ptask); 36 #endif 37 38 #ifdef BUILD_TIED_TASK_STACK 39 40 // __kmp_trace_task_stack: print the tied tasks from the task stack in order 41 // from top do bottom 42 // 43 // gtid: global thread identifier for thread containing stack 44 // thread_data: thread data for task team thread containing stack 45 // threshold: value above which the trace statement triggers 46 // location: string identifying call site of this function (for trace) 47 static void __kmp_trace_task_stack(kmp_int32 gtid, 48 kmp_thread_data_t *thread_data, 49 int threshold, char *location) { 50 kmp_task_stack_t *task_stack = &thread_data->td.td_susp_tied_tasks; 51 kmp_taskdata_t **stack_top = task_stack->ts_top; 52 kmp_int32 entries = task_stack->ts_entries; 53 kmp_taskdata_t *tied_task; 54 55 KA_TRACE( 56 threshold, 57 ("__kmp_trace_task_stack(start): location = %s, gtid = %d, entries = %d, " 58 "first_block = %p, stack_top = %p \n", 59 location, gtid, entries, task_stack->ts_first_block, stack_top)); 60 61 KMP_DEBUG_ASSERT(stack_top != NULL); 62 KMP_DEBUG_ASSERT(entries > 0); 63 64 while (entries != 0) { 65 KMP_DEBUG_ASSERT(stack_top != &task_stack->ts_first_block.sb_block[0]); 66 // fix up ts_top if we need to pop from previous block 67 if (entries & TASK_STACK_INDEX_MASK == 0) { 68 kmp_stack_block_t *stack_block = (kmp_stack_block_t *)(stack_top); 69 70 stack_block = stack_block->sb_prev; 71 stack_top = &stack_block->sb_block[TASK_STACK_BLOCK_SIZE]; 72 } 73 74 // finish bookkeeping 75 stack_top--; 76 entries--; 77 78 tied_task = *stack_top; 79 80 KMP_DEBUG_ASSERT(tied_task != NULL); 81 KMP_DEBUG_ASSERT(tied_task->td_flags.tasktype == TASK_TIED); 82 83 KA_TRACE(threshold, 84 ("__kmp_trace_task_stack(%s): gtid=%d, entry=%d, " 85 "stack_top=%p, tied_task=%p\n", 86 location, gtid, entries, stack_top, tied_task)); 87 } 88 KMP_DEBUG_ASSERT(stack_top == &task_stack->ts_first_block.sb_block[0]); 89 90 KA_TRACE(threshold, 91 ("__kmp_trace_task_stack(exit): location = %s, gtid = %d\n", 92 location, gtid)); 93 } 94 95 // __kmp_init_task_stack: initialize the task stack for the first time 96 // after a thread_data structure is created. 97 // It should not be necessary to do this again (assuming the stack works). 98 // 99 // gtid: global thread identifier of calling thread 100 // thread_data: thread data for task team thread containing stack 101 static void __kmp_init_task_stack(kmp_int32 gtid, 102 kmp_thread_data_t *thread_data) { 103 kmp_task_stack_t *task_stack = &thread_data->td.td_susp_tied_tasks; 104 kmp_stack_block_t *first_block; 105 106 // set up the first block of the stack 107 first_block = &task_stack->ts_first_block; 108 task_stack->ts_top = (kmp_taskdata_t **)first_block; 109 memset((void *)first_block, '\0', 110 TASK_STACK_BLOCK_SIZE * sizeof(kmp_taskdata_t *)); 111 112 // initialize the stack to be empty 113 task_stack->ts_entries = TASK_STACK_EMPTY; 114 first_block->sb_next = NULL; 115 first_block->sb_prev = NULL; 116 } 117 118 // __kmp_free_task_stack: free the task stack when thread_data is destroyed. 119 // 120 // gtid: global thread identifier for calling thread 121 // thread_data: thread info for thread containing stack 122 static void __kmp_free_task_stack(kmp_int32 gtid, 123 kmp_thread_data_t *thread_data) { 124 kmp_task_stack_t *task_stack = &thread_data->td.td_susp_tied_tasks; 125 kmp_stack_block_t *stack_block = &task_stack->ts_first_block; 126 127 KMP_DEBUG_ASSERT(task_stack->ts_entries == TASK_STACK_EMPTY); 128 // free from the second block of the stack 129 while (stack_block != NULL) { 130 kmp_stack_block_t *next_block = (stack_block) ? stack_block->sb_next : NULL; 131 132 stack_block->sb_next = NULL; 133 stack_block->sb_prev = NULL; 134 if (stack_block != &task_stack->ts_first_block) { 135 __kmp_thread_free(thread, 136 stack_block); // free the block, if not the first 137 } 138 stack_block = next_block; 139 } 140 // initialize the stack to be empty 141 task_stack->ts_entries = 0; 142 task_stack->ts_top = NULL; 143 } 144 145 // __kmp_push_task_stack: Push the tied task onto the task stack. 146 // Grow the stack if necessary by allocating another block. 147 // 148 // gtid: global thread identifier for calling thread 149 // thread: thread info for thread containing stack 150 // tied_task: the task to push on the stack 151 static void __kmp_push_task_stack(kmp_int32 gtid, kmp_info_t *thread, 152 kmp_taskdata_t *tied_task) { 153 // GEH - need to consider what to do if tt_threads_data not allocated yet 154 kmp_thread_data_t *thread_data = 155 &thread->th.th_task_team->tt.tt_threads_data[__kmp_tid_from_gtid(gtid)]; 156 kmp_task_stack_t *task_stack = &thread_data->td.td_susp_tied_tasks; 157 158 if (tied_task->td_flags.team_serial || tied_task->td_flags.tasking_ser) { 159 return; // Don't push anything on stack if team or team tasks are serialized 160 } 161 162 KMP_DEBUG_ASSERT(tied_task->td_flags.tasktype == TASK_TIED); 163 KMP_DEBUG_ASSERT(task_stack->ts_top != NULL); 164 165 KA_TRACE(20, 166 ("__kmp_push_task_stack(enter): GTID: %d; THREAD: %p; TASK: %p\n", 167 gtid, thread, tied_task)); 168 // Store entry 169 *(task_stack->ts_top) = tied_task; 170 171 // Do bookkeeping for next push 172 task_stack->ts_top++; 173 task_stack->ts_entries++; 174 175 if (task_stack->ts_entries & TASK_STACK_INDEX_MASK == 0) { 176 // Find beginning of this task block 177 kmp_stack_block_t *stack_block = 178 (kmp_stack_block_t *)(task_stack->ts_top - TASK_STACK_BLOCK_SIZE); 179 180 // Check if we already have a block 181 if (stack_block->sb_next != 182 NULL) { // reset ts_top to beginning of next block 183 task_stack->ts_top = &stack_block->sb_next->sb_block[0]; 184 } else { // Alloc new block and link it up 185 kmp_stack_block_t *new_block = (kmp_stack_block_t *)__kmp_thread_calloc( 186 thread, sizeof(kmp_stack_block_t)); 187 188 task_stack->ts_top = &new_block->sb_block[0]; 189 stack_block->sb_next = new_block; 190 new_block->sb_prev = stack_block; 191 new_block->sb_next = NULL; 192 193 KA_TRACE( 194 30, 195 ("__kmp_push_task_stack(): GTID: %d; TASK: %p; Alloc new block: %p\n", 196 gtid, tied_task, new_block)); 197 } 198 } 199 KA_TRACE(20, ("__kmp_push_task_stack(exit): GTID: %d; TASK: %p\n", gtid, 200 tied_task)); 201 } 202 203 // __kmp_pop_task_stack: Pop the tied task from the task stack. Don't return 204 // the task, just check to make sure it matches the ending task passed in. 205 // 206 // gtid: global thread identifier for the calling thread 207 // thread: thread info structure containing stack 208 // tied_task: the task popped off the stack 209 // ending_task: the task that is ending (should match popped task) 210 static void __kmp_pop_task_stack(kmp_int32 gtid, kmp_info_t *thread, 211 kmp_taskdata_t *ending_task) { 212 // GEH - need to consider what to do if tt_threads_data not allocated yet 213 kmp_thread_data_t *thread_data = 214 &thread->th.th_task_team->tt_threads_data[__kmp_tid_from_gtid(gtid)]; 215 kmp_task_stack_t *task_stack = &thread_data->td.td_susp_tied_tasks; 216 kmp_taskdata_t *tied_task; 217 218 if (ending_task->td_flags.team_serial || ending_task->td_flags.tasking_ser) { 219 // Don't pop anything from stack if team or team tasks are serialized 220 return; 221 } 222 223 KMP_DEBUG_ASSERT(task_stack->ts_top != NULL); 224 KMP_DEBUG_ASSERT(task_stack->ts_entries > 0); 225 226 KA_TRACE(20, ("__kmp_pop_task_stack(enter): GTID: %d; THREAD: %p\n", gtid, 227 thread)); 228 229 // fix up ts_top if we need to pop from previous block 230 if (task_stack->ts_entries & TASK_STACK_INDEX_MASK == 0) { 231 kmp_stack_block_t *stack_block = (kmp_stack_block_t *)(task_stack->ts_top); 232 233 stack_block = stack_block->sb_prev; 234 task_stack->ts_top = &stack_block->sb_block[TASK_STACK_BLOCK_SIZE]; 235 } 236 237 // finish bookkeeping 238 task_stack->ts_top--; 239 task_stack->ts_entries--; 240 241 tied_task = *(task_stack->ts_top); 242 243 KMP_DEBUG_ASSERT(tied_task != NULL); 244 KMP_DEBUG_ASSERT(tied_task->td_flags.tasktype == TASK_TIED); 245 KMP_DEBUG_ASSERT(tied_task == ending_task); // If we built the stack correctly 246 247 KA_TRACE(20, ("__kmp_pop_task_stack(exit): GTID: %d; TASK: %p\n", gtid, 248 tied_task)); 249 return; 250 } 251 #endif /* BUILD_TIED_TASK_STACK */ 252 253 // returns 1 if new task is allowed to execute, 0 otherwise 254 // checks Task Scheduling constraint (if requested) and 255 // mutexinoutset dependencies if any 256 static bool __kmp_task_is_allowed(int gtid, const kmp_int32 is_constrained, 257 const kmp_taskdata_t *tasknew, 258 const kmp_taskdata_t *taskcurr) { 259 if (is_constrained && (tasknew->td_flags.tiedness == TASK_TIED)) { 260 // Check if the candidate obeys the Task Scheduling Constraints (TSC) 261 // only descendant of all deferred tied tasks can be scheduled, checking 262 // the last one is enough, as it in turn is the descendant of all others 263 kmp_taskdata_t *current = taskcurr->td_last_tied; 264 KMP_DEBUG_ASSERT(current != NULL); 265 // check if the task is not suspended on barrier 266 if (current->td_flags.tasktype == TASK_EXPLICIT || 267 current->td_taskwait_thread > 0) { // <= 0 on barrier 268 kmp_int32 level = current->td_level; 269 kmp_taskdata_t *parent = tasknew->td_parent; 270 while (parent != current && parent->td_level > level) { 271 // check generation up to the level of the current task 272 parent = parent->td_parent; 273 KMP_DEBUG_ASSERT(parent != NULL); 274 } 275 if (parent != current) 276 return false; 277 } 278 } 279 // Check mutexinoutset dependencies, acquire locks 280 kmp_depnode_t *node = tasknew->td_depnode; 281 if (node && (node->dn.mtx_num_locks > 0)) { 282 for (int i = 0; i < node->dn.mtx_num_locks; ++i) { 283 KMP_DEBUG_ASSERT(node->dn.mtx_locks[i] != NULL); 284 if (__kmp_test_lock(node->dn.mtx_locks[i], gtid)) 285 continue; 286 // could not get the lock, release previous locks 287 for (int j = i - 1; j >= 0; --j) 288 __kmp_release_lock(node->dn.mtx_locks[j], gtid); 289 return false; 290 } 291 // negative num_locks means all locks acquired successfully 292 node->dn.mtx_num_locks = -node->dn.mtx_num_locks; 293 } 294 return true; 295 } 296 297 // __kmp_realloc_task_deque: 298 // Re-allocates a task deque for a particular thread, copies the content from 299 // the old deque and adjusts the necessary data structures relating to the 300 // deque. This operation must be done with the deque_lock being held 301 static void __kmp_realloc_task_deque(kmp_info_t *thread, 302 kmp_thread_data_t *thread_data) { 303 kmp_int32 size = TASK_DEQUE_SIZE(thread_data->td); 304 kmp_int32 new_size = 2 * size; 305 306 KE_TRACE(10, ("__kmp_realloc_task_deque: T#%d reallocating deque[from %d to " 307 "%d] for thread_data %p\n", 308 __kmp_gtid_from_thread(thread), size, new_size, thread_data)); 309 310 kmp_taskdata_t **new_deque = 311 (kmp_taskdata_t **)__kmp_allocate(new_size * sizeof(kmp_taskdata_t *)); 312 313 int i, j; 314 for (i = thread_data->td.td_deque_head, j = 0; j < size; 315 i = (i + 1) & TASK_DEQUE_MASK(thread_data->td), j++) 316 new_deque[j] = thread_data->td.td_deque[i]; 317 318 __kmp_free(thread_data->td.td_deque); 319 320 thread_data->td.td_deque_head = 0; 321 thread_data->td.td_deque_tail = size; 322 thread_data->td.td_deque = new_deque; 323 thread_data->td.td_deque_size = new_size; 324 } 325 326 // __kmp_push_task: Add a task to the thread's deque 327 static kmp_int32 __kmp_push_task(kmp_int32 gtid, kmp_task_t *task) { 328 kmp_info_t *thread = __kmp_threads[gtid]; 329 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 330 kmp_task_team_t *task_team = thread->th.th_task_team; 331 kmp_int32 tid = __kmp_tid_from_gtid(gtid); 332 kmp_thread_data_t *thread_data; 333 334 KA_TRACE(20, 335 ("__kmp_push_task: T#%d trying to push task %p.\n", gtid, taskdata)); 336 337 if (taskdata->td_flags.tiedness == TASK_UNTIED) { 338 // untied task needs to increment counter so that the task structure is not 339 // freed prematurely 340 kmp_int32 counter = 1 + KMP_ATOMIC_INC(&taskdata->td_untied_count); 341 KMP_DEBUG_USE_VAR(counter); 342 KA_TRACE( 343 20, 344 ("__kmp_push_task: T#%d untied_count (%d) incremented for task %p\n", 345 gtid, counter, taskdata)); 346 } 347 348 // The first check avoids building task_team thread data if serialized 349 if (taskdata->td_flags.task_serial) { 350 KA_TRACE(20, ("__kmp_push_task: T#%d team serialized; returning " 351 "TASK_NOT_PUSHED for task %p\n", 352 gtid, taskdata)); 353 return TASK_NOT_PUSHED; 354 } 355 356 // Now that serialized tasks have returned, we can assume that we are not in 357 // immediate exec mode 358 KMP_DEBUG_ASSERT(__kmp_tasking_mode != tskm_immediate_exec); 359 if (!KMP_TASKING_ENABLED(task_team)) { 360 __kmp_enable_tasking(task_team, thread); 361 } 362 KMP_DEBUG_ASSERT(TCR_4(task_team->tt.tt_found_tasks) == TRUE); 363 KMP_DEBUG_ASSERT(TCR_PTR(task_team->tt.tt_threads_data) != NULL); 364 365 // Find tasking deque specific to encountering thread 366 thread_data = &task_team->tt.tt_threads_data[tid]; 367 368 // No lock needed since only owner can allocate 369 if (thread_data->td.td_deque == NULL) { 370 __kmp_alloc_task_deque(thread, thread_data); 371 } 372 373 int locked = 0; 374 // Check if deque is full 375 if (TCR_4(thread_data->td.td_deque_ntasks) >= 376 TASK_DEQUE_SIZE(thread_data->td)) { 377 if (__kmp_task_is_allowed(gtid, __kmp_task_stealing_constraint, taskdata, 378 thread->th.th_current_task)) { 379 KA_TRACE(20, ("__kmp_push_task: T#%d deque is full; returning " 380 "TASK_NOT_PUSHED for task %p\n", 381 gtid, taskdata)); 382 return TASK_NOT_PUSHED; 383 } else { 384 __kmp_acquire_bootstrap_lock(&thread_data->td.td_deque_lock); 385 locked = 1; 386 // expand deque to push the task which is not allowed to execute 387 __kmp_realloc_task_deque(thread, thread_data); 388 } 389 } 390 // Lock the deque for the task push operation 391 if (!locked) { 392 __kmp_acquire_bootstrap_lock(&thread_data->td.td_deque_lock); 393 #if OMP_45_ENABLED 394 // Need to recheck as we can get a proxy task from thread outside of OpenMP 395 if (TCR_4(thread_data->td.td_deque_ntasks) >= 396 TASK_DEQUE_SIZE(thread_data->td)) { 397 if (__kmp_task_is_allowed(gtid, __kmp_task_stealing_constraint, taskdata, 398 thread->th.th_current_task)) { 399 __kmp_release_bootstrap_lock(&thread_data->td.td_deque_lock); 400 KA_TRACE(20, ("__kmp_push_task: T#%d deque is full on 2nd check; " 401 "returning TASK_NOT_PUSHED for task %p\n", 402 gtid, taskdata)); 403 return TASK_NOT_PUSHED; 404 } else { 405 // expand deque to push the task which is not allowed to execute 406 __kmp_realloc_task_deque(thread, thread_data); 407 } 408 } 409 #endif 410 } 411 // Must have room since no thread can add tasks but calling thread 412 KMP_DEBUG_ASSERT(TCR_4(thread_data->td.td_deque_ntasks) < 413 TASK_DEQUE_SIZE(thread_data->td)); 414 415 thread_data->td.td_deque[thread_data->td.td_deque_tail] = 416 taskdata; // Push taskdata 417 // Wrap index. 418 thread_data->td.td_deque_tail = 419 (thread_data->td.td_deque_tail + 1) & TASK_DEQUE_MASK(thread_data->td); 420 TCW_4(thread_data->td.td_deque_ntasks, 421 TCR_4(thread_data->td.td_deque_ntasks) + 1); // Adjust task count 422 423 KA_TRACE(20, ("__kmp_push_task: T#%d returning TASK_SUCCESSFULLY_PUSHED: " 424 "task=%p ntasks=%d head=%u tail=%u\n", 425 gtid, taskdata, thread_data->td.td_deque_ntasks, 426 thread_data->td.td_deque_head, thread_data->td.td_deque_tail)); 427 428 __kmp_release_bootstrap_lock(&thread_data->td.td_deque_lock); 429 430 return TASK_SUCCESSFULLY_PUSHED; 431 } 432 433 // __kmp_pop_current_task_from_thread: set up current task from called thread 434 // when team ends 435 // 436 // this_thr: thread structure to set current_task in. 437 void __kmp_pop_current_task_from_thread(kmp_info_t *this_thr) { 438 KF_TRACE(10, ("__kmp_pop_current_task_from_thread(enter): T#%d " 439 "this_thread=%p, curtask=%p, " 440 "curtask_parent=%p\n", 441 0, this_thr, this_thr->th.th_current_task, 442 this_thr->th.th_current_task->td_parent)); 443 444 this_thr->th.th_current_task = this_thr->th.th_current_task->td_parent; 445 446 KF_TRACE(10, ("__kmp_pop_current_task_from_thread(exit): T#%d " 447 "this_thread=%p, curtask=%p, " 448 "curtask_parent=%p\n", 449 0, this_thr, this_thr->th.th_current_task, 450 this_thr->th.th_current_task->td_parent)); 451 } 452 453 // __kmp_push_current_task_to_thread: set up current task in called thread for a 454 // new team 455 // 456 // this_thr: thread structure to set up 457 // team: team for implicit task data 458 // tid: thread within team to set up 459 void __kmp_push_current_task_to_thread(kmp_info_t *this_thr, kmp_team_t *team, 460 int tid) { 461 // current task of the thread is a parent of the new just created implicit 462 // tasks of new team 463 KF_TRACE(10, ("__kmp_push_current_task_to_thread(enter): T#%d this_thread=%p " 464 "curtask=%p " 465 "parent_task=%p\n", 466 tid, this_thr, this_thr->th.th_current_task, 467 team->t.t_implicit_task_taskdata[tid].td_parent)); 468 469 KMP_DEBUG_ASSERT(this_thr != NULL); 470 471 if (tid == 0) { 472 if (this_thr->th.th_current_task != &team->t.t_implicit_task_taskdata[0]) { 473 team->t.t_implicit_task_taskdata[0].td_parent = 474 this_thr->th.th_current_task; 475 this_thr->th.th_current_task = &team->t.t_implicit_task_taskdata[0]; 476 } 477 } else { 478 team->t.t_implicit_task_taskdata[tid].td_parent = 479 team->t.t_implicit_task_taskdata[0].td_parent; 480 this_thr->th.th_current_task = &team->t.t_implicit_task_taskdata[tid]; 481 } 482 483 KF_TRACE(10, ("__kmp_push_current_task_to_thread(exit): T#%d this_thread=%p " 484 "curtask=%p " 485 "parent_task=%p\n", 486 tid, this_thr, this_thr->th.th_current_task, 487 team->t.t_implicit_task_taskdata[tid].td_parent)); 488 } 489 490 // __kmp_task_start: bookkeeping for a task starting execution 491 // 492 // GTID: global thread id of calling thread 493 // task: task starting execution 494 // current_task: task suspending 495 static void __kmp_task_start(kmp_int32 gtid, kmp_task_t *task, 496 kmp_taskdata_t *current_task) { 497 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 498 kmp_info_t *thread = __kmp_threads[gtid]; 499 500 KA_TRACE(10, 501 ("__kmp_task_start(enter): T#%d starting task %p: current_task=%p\n", 502 gtid, taskdata, current_task)); 503 504 KMP_DEBUG_ASSERT(taskdata->td_flags.tasktype == TASK_EXPLICIT); 505 506 // mark currently executing task as suspended 507 // TODO: GEH - make sure root team implicit task is initialized properly. 508 // KMP_DEBUG_ASSERT( current_task -> td_flags.executing == 1 ); 509 current_task->td_flags.executing = 0; 510 511 // Add task to stack if tied 512 #ifdef BUILD_TIED_TASK_STACK 513 if (taskdata->td_flags.tiedness == TASK_TIED) { 514 __kmp_push_task_stack(gtid, thread, taskdata); 515 } 516 #endif /* BUILD_TIED_TASK_STACK */ 517 518 // mark starting task as executing and as current task 519 thread->th.th_current_task = taskdata; 520 521 KMP_DEBUG_ASSERT(taskdata->td_flags.started == 0 || 522 taskdata->td_flags.tiedness == TASK_UNTIED); 523 KMP_DEBUG_ASSERT(taskdata->td_flags.executing == 0 || 524 taskdata->td_flags.tiedness == TASK_UNTIED); 525 taskdata->td_flags.started = 1; 526 taskdata->td_flags.executing = 1; 527 KMP_DEBUG_ASSERT(taskdata->td_flags.complete == 0); 528 KMP_DEBUG_ASSERT(taskdata->td_flags.freed == 0); 529 530 // GEH TODO: shouldn't we pass some sort of location identifier here? 531 // APT: yes, we will pass location here. 532 // need to store current thread state (in a thread or taskdata structure) 533 // before setting work_state, otherwise wrong state is set after end of task 534 535 KA_TRACE(10, ("__kmp_task_start(exit): T#%d task=%p\n", gtid, taskdata)); 536 537 return; 538 } 539 540 #if OMPT_SUPPORT 541 //------------------------------------------------------------------------------ 542 // __ompt_task_init: 543 // Initialize OMPT fields maintained by a task. This will only be called after 544 // ompt_start_tool, so we already know whether ompt is enabled or not. 545 546 static inline void __ompt_task_init(kmp_taskdata_t *task, int tid) { 547 // The calls to __ompt_task_init already have the ompt_enabled condition. 548 task->ompt_task_info.task_data.value = 0; 549 task->ompt_task_info.frame.exit_frame = ompt_data_none; 550 task->ompt_task_info.frame.enter_frame = ompt_data_none; 551 task->ompt_task_info.frame.exit_frame_flags = ompt_frame_runtime | ompt_frame_framepointer; 552 task->ompt_task_info.frame.enter_frame_flags = ompt_frame_runtime | ompt_frame_framepointer; 553 #if OMP_40_ENABLED 554 task->ompt_task_info.ndeps = 0; 555 task->ompt_task_info.deps = NULL; 556 #endif /* OMP_40_ENABLED */ 557 } 558 559 // __ompt_task_start: 560 // Build and trigger task-begin event 561 static inline void __ompt_task_start(kmp_task_t *task, 562 kmp_taskdata_t *current_task, 563 kmp_int32 gtid) { 564 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 565 ompt_task_status_t status = ompt_task_switch; 566 if (__kmp_threads[gtid]->th.ompt_thread_info.ompt_task_yielded) { 567 status = ompt_task_yield; 568 __kmp_threads[gtid]->th.ompt_thread_info.ompt_task_yielded = 0; 569 } 570 /* let OMPT know that we're about to run this task */ 571 if (ompt_enabled.ompt_callback_task_schedule) { 572 ompt_callbacks.ompt_callback(ompt_callback_task_schedule)( 573 &(current_task->ompt_task_info.task_data), status, 574 &(taskdata->ompt_task_info.task_data)); 575 } 576 taskdata->ompt_task_info.scheduling_parent = current_task; 577 } 578 579 // __ompt_task_finish: 580 // Build and trigger final task-schedule event 581 static inline void 582 __ompt_task_finish(kmp_task_t *task, kmp_taskdata_t *resumed_task, 583 ompt_task_status_t status = ompt_task_complete) { 584 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 585 if (__kmp_omp_cancellation && taskdata->td_taskgroup && 586 taskdata->td_taskgroup->cancel_request == cancel_taskgroup) { 587 status = ompt_task_cancel; 588 } 589 590 /* let OMPT know that we're returning to the callee task */ 591 if (ompt_enabled.ompt_callback_task_schedule) { 592 ompt_callbacks.ompt_callback(ompt_callback_task_schedule)( 593 &(taskdata->ompt_task_info.task_data), status, 594 &((resumed_task ? resumed_task 595 : (taskdata->ompt_task_info.scheduling_parent 596 ? taskdata->ompt_task_info.scheduling_parent 597 : taskdata->td_parent)) 598 ->ompt_task_info.task_data)); 599 } 600 } 601 #endif 602 603 template <bool ompt> 604 static void __kmpc_omp_task_begin_if0_template(ident_t *loc_ref, kmp_int32 gtid, 605 kmp_task_t *task, 606 void *frame_address, 607 void *return_address) { 608 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 609 kmp_taskdata_t *current_task = __kmp_threads[gtid]->th.th_current_task; 610 611 KA_TRACE(10, ("__kmpc_omp_task_begin_if0(enter): T#%d loc=%p task=%p " 612 "current_task=%p\n", 613 gtid, loc_ref, taskdata, current_task)); 614 615 if (taskdata->td_flags.tiedness == TASK_UNTIED) { 616 // untied task needs to increment counter so that the task structure is not 617 // freed prematurely 618 kmp_int32 counter = 1 + KMP_ATOMIC_INC(&taskdata->td_untied_count); 619 KMP_DEBUG_USE_VAR(counter); 620 KA_TRACE(20, ("__kmpc_omp_task_begin_if0: T#%d untied_count (%d) " 621 "incremented for task %p\n", 622 gtid, counter, taskdata)); 623 } 624 625 taskdata->td_flags.task_serial = 626 1; // Execute this task immediately, not deferred. 627 __kmp_task_start(gtid, task, current_task); 628 629 #if OMPT_SUPPORT 630 if (ompt) { 631 if (current_task->ompt_task_info.frame.enter_frame.ptr == NULL) { 632 current_task->ompt_task_info.frame.enter_frame.ptr = 633 taskdata->ompt_task_info.frame.exit_frame.ptr = frame_address; 634 current_task->ompt_task_info.frame.enter_frame_flags = 635 taskdata->ompt_task_info.frame.exit_frame_flags = ompt_frame_application | ompt_frame_framepointer; 636 } 637 if (ompt_enabled.ompt_callback_task_create) { 638 ompt_task_info_t *parent_info = &(current_task->ompt_task_info); 639 ompt_callbacks.ompt_callback(ompt_callback_task_create)( 640 &(parent_info->task_data), &(parent_info->frame), 641 &(taskdata->ompt_task_info.task_data), 642 ompt_task_explicit | TASK_TYPE_DETAILS_FORMAT(taskdata), 0, 643 return_address); 644 } 645 __ompt_task_start(task, current_task, gtid); 646 } 647 #endif // OMPT_SUPPORT 648 649 KA_TRACE(10, ("__kmpc_omp_task_begin_if0(exit): T#%d loc=%p task=%p,\n", gtid, 650 loc_ref, taskdata)); 651 } 652 653 #if OMPT_SUPPORT 654 OMPT_NOINLINE 655 static void __kmpc_omp_task_begin_if0_ompt(ident_t *loc_ref, kmp_int32 gtid, 656 kmp_task_t *task, 657 void *frame_address, 658 void *return_address) { 659 __kmpc_omp_task_begin_if0_template<true>(loc_ref, gtid, task, frame_address, 660 return_address); 661 } 662 #endif // OMPT_SUPPORT 663 664 // __kmpc_omp_task_begin_if0: report that a given serialized task has started 665 // execution 666 // 667 // loc_ref: source location information; points to beginning of task block. 668 // gtid: global thread number. 669 // task: task thunk for the started task. 670 void __kmpc_omp_task_begin_if0(ident_t *loc_ref, kmp_int32 gtid, 671 kmp_task_t *task) { 672 #if OMPT_SUPPORT 673 if (UNLIKELY(ompt_enabled.enabled)) { 674 OMPT_STORE_RETURN_ADDRESS(gtid); 675 __kmpc_omp_task_begin_if0_ompt(loc_ref, gtid, task, 676 OMPT_GET_FRAME_ADDRESS(1), 677 OMPT_LOAD_RETURN_ADDRESS(gtid)); 678 return; 679 } 680 #endif 681 __kmpc_omp_task_begin_if0_template<false>(loc_ref, gtid, task, NULL, NULL); 682 } 683 684 #ifdef TASK_UNUSED 685 // __kmpc_omp_task_begin: report that a given task has started execution 686 // NEVER GENERATED BY COMPILER, DEPRECATED!!! 687 void __kmpc_omp_task_begin(ident_t *loc_ref, kmp_int32 gtid, kmp_task_t *task) { 688 kmp_taskdata_t *current_task = __kmp_threads[gtid]->th.th_current_task; 689 690 KA_TRACE( 691 10, 692 ("__kmpc_omp_task_begin(enter): T#%d loc=%p task=%p current_task=%p\n", 693 gtid, loc_ref, KMP_TASK_TO_TASKDATA(task), current_task)); 694 695 __kmp_task_start(gtid, task, current_task); 696 697 KA_TRACE(10, ("__kmpc_omp_task_begin(exit): T#%d loc=%p task=%p,\n", gtid, 698 loc_ref, KMP_TASK_TO_TASKDATA(task))); 699 return; 700 } 701 #endif // TASK_UNUSED 702 703 // __kmp_free_task: free the current task space and the space for shareds 704 // 705 // gtid: Global thread ID of calling thread 706 // taskdata: task to free 707 // thread: thread data structure of caller 708 static void __kmp_free_task(kmp_int32 gtid, kmp_taskdata_t *taskdata, 709 kmp_info_t *thread) { 710 KA_TRACE(30, ("__kmp_free_task: T#%d freeing data from task %p\n", gtid, 711 taskdata)); 712 713 // Check to make sure all flags and counters have the correct values 714 KMP_DEBUG_ASSERT(taskdata->td_flags.tasktype == TASK_EXPLICIT); 715 KMP_DEBUG_ASSERT(taskdata->td_flags.executing == 0); 716 KMP_DEBUG_ASSERT(taskdata->td_flags.complete == 1); 717 KMP_DEBUG_ASSERT(taskdata->td_flags.freed == 0); 718 KMP_DEBUG_ASSERT(taskdata->td_allocated_child_tasks == 0 || 719 taskdata->td_flags.task_serial == 1); 720 KMP_DEBUG_ASSERT(taskdata->td_incomplete_child_tasks == 0); 721 722 taskdata->td_flags.freed = 1; 723 ANNOTATE_HAPPENS_BEFORE(taskdata); 724 // deallocate the taskdata and shared variable blocks associated with this task 725 #if USE_FAST_MEMORY 726 __kmp_fast_free(thread, taskdata); 727 #else /* ! USE_FAST_MEMORY */ 728 __kmp_thread_free(thread, taskdata); 729 #endif 730 731 KA_TRACE(20, ("__kmp_free_task: T#%d freed task %p\n", gtid, taskdata)); 732 } 733 734 // __kmp_free_task_and_ancestors: free the current task and ancestors without 735 // children 736 // 737 // gtid: Global thread ID of calling thread 738 // taskdata: task to free 739 // thread: thread data structure of caller 740 static void __kmp_free_task_and_ancestors(kmp_int32 gtid, 741 kmp_taskdata_t *taskdata, 742 kmp_info_t *thread) { 743 #if OMP_45_ENABLED 744 // Proxy tasks must always be allowed to free their parents 745 // because they can be run in background even in serial mode. 746 kmp_int32 team_serial = 747 (taskdata->td_flags.team_serial || taskdata->td_flags.tasking_ser) && 748 !taskdata->td_flags.proxy; 749 #else 750 kmp_int32 team_serial = 751 taskdata->td_flags.team_serial || taskdata->td_flags.tasking_ser; 752 #endif 753 KMP_DEBUG_ASSERT(taskdata->td_flags.tasktype == TASK_EXPLICIT); 754 755 kmp_int32 children = KMP_ATOMIC_DEC(&taskdata->td_allocated_child_tasks) - 1; 756 KMP_DEBUG_ASSERT(children >= 0); 757 758 // Now, go up the ancestor tree to see if any ancestors can now be freed. 759 while (children == 0) { 760 kmp_taskdata_t *parent_taskdata = taskdata->td_parent; 761 762 KA_TRACE(20, ("__kmp_free_task_and_ancestors(enter): T#%d task %p complete " 763 "and freeing itself\n", 764 gtid, taskdata)); 765 766 // --- Deallocate my ancestor task --- 767 __kmp_free_task(gtid, taskdata, thread); 768 769 taskdata = parent_taskdata; 770 771 if (team_serial) 772 return; 773 // Stop checking ancestors at implicit task instead of walking up ancestor 774 // tree to avoid premature deallocation of ancestors. 775 if (taskdata->td_flags.tasktype == TASK_IMPLICIT) { 776 if (taskdata->td_dephash) { // do we need to cleanup dephash? 777 int children = KMP_ATOMIC_LD_ACQ(&taskdata->td_incomplete_child_tasks); 778 kmp_tasking_flags_t flags_old = taskdata->td_flags; 779 if (children == 0 && flags_old.complete == 1) { 780 kmp_tasking_flags_t flags_new = flags_old; 781 flags_new.complete = 0; 782 if (KMP_COMPARE_AND_STORE_ACQ32( 783 RCAST(kmp_int32 *, &taskdata->td_flags), 784 *RCAST(kmp_int32 *, &flags_old), 785 *RCAST(kmp_int32 *, &flags_new))) { 786 KA_TRACE(100, ("__kmp_free_task_and_ancestors: T#%d cleans " 787 "dephash of implicit task %p\n", 788 gtid, taskdata)); 789 // cleanup dephash of finished implicit task 790 __kmp_dephash_free_entries(thread, taskdata->td_dephash); 791 } 792 } 793 } 794 return; 795 } 796 // Predecrement simulated by "- 1" calculation 797 children = KMP_ATOMIC_DEC(&taskdata->td_allocated_child_tasks) - 1; 798 KMP_DEBUG_ASSERT(children >= 0); 799 } 800 801 KA_TRACE( 802 20, ("__kmp_free_task_and_ancestors(exit): T#%d task %p has %d children; " 803 "not freeing it yet\n", 804 gtid, taskdata, children)); 805 } 806 807 // __kmp_task_finish: bookkeeping to do when a task finishes execution 808 // 809 // gtid: global thread ID for calling thread 810 // task: task to be finished 811 // resumed_task: task to be resumed. (may be NULL if task is serialized) 812 template <bool ompt> 813 static void __kmp_task_finish(kmp_int32 gtid, kmp_task_t *task, 814 kmp_taskdata_t *resumed_task) { 815 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 816 kmp_info_t *thread = __kmp_threads[gtid]; 817 #if OMP_45_ENABLED 818 kmp_task_team_t *task_team = 819 thread->th.th_task_team; // might be NULL for serial teams... 820 #endif // OMP_45_ENABLED 821 kmp_int32 children = 0; 822 823 KA_TRACE(10, ("__kmp_task_finish(enter): T#%d finishing task %p and resuming " 824 "task %p\n", 825 gtid, taskdata, resumed_task)); 826 827 KMP_DEBUG_ASSERT(taskdata->td_flags.tasktype == TASK_EXPLICIT); 828 829 // Pop task from stack if tied 830 #ifdef BUILD_TIED_TASK_STACK 831 if (taskdata->td_flags.tiedness == TASK_TIED) { 832 __kmp_pop_task_stack(gtid, thread, taskdata); 833 } 834 #endif /* BUILD_TIED_TASK_STACK */ 835 836 if (taskdata->td_flags.tiedness == TASK_UNTIED) { 837 // untied task needs to check the counter so that the task structure is not 838 // freed prematurely 839 kmp_int32 counter = KMP_ATOMIC_DEC(&taskdata->td_untied_count) - 1; 840 KA_TRACE( 841 20, 842 ("__kmp_task_finish: T#%d untied_count (%d) decremented for task %p\n", 843 gtid, counter, taskdata)); 844 if (counter > 0) { 845 // untied task is not done, to be continued possibly by other thread, do 846 // not free it now 847 if (resumed_task == NULL) { 848 KMP_DEBUG_ASSERT(taskdata->td_flags.task_serial); 849 resumed_task = taskdata->td_parent; // In a serialized task, the resumed 850 // task is the parent 851 } 852 thread->th.th_current_task = resumed_task; // restore current_task 853 resumed_task->td_flags.executing = 1; // resume previous task 854 KA_TRACE(10, ("__kmp_task_finish(exit): T#%d partially done task %p, " 855 "resuming task %p\n", 856 gtid, taskdata, resumed_task)); 857 return; 858 } 859 } 860 #if OMPT_SUPPORT 861 if (ompt) 862 __ompt_task_finish(task, resumed_task); 863 #endif 864 865 // Check mutexinoutset dependencies, release locks 866 kmp_depnode_t *node = taskdata->td_depnode; 867 if (node && (node->dn.mtx_num_locks < 0)) { 868 // negative num_locks means all locks were acquired 869 node->dn.mtx_num_locks = -node->dn.mtx_num_locks; 870 for (int i = node->dn.mtx_num_locks - 1; i >= 0; --i) { 871 KMP_DEBUG_ASSERT(node->dn.mtx_locks[i] != NULL); 872 __kmp_release_lock(node->dn.mtx_locks[i], gtid); 873 } 874 } 875 876 KMP_DEBUG_ASSERT(taskdata->td_flags.complete == 0); 877 taskdata->td_flags.complete = 1; // mark the task as completed 878 KMP_DEBUG_ASSERT(taskdata->td_flags.started == 1); 879 KMP_DEBUG_ASSERT(taskdata->td_flags.freed == 0); 880 881 // Only need to keep track of count if team parallel and tasking not 882 // serialized 883 if (!(taskdata->td_flags.team_serial || taskdata->td_flags.tasking_ser)) { 884 // Predecrement simulated by "- 1" calculation 885 children = 886 KMP_ATOMIC_DEC(&taskdata->td_parent->td_incomplete_child_tasks) - 1; 887 KMP_DEBUG_ASSERT(children >= 0); 888 #if OMP_40_ENABLED 889 if (taskdata->td_taskgroup) 890 KMP_ATOMIC_DEC(&taskdata->td_taskgroup->count); 891 __kmp_release_deps(gtid, taskdata); 892 #if OMP_45_ENABLED 893 } else if (task_team && task_team->tt.tt_found_proxy_tasks) { 894 // if we found proxy tasks there could exist a dependency chain 895 // with the proxy task as origin 896 __kmp_release_deps(gtid, taskdata); 897 #endif // OMP_45_ENABLED 898 #endif // OMP_40_ENABLED 899 } 900 901 // td_flags.executing must be marked as 0 after __kmp_release_deps has been 902 // called. Othertwise, if a task is executed immediately from the release_deps 903 // code, the flag will be reset to 1 again by this same function 904 KMP_DEBUG_ASSERT(taskdata->td_flags.executing == 1); 905 taskdata->td_flags.executing = 0; // suspend the finishing task 906 907 KA_TRACE( 908 20, ("__kmp_task_finish: T#%d finished task %p, %d incomplete children\n", 909 gtid, taskdata, children)); 910 911 #if OMP_40_ENABLED 912 /* If the tasks' destructor thunk flag has been set, we need to invoke the 913 destructor thunk that has been generated by the compiler. The code is 914 placed here, since at this point other tasks might have been released 915 hence overlapping the destructor invokations with some other work in the 916 released tasks. The OpenMP spec is not specific on when the destructors 917 are invoked, so we should be free to choose. */ 918 if (taskdata->td_flags.destructors_thunk) { 919 kmp_routine_entry_t destr_thunk = task->data1.destructors; 920 KMP_ASSERT(destr_thunk); 921 destr_thunk(gtid, task); 922 } 923 #endif // OMP_40_ENABLED 924 925 // bookkeeping for resuming task: 926 // GEH - note tasking_ser => task_serial 927 KMP_DEBUG_ASSERT( 928 (taskdata->td_flags.tasking_ser || taskdata->td_flags.task_serial) == 929 taskdata->td_flags.task_serial); 930 if (taskdata->td_flags.task_serial) { 931 if (resumed_task == NULL) { 932 resumed_task = taskdata->td_parent; // In a serialized task, the resumed 933 // task is the parent 934 } 935 } else { 936 KMP_DEBUG_ASSERT(resumed_task != 937 NULL); // verify that resumed task is passed as arguemnt 938 } 939 940 // Free this task and then ancestor tasks if they have no children. 941 // Restore th_current_task first as suggested by John: 942 // johnmc: if an asynchronous inquiry peers into the runtime system 943 // it doesn't see the freed task as the current task. 944 thread->th.th_current_task = resumed_task; 945 __kmp_free_task_and_ancestors(gtid, taskdata, thread); 946 947 // TODO: GEH - make sure root team implicit task is initialized properly. 948 // KMP_DEBUG_ASSERT( resumed_task->td_flags.executing == 0 ); 949 resumed_task->td_flags.executing = 1; // resume previous task 950 951 KA_TRACE( 952 10, ("__kmp_task_finish(exit): T#%d finished task %p, resuming task %p\n", 953 gtid, taskdata, resumed_task)); 954 955 return; 956 } 957 958 template <bool ompt> 959 static void __kmpc_omp_task_complete_if0_template(ident_t *loc_ref, 960 kmp_int32 gtid, 961 kmp_task_t *task) { 962 KA_TRACE(10, ("__kmpc_omp_task_complete_if0(enter): T#%d loc=%p task=%p\n", 963 gtid, loc_ref, KMP_TASK_TO_TASKDATA(task))); 964 // this routine will provide task to resume 965 __kmp_task_finish<ompt>(gtid, task, NULL); 966 967 KA_TRACE(10, ("__kmpc_omp_task_complete_if0(exit): T#%d loc=%p task=%p\n", 968 gtid, loc_ref, KMP_TASK_TO_TASKDATA(task))); 969 970 #if OMPT_SUPPORT 971 if (ompt) { 972 ompt_frame_t *ompt_frame; 973 __ompt_get_task_info_internal(0, NULL, NULL, &ompt_frame, NULL, NULL); 974 ompt_frame->enter_frame = ompt_data_none; 975 ompt_frame->enter_frame_flags = ompt_frame_runtime | ompt_frame_framepointer; 976 } 977 #endif 978 979 return; 980 } 981 982 #if OMPT_SUPPORT 983 OMPT_NOINLINE 984 void __kmpc_omp_task_complete_if0_ompt(ident_t *loc_ref, kmp_int32 gtid, 985 kmp_task_t *task) { 986 __kmpc_omp_task_complete_if0_template<true>(loc_ref, gtid, task); 987 } 988 #endif // OMPT_SUPPORT 989 990 // __kmpc_omp_task_complete_if0: report that a task has completed execution 991 // 992 // loc_ref: source location information; points to end of task block. 993 // gtid: global thread number. 994 // task: task thunk for the completed task. 995 void __kmpc_omp_task_complete_if0(ident_t *loc_ref, kmp_int32 gtid, 996 kmp_task_t *task) { 997 #if OMPT_SUPPORT 998 if (UNLIKELY(ompt_enabled.enabled)) { 999 __kmpc_omp_task_complete_if0_ompt(loc_ref, gtid, task); 1000 return; 1001 } 1002 #endif 1003 __kmpc_omp_task_complete_if0_template<false>(loc_ref, gtid, task); 1004 } 1005 1006 #ifdef TASK_UNUSED 1007 // __kmpc_omp_task_complete: report that a task has completed execution 1008 // NEVER GENERATED BY COMPILER, DEPRECATED!!! 1009 void __kmpc_omp_task_complete(ident_t *loc_ref, kmp_int32 gtid, 1010 kmp_task_t *task) { 1011 KA_TRACE(10, ("__kmpc_omp_task_complete(enter): T#%d loc=%p task=%p\n", gtid, 1012 loc_ref, KMP_TASK_TO_TASKDATA(task))); 1013 1014 __kmp_task_finish<false>(gtid, task, 1015 NULL); // Not sure how to find task to resume 1016 1017 KA_TRACE(10, ("__kmpc_omp_task_complete(exit): T#%d loc=%p task=%p\n", gtid, 1018 loc_ref, KMP_TASK_TO_TASKDATA(task))); 1019 return; 1020 } 1021 #endif // TASK_UNUSED 1022 1023 // __kmp_init_implicit_task: Initialize the appropriate fields in the implicit 1024 // task for a given thread 1025 // 1026 // loc_ref: reference to source location of parallel region 1027 // this_thr: thread data structure corresponding to implicit task 1028 // team: team for this_thr 1029 // tid: thread id of given thread within team 1030 // set_curr_task: TRUE if need to push current task to thread 1031 // NOTE: Routine does not set up the implicit task ICVS. This is assumed to 1032 // have already been done elsewhere. 1033 // TODO: Get better loc_ref. Value passed in may be NULL 1034 void __kmp_init_implicit_task(ident_t *loc_ref, kmp_info_t *this_thr, 1035 kmp_team_t *team, int tid, int set_curr_task) { 1036 kmp_taskdata_t *task = &team->t.t_implicit_task_taskdata[tid]; 1037 1038 KF_TRACE( 1039 10, 1040 ("__kmp_init_implicit_task(enter): T#:%d team=%p task=%p, reinit=%s\n", 1041 tid, team, task, set_curr_task ? "TRUE" : "FALSE")); 1042 1043 task->td_task_id = KMP_GEN_TASK_ID(); 1044 task->td_team = team; 1045 // task->td_parent = NULL; // fix for CQ230101 (broken parent task info 1046 // in debugger) 1047 task->td_ident = loc_ref; 1048 task->td_taskwait_ident = NULL; 1049 task->td_taskwait_counter = 0; 1050 task->td_taskwait_thread = 0; 1051 1052 task->td_flags.tiedness = TASK_TIED; 1053 task->td_flags.tasktype = TASK_IMPLICIT; 1054 #if OMP_45_ENABLED 1055 task->td_flags.proxy = TASK_FULL; 1056 #endif 1057 1058 // All implicit tasks are executed immediately, not deferred 1059 task->td_flags.task_serial = 1; 1060 task->td_flags.tasking_ser = (__kmp_tasking_mode == tskm_immediate_exec); 1061 task->td_flags.team_serial = (team->t.t_serialized) ? 1 : 0; 1062 1063 task->td_flags.started = 1; 1064 task->td_flags.executing = 1; 1065 task->td_flags.complete = 0; 1066 task->td_flags.freed = 0; 1067 1068 #if OMP_40_ENABLED 1069 task->td_depnode = NULL; 1070 #endif 1071 task->td_last_tied = task; 1072 1073 if (set_curr_task) { // only do this init first time thread is created 1074 KMP_ATOMIC_ST_REL(&task->td_incomplete_child_tasks, 0); 1075 // Not used: don't need to deallocate implicit task 1076 KMP_ATOMIC_ST_REL(&task->td_allocated_child_tasks, 0); 1077 #if OMP_40_ENABLED 1078 task->td_taskgroup = NULL; // An implicit task does not have taskgroup 1079 task->td_dephash = NULL; 1080 #endif 1081 __kmp_push_current_task_to_thread(this_thr, team, tid); 1082 } else { 1083 KMP_DEBUG_ASSERT(task->td_incomplete_child_tasks == 0); 1084 KMP_DEBUG_ASSERT(task->td_allocated_child_tasks == 0); 1085 } 1086 1087 #if OMPT_SUPPORT 1088 if (UNLIKELY(ompt_enabled.enabled)) 1089 __ompt_task_init(task, tid); 1090 #endif 1091 1092 KF_TRACE(10, ("__kmp_init_implicit_task(exit): T#:%d team=%p task=%p\n", tid, 1093 team, task)); 1094 } 1095 1096 // __kmp_finish_implicit_task: Release resources associated to implicit tasks 1097 // at the end of parallel regions. Some resources are kept for reuse in the next 1098 // parallel region. 1099 // 1100 // thread: thread data structure corresponding to implicit task 1101 void __kmp_finish_implicit_task(kmp_info_t *thread) { 1102 kmp_taskdata_t *task = thread->th.th_current_task; 1103 if (task->td_dephash) { 1104 int children; 1105 task->td_flags.complete = 1; 1106 children = KMP_ATOMIC_LD_ACQ(&task->td_incomplete_child_tasks); 1107 kmp_tasking_flags_t flags_old = task->td_flags; 1108 if (children == 0 && flags_old.complete == 1) { 1109 kmp_tasking_flags_t flags_new = flags_old; 1110 flags_new.complete = 0; 1111 if (KMP_COMPARE_AND_STORE_ACQ32(RCAST(kmp_int32 *, &task->td_flags), 1112 *RCAST(kmp_int32 *, &flags_old), 1113 *RCAST(kmp_int32 *, &flags_new))) { 1114 KA_TRACE(100, ("__kmp_finish_implicit_task: T#%d cleans " 1115 "dephash of implicit task %p\n", 1116 thread->th.th_info.ds.ds_gtid, task)); 1117 __kmp_dephash_free_entries(thread, task->td_dephash); 1118 } 1119 } 1120 } 1121 } 1122 1123 // __kmp_free_implicit_task: Release resources associated to implicit tasks 1124 // when these are destroyed regions 1125 // 1126 // thread: thread data structure corresponding to implicit task 1127 void __kmp_free_implicit_task(kmp_info_t *thread) { 1128 kmp_taskdata_t *task = thread->th.th_current_task; 1129 if (task && task->td_dephash) { 1130 __kmp_dephash_free(thread, task->td_dephash); 1131 task->td_dephash = NULL; 1132 } 1133 } 1134 1135 // Round up a size to a power of two specified by val: Used to insert padding 1136 // between structures co-allocated using a single malloc() call 1137 static size_t __kmp_round_up_to_val(size_t size, size_t val) { 1138 if (size & (val - 1)) { 1139 size &= ~(val - 1); 1140 if (size <= KMP_SIZE_T_MAX - val) { 1141 size += val; // Round up if there is no overflow. 1142 } 1143 } 1144 return size; 1145 } // __kmp_round_up_to_va 1146 1147 // __kmp_task_alloc: Allocate the taskdata and task data structures for a task 1148 // 1149 // loc_ref: source location information 1150 // gtid: global thread number. 1151 // flags: include tiedness & task type (explicit vs. implicit) of the ''new'' 1152 // task encountered. Converted from kmp_int32 to kmp_tasking_flags_t in routine. 1153 // sizeof_kmp_task_t: Size in bytes of kmp_task_t data structure including 1154 // private vars accessed in task. 1155 // sizeof_shareds: Size in bytes of array of pointers to shared vars accessed 1156 // in task. 1157 // task_entry: Pointer to task code entry point generated by compiler. 1158 // returns: a pointer to the allocated kmp_task_t structure (task). 1159 kmp_task_t *__kmp_task_alloc(ident_t *loc_ref, kmp_int32 gtid, 1160 kmp_tasking_flags_t *flags, 1161 size_t sizeof_kmp_task_t, size_t sizeof_shareds, 1162 kmp_routine_entry_t task_entry) { 1163 kmp_task_t *task; 1164 kmp_taskdata_t *taskdata; 1165 kmp_info_t *thread = __kmp_threads[gtid]; 1166 kmp_team_t *team = thread->th.th_team; 1167 kmp_taskdata_t *parent_task = thread->th.th_current_task; 1168 size_t shareds_offset; 1169 1170 if (!TCR_4(__kmp_init_middle)) 1171 __kmp_middle_initialize(); 1172 1173 KA_TRACE(10, ("__kmp_task_alloc(enter): T#%d loc=%p, flags=(0x%x) " 1174 "sizeof_task=%ld sizeof_shared=%ld entry=%p\n", 1175 gtid, loc_ref, *((kmp_int32 *)flags), sizeof_kmp_task_t, 1176 sizeof_shareds, task_entry)); 1177 1178 if (parent_task->td_flags.final) { 1179 if (flags->merged_if0) { 1180 } 1181 flags->final = 1; 1182 } 1183 if (flags->tiedness == TASK_UNTIED && !team->t.t_serialized) { 1184 // Untied task encountered causes the TSC algorithm to check entire deque of 1185 // the victim thread. If no untied task encountered, then checking the head 1186 // of the deque should be enough. 1187 KMP_CHECK_UPDATE(thread->th.th_task_team->tt.tt_untied_task_encountered, 1); 1188 } 1189 1190 #if OMP_45_ENABLED 1191 if (flags->proxy == TASK_PROXY) { 1192 flags->tiedness = TASK_UNTIED; 1193 flags->merged_if0 = 1; 1194 1195 /* are we running in a sequential parallel or tskm_immediate_exec... we need 1196 tasking support enabled */ 1197 if ((thread->th.th_task_team) == NULL) { 1198 /* This should only happen if the team is serialized 1199 setup a task team and propagate it to the thread */ 1200 KMP_DEBUG_ASSERT(team->t.t_serialized); 1201 KA_TRACE(30, 1202 ("T#%d creating task team in __kmp_task_alloc for proxy task\n", 1203 gtid)); 1204 __kmp_task_team_setup( 1205 thread, team, 1206 1); // 1 indicates setup the current team regardless of nthreads 1207 thread->th.th_task_team = team->t.t_task_team[thread->th.th_task_state]; 1208 } 1209 kmp_task_team_t *task_team = thread->th.th_task_team; 1210 1211 /* tasking must be enabled now as the task might not be pushed */ 1212 if (!KMP_TASKING_ENABLED(task_team)) { 1213 KA_TRACE( 1214 30, 1215 ("T#%d enabling tasking in __kmp_task_alloc for proxy task\n", gtid)); 1216 __kmp_enable_tasking(task_team, thread); 1217 kmp_int32 tid = thread->th.th_info.ds.ds_tid; 1218 kmp_thread_data_t *thread_data = &task_team->tt.tt_threads_data[tid]; 1219 // No lock needed since only owner can allocate 1220 if (thread_data->td.td_deque == NULL) { 1221 __kmp_alloc_task_deque(thread, thread_data); 1222 } 1223 } 1224 1225 if (task_team->tt.tt_found_proxy_tasks == FALSE) 1226 TCW_4(task_team->tt.tt_found_proxy_tasks, TRUE); 1227 } 1228 #endif 1229 1230 // Calculate shared structure offset including padding after kmp_task_t struct 1231 // to align pointers in shared struct 1232 shareds_offset = sizeof(kmp_taskdata_t) + sizeof_kmp_task_t; 1233 shareds_offset = __kmp_round_up_to_val(shareds_offset, sizeof(void *)); 1234 1235 // Allocate a kmp_taskdata_t block and a kmp_task_t block. 1236 KA_TRACE(30, ("__kmp_task_alloc: T#%d First malloc size: %ld\n", gtid, 1237 shareds_offset)); 1238 KA_TRACE(30, ("__kmp_task_alloc: T#%d Second malloc size: %ld\n", gtid, 1239 sizeof_shareds)); 1240 1241 // Avoid double allocation here by combining shareds with taskdata 1242 #if USE_FAST_MEMORY 1243 taskdata = (kmp_taskdata_t *)__kmp_fast_allocate(thread, shareds_offset + 1244 sizeof_shareds); 1245 #else /* ! USE_FAST_MEMORY */ 1246 taskdata = (kmp_taskdata_t *)__kmp_thread_malloc(thread, shareds_offset + 1247 sizeof_shareds); 1248 #endif /* USE_FAST_MEMORY */ 1249 ANNOTATE_HAPPENS_AFTER(taskdata); 1250 1251 task = KMP_TASKDATA_TO_TASK(taskdata); 1252 1253 // Make sure task & taskdata are aligned appropriately 1254 #if KMP_ARCH_X86 || KMP_ARCH_PPC64 || !KMP_HAVE_QUAD 1255 KMP_DEBUG_ASSERT((((kmp_uintptr_t)taskdata) & (sizeof(double) - 1)) == 0); 1256 KMP_DEBUG_ASSERT((((kmp_uintptr_t)task) & (sizeof(double) - 1)) == 0); 1257 #else 1258 KMP_DEBUG_ASSERT((((kmp_uintptr_t)taskdata) & (sizeof(_Quad) - 1)) == 0); 1259 KMP_DEBUG_ASSERT((((kmp_uintptr_t)task) & (sizeof(_Quad) - 1)) == 0); 1260 #endif 1261 if (sizeof_shareds > 0) { 1262 // Avoid double allocation here by combining shareds with taskdata 1263 task->shareds = &((char *)taskdata)[shareds_offset]; 1264 // Make sure shareds struct is aligned to pointer size 1265 KMP_DEBUG_ASSERT((((kmp_uintptr_t)task->shareds) & (sizeof(void *) - 1)) == 1266 0); 1267 } else { 1268 task->shareds = NULL; 1269 } 1270 task->routine = task_entry; 1271 task->part_id = 0; // AC: Always start with 0 part id 1272 1273 taskdata->td_task_id = KMP_GEN_TASK_ID(); 1274 taskdata->td_team = team; 1275 taskdata->td_alloc_thread = thread; 1276 taskdata->td_parent = parent_task; 1277 taskdata->td_level = parent_task->td_level + 1; // increment nesting level 1278 KMP_ATOMIC_ST_RLX(&taskdata->td_untied_count, 0); 1279 taskdata->td_ident = loc_ref; 1280 taskdata->td_taskwait_ident = NULL; 1281 taskdata->td_taskwait_counter = 0; 1282 taskdata->td_taskwait_thread = 0; 1283 KMP_DEBUG_ASSERT(taskdata->td_parent != NULL); 1284 #if OMP_45_ENABLED 1285 // avoid copying icvs for proxy tasks 1286 if (flags->proxy == TASK_FULL) 1287 #endif 1288 copy_icvs(&taskdata->td_icvs, &taskdata->td_parent->td_icvs); 1289 1290 taskdata->td_flags.tiedness = flags->tiedness; 1291 taskdata->td_flags.final = flags->final; 1292 taskdata->td_flags.merged_if0 = flags->merged_if0; 1293 #if OMP_40_ENABLED 1294 taskdata->td_flags.destructors_thunk = flags->destructors_thunk; 1295 #endif // OMP_40_ENABLED 1296 #if OMP_45_ENABLED 1297 taskdata->td_flags.proxy = flags->proxy; 1298 taskdata->td_task_team = thread->th.th_task_team; 1299 taskdata->td_size_alloc = shareds_offset + sizeof_shareds; 1300 #endif 1301 taskdata->td_flags.tasktype = TASK_EXPLICIT; 1302 1303 // GEH - TODO: fix this to copy parent task's value of tasking_ser flag 1304 taskdata->td_flags.tasking_ser = (__kmp_tasking_mode == tskm_immediate_exec); 1305 1306 // GEH - TODO: fix this to copy parent task's value of team_serial flag 1307 taskdata->td_flags.team_serial = (team->t.t_serialized) ? 1 : 0; 1308 1309 // GEH - Note we serialize the task if the team is serialized to make sure 1310 // implicit parallel region tasks are not left until program termination to 1311 // execute. Also, it helps locality to execute immediately. 1312 1313 taskdata->td_flags.task_serial = 1314 (parent_task->td_flags.final || taskdata->td_flags.team_serial || 1315 taskdata->td_flags.tasking_ser); 1316 1317 taskdata->td_flags.started = 0; 1318 taskdata->td_flags.executing = 0; 1319 taskdata->td_flags.complete = 0; 1320 taskdata->td_flags.freed = 0; 1321 1322 taskdata->td_flags.native = flags->native; 1323 1324 KMP_ATOMIC_ST_RLX(&taskdata->td_incomplete_child_tasks, 0); 1325 // start at one because counts current task and children 1326 KMP_ATOMIC_ST_RLX(&taskdata->td_allocated_child_tasks, 1); 1327 #if OMP_40_ENABLED 1328 taskdata->td_taskgroup = 1329 parent_task->td_taskgroup; // task inherits taskgroup from the parent task 1330 taskdata->td_dephash = NULL; 1331 taskdata->td_depnode = NULL; 1332 #endif 1333 if (flags->tiedness == TASK_UNTIED) 1334 taskdata->td_last_tied = NULL; // will be set when the task is scheduled 1335 else 1336 taskdata->td_last_tied = taskdata; 1337 1338 #if OMPT_SUPPORT 1339 if (UNLIKELY(ompt_enabled.enabled)) 1340 __ompt_task_init(taskdata, gtid); 1341 #endif 1342 // Only need to keep track of child task counts if team parallel and tasking not 1343 // serialized or if it is a proxy task 1344 #if OMP_45_ENABLED 1345 if (flags->proxy == TASK_PROXY || 1346 !(taskdata->td_flags.team_serial || taskdata->td_flags.tasking_ser)) 1347 #else 1348 if (!(taskdata->td_flags.team_serial || taskdata->td_flags.tasking_ser)) 1349 #endif 1350 { 1351 KMP_ATOMIC_INC(&parent_task->td_incomplete_child_tasks); 1352 #if OMP_40_ENABLED 1353 if (parent_task->td_taskgroup) 1354 KMP_ATOMIC_INC(&parent_task->td_taskgroup->count); 1355 #endif 1356 // Only need to keep track of allocated child tasks for explicit tasks since 1357 // implicit not deallocated 1358 if (taskdata->td_parent->td_flags.tasktype == TASK_EXPLICIT) { 1359 KMP_ATOMIC_INC(&taskdata->td_parent->td_allocated_child_tasks); 1360 } 1361 } 1362 1363 KA_TRACE(20, ("__kmp_task_alloc(exit): T#%d created task %p parent=%p\n", 1364 gtid, taskdata, taskdata->td_parent)); 1365 ANNOTATE_HAPPENS_BEFORE(task); 1366 1367 return task; 1368 } 1369 1370 kmp_task_t *__kmpc_omp_task_alloc(ident_t *loc_ref, kmp_int32 gtid, 1371 kmp_int32 flags, size_t sizeof_kmp_task_t, 1372 size_t sizeof_shareds, 1373 kmp_routine_entry_t task_entry) { 1374 kmp_task_t *retval; 1375 kmp_tasking_flags_t *input_flags = (kmp_tasking_flags_t *)&flags; 1376 1377 input_flags->native = FALSE; 1378 // __kmp_task_alloc() sets up all other runtime flags 1379 1380 #if OMP_45_ENABLED 1381 KA_TRACE(10, ("__kmpc_omp_task_alloc(enter): T#%d loc=%p, flags=(%s %s) " 1382 "sizeof_task=%ld sizeof_shared=%ld entry=%p\n", 1383 gtid, loc_ref, input_flags->tiedness ? "tied " : "untied", 1384 input_flags->proxy ? "proxy" : "", sizeof_kmp_task_t, 1385 sizeof_shareds, task_entry)); 1386 #else 1387 KA_TRACE(10, ("__kmpc_omp_task_alloc(enter): T#%d loc=%p, flags=(%s) " 1388 "sizeof_task=%ld sizeof_shared=%ld entry=%p\n", 1389 gtid, loc_ref, input_flags->tiedness ? "tied " : "untied", 1390 sizeof_kmp_task_t, sizeof_shareds, task_entry)); 1391 #endif 1392 1393 retval = __kmp_task_alloc(loc_ref, gtid, input_flags, sizeof_kmp_task_t, 1394 sizeof_shareds, task_entry); 1395 1396 KA_TRACE(20, ("__kmpc_omp_task_alloc(exit): T#%d retval %p\n", gtid, retval)); 1397 1398 return retval; 1399 } 1400 1401 kmp_task_t *__kmpc_omp_target_task_alloc(ident_t *loc_ref, kmp_int32 gtid, 1402 kmp_int32 flags, 1403 size_t sizeof_kmp_task_t, 1404 size_t sizeof_shareds, 1405 kmp_routine_entry_t task_entry, 1406 kmp_int64 device_id) { 1407 return __kmpc_omp_task_alloc(loc_ref, gtid, flags, sizeof_kmp_task_t, 1408 sizeof_shareds, task_entry); 1409 } 1410 1411 #if OMP_50_ENABLED 1412 /*! 1413 @ingroup TASKING 1414 @param loc_ref location of the original task directive 1415 @param gtid Global Thread ID of encountering thread 1416 @param new_task task thunk allocated by __kmpc_omp_task_alloc() for the ''new 1417 task'' 1418 @param naffins Number of affinity items 1419 @param affin_list List of affinity items 1420 @return Returns non-zero if registering affinity information was not successful. 1421 Returns 0 if registration was successful 1422 This entry registers the affinity information attached to a task with the task 1423 thunk structure kmp_taskdata_t. 1424 */ 1425 kmp_int32 1426 __kmpc_omp_reg_task_with_affinity(ident_t *loc_ref, kmp_int32 gtid, 1427 kmp_task_t *new_task, kmp_int32 naffins, 1428 kmp_task_affinity_info_t *affin_list) { 1429 return 0; 1430 } 1431 #endif 1432 1433 // __kmp_invoke_task: invoke the specified task 1434 // 1435 // gtid: global thread ID of caller 1436 // task: the task to invoke 1437 // current_task: the task to resume after task invokation 1438 static void __kmp_invoke_task(kmp_int32 gtid, kmp_task_t *task, 1439 kmp_taskdata_t *current_task) { 1440 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 1441 kmp_info_t *thread; 1442 #if OMP_40_ENABLED 1443 int discard = 0 /* false */; 1444 #endif 1445 KA_TRACE( 1446 30, ("__kmp_invoke_task(enter): T#%d invoking task %p, current_task=%p\n", 1447 gtid, taskdata, current_task)); 1448 KMP_DEBUG_ASSERT(task); 1449 #if OMP_45_ENABLED 1450 if (taskdata->td_flags.proxy == TASK_PROXY && 1451 taskdata->td_flags.complete == 1) { 1452 // This is a proxy task that was already completed but it needs to run 1453 // its bottom-half finish 1454 KA_TRACE( 1455 30, 1456 ("__kmp_invoke_task: T#%d running bottom finish for proxy task %p\n", 1457 gtid, taskdata)); 1458 1459 __kmp_bottom_half_finish_proxy(gtid, task); 1460 1461 KA_TRACE(30, ("__kmp_invoke_task(exit): T#%d completed bottom finish for " 1462 "proxy task %p, resuming task %p\n", 1463 gtid, taskdata, current_task)); 1464 1465 return; 1466 } 1467 #endif 1468 1469 #if OMPT_SUPPORT 1470 // For untied tasks, the first task executed only calls __kmpc_omp_task and 1471 // does not execute code. 1472 ompt_thread_info_t oldInfo; 1473 if (UNLIKELY(ompt_enabled.enabled)) { 1474 // Store the threads states and restore them after the task 1475 thread = __kmp_threads[gtid]; 1476 oldInfo = thread->th.ompt_thread_info; 1477 thread->th.ompt_thread_info.wait_id = 0; 1478 thread->th.ompt_thread_info.state = (thread->th.th_team_serialized) 1479 ? ompt_state_work_serial 1480 : ompt_state_work_parallel; 1481 taskdata->ompt_task_info.frame.exit_frame.ptr = OMPT_GET_FRAME_ADDRESS(0); 1482 } 1483 #endif 1484 1485 #if OMP_45_ENABLED 1486 // Proxy tasks are not handled by the runtime 1487 if (taskdata->td_flags.proxy != TASK_PROXY) { 1488 #endif 1489 ANNOTATE_HAPPENS_AFTER(task); 1490 __kmp_task_start(gtid, task, current_task); // OMPT only if not discarded 1491 #if OMP_45_ENABLED 1492 } 1493 #endif 1494 1495 #if OMP_40_ENABLED 1496 // TODO: cancel tasks if the parallel region has also been cancelled 1497 // TODO: check if this sequence can be hoisted above __kmp_task_start 1498 // if cancellation has been enabled for this run ... 1499 if (__kmp_omp_cancellation) { 1500 thread = __kmp_threads[gtid]; 1501 kmp_team_t *this_team = thread->th.th_team; 1502 kmp_taskgroup_t *taskgroup = taskdata->td_taskgroup; 1503 if ((taskgroup && taskgroup->cancel_request) || 1504 (this_team->t.t_cancel_request == cancel_parallel)) { 1505 #if OMPT_SUPPORT && OMPT_OPTIONAL 1506 ompt_data_t *task_data; 1507 if (UNLIKELY(ompt_enabled.ompt_callback_cancel)) { 1508 __ompt_get_task_info_internal(0, NULL, &task_data, NULL, NULL, NULL); 1509 ompt_callbacks.ompt_callback(ompt_callback_cancel)( 1510 task_data, 1511 ((taskgroup && taskgroup->cancel_request) ? ompt_cancel_taskgroup 1512 : ompt_cancel_parallel) | 1513 ompt_cancel_discarded_task, 1514 NULL); 1515 } 1516 #endif 1517 KMP_COUNT_BLOCK(TASK_cancelled); 1518 // this task belongs to a task group and we need to cancel it 1519 discard = 1 /* true */; 1520 } 1521 } 1522 1523 // Invoke the task routine and pass in relevant data. 1524 // Thunks generated by gcc take a different argument list. 1525 if (!discard) { 1526 if (taskdata->td_flags.tiedness == TASK_UNTIED) { 1527 taskdata->td_last_tied = current_task->td_last_tied; 1528 KMP_DEBUG_ASSERT(taskdata->td_last_tied); 1529 } 1530 #if KMP_STATS_ENABLED 1531 KMP_COUNT_BLOCK(TASK_executed); 1532 switch (KMP_GET_THREAD_STATE()) { 1533 case FORK_JOIN_BARRIER: 1534 KMP_PUSH_PARTITIONED_TIMER(OMP_task_join_bar); 1535 break; 1536 case PLAIN_BARRIER: 1537 KMP_PUSH_PARTITIONED_TIMER(OMP_task_plain_bar); 1538 break; 1539 case TASKYIELD: 1540 KMP_PUSH_PARTITIONED_TIMER(OMP_task_taskyield); 1541 break; 1542 case TASKWAIT: 1543 KMP_PUSH_PARTITIONED_TIMER(OMP_task_taskwait); 1544 break; 1545 case TASKGROUP: 1546 KMP_PUSH_PARTITIONED_TIMER(OMP_task_taskgroup); 1547 break; 1548 default: 1549 KMP_PUSH_PARTITIONED_TIMER(OMP_task_immediate); 1550 break; 1551 } 1552 #endif // KMP_STATS_ENABLED 1553 #endif // OMP_40_ENABLED 1554 1555 // OMPT task begin 1556 #if OMPT_SUPPORT 1557 if (UNLIKELY(ompt_enabled.enabled)) 1558 __ompt_task_start(task, current_task, gtid); 1559 #endif 1560 1561 #if USE_ITT_BUILD && USE_ITT_NOTIFY 1562 kmp_uint64 cur_time; 1563 kmp_int32 kmp_itt_count_task = 1564 __kmp_forkjoin_frames_mode == 3 && !taskdata->td_flags.task_serial && 1565 current_task->td_flags.tasktype == TASK_IMPLICIT; 1566 if (kmp_itt_count_task) { 1567 thread = __kmp_threads[gtid]; 1568 // Time outer level explicit task on barrier for adjusting imbalance time 1569 if (thread->th.th_bar_arrive_time) 1570 cur_time = __itt_get_timestamp(); 1571 else 1572 kmp_itt_count_task = 0; // thread is not on a barrier - skip timing 1573 } 1574 #endif 1575 1576 #ifdef KMP_GOMP_COMPAT 1577 if (taskdata->td_flags.native) { 1578 ((void (*)(void *))(*(task->routine)))(task->shareds); 1579 } else 1580 #endif /* KMP_GOMP_COMPAT */ 1581 { 1582 (*(task->routine))(gtid, task); 1583 } 1584 KMP_POP_PARTITIONED_TIMER(); 1585 1586 #if USE_ITT_BUILD && USE_ITT_NOTIFY 1587 if (kmp_itt_count_task) { 1588 // Barrier imbalance - adjust arrive time with the task duration 1589 thread->th.th_bar_arrive_time += (__itt_get_timestamp() - cur_time); 1590 } 1591 #endif 1592 1593 #if OMP_40_ENABLED 1594 } 1595 #endif // OMP_40_ENABLED 1596 1597 1598 #if OMP_45_ENABLED 1599 // Proxy tasks are not handled by the runtime 1600 if (taskdata->td_flags.proxy != TASK_PROXY) { 1601 #endif 1602 ANNOTATE_HAPPENS_BEFORE(taskdata->td_parent); 1603 #if OMPT_SUPPORT 1604 if (UNLIKELY(ompt_enabled.enabled)) { 1605 thread->th.ompt_thread_info = oldInfo; 1606 if (taskdata->td_flags.tiedness == TASK_TIED) { 1607 taskdata->ompt_task_info.frame.exit_frame = ompt_data_none; 1608 } 1609 __kmp_task_finish<true>(gtid, task, current_task); 1610 } else 1611 #endif 1612 __kmp_task_finish<false>(gtid, task, current_task); 1613 #if OMP_45_ENABLED 1614 } 1615 #endif 1616 1617 KA_TRACE( 1618 30, 1619 ("__kmp_invoke_task(exit): T#%d completed task %p, resuming task %p\n", 1620 gtid, taskdata, current_task)); 1621 return; 1622 } 1623 1624 // __kmpc_omp_task_parts: Schedule a thread-switchable task for execution 1625 // 1626 // loc_ref: location of original task pragma (ignored) 1627 // gtid: Global Thread ID of encountering thread 1628 // new_task: task thunk allocated by __kmp_omp_task_alloc() for the ''new task'' 1629 // Returns: 1630 // TASK_CURRENT_NOT_QUEUED (0) if did not suspend and queue current task to 1631 // be resumed later. 1632 // TASK_CURRENT_QUEUED (1) if suspended and queued the current task to be 1633 // resumed later. 1634 kmp_int32 __kmpc_omp_task_parts(ident_t *loc_ref, kmp_int32 gtid, 1635 kmp_task_t *new_task) { 1636 kmp_taskdata_t *new_taskdata = KMP_TASK_TO_TASKDATA(new_task); 1637 1638 KA_TRACE(10, ("__kmpc_omp_task_parts(enter): T#%d loc=%p task=%p\n", gtid, 1639 loc_ref, new_taskdata)); 1640 1641 #if OMPT_SUPPORT 1642 kmp_taskdata_t *parent; 1643 if (UNLIKELY(ompt_enabled.enabled)) { 1644 parent = new_taskdata->td_parent; 1645 if (ompt_enabled.ompt_callback_task_create) { 1646 ompt_data_t task_data = ompt_data_none; 1647 ompt_callbacks.ompt_callback(ompt_callback_task_create)( 1648 parent ? &(parent->ompt_task_info.task_data) : &task_data, 1649 parent ? &(parent->ompt_task_info.frame) : NULL, 1650 &(new_taskdata->ompt_task_info.task_data), ompt_task_explicit, 0, 1651 OMPT_GET_RETURN_ADDRESS(0)); 1652 } 1653 } 1654 #endif 1655 1656 /* Should we execute the new task or queue it? For now, let's just always try 1657 to queue it. If the queue fills up, then we'll execute it. */ 1658 1659 if (__kmp_push_task(gtid, new_task) == TASK_NOT_PUSHED) // if cannot defer 1660 { // Execute this task immediately 1661 kmp_taskdata_t *current_task = __kmp_threads[gtid]->th.th_current_task; 1662 new_taskdata->td_flags.task_serial = 1; 1663 __kmp_invoke_task(gtid, new_task, current_task); 1664 } 1665 1666 KA_TRACE( 1667 10, 1668 ("__kmpc_omp_task_parts(exit): T#%d returning TASK_CURRENT_NOT_QUEUED: " 1669 "loc=%p task=%p, return: TASK_CURRENT_NOT_QUEUED\n", 1670 gtid, loc_ref, new_taskdata)); 1671 1672 ANNOTATE_HAPPENS_BEFORE(new_task); 1673 #if OMPT_SUPPORT 1674 if (UNLIKELY(ompt_enabled.enabled)) { 1675 parent->ompt_task_info.frame.enter_frame = ompt_data_none; 1676 } 1677 #endif 1678 return TASK_CURRENT_NOT_QUEUED; 1679 } 1680 1681 // __kmp_omp_task: Schedule a non-thread-switchable task for execution 1682 // 1683 // gtid: Global Thread ID of encountering thread 1684 // new_task:non-thread-switchable task thunk allocated by __kmp_omp_task_alloc() 1685 // serialize_immediate: if TRUE then if the task is executed immediately its 1686 // execution will be serialized 1687 // Returns: 1688 // TASK_CURRENT_NOT_QUEUED (0) if did not suspend and queue current task to 1689 // be resumed later. 1690 // TASK_CURRENT_QUEUED (1) if suspended and queued the current task to be 1691 // resumed later. 1692 kmp_int32 __kmp_omp_task(kmp_int32 gtid, kmp_task_t *new_task, 1693 bool serialize_immediate) { 1694 kmp_taskdata_t *new_taskdata = KMP_TASK_TO_TASKDATA(new_task); 1695 1696 /* Should we execute the new task or queue it? For now, let's just always try to 1697 queue it. If the queue fills up, then we'll execute it. */ 1698 #if OMP_45_ENABLED 1699 if (new_taskdata->td_flags.proxy == TASK_PROXY || 1700 __kmp_push_task(gtid, new_task) == TASK_NOT_PUSHED) // if cannot defer 1701 #else 1702 if (__kmp_push_task(gtid, new_task) == TASK_NOT_PUSHED) // if cannot defer 1703 #endif 1704 { // Execute this task immediately 1705 kmp_taskdata_t *current_task = __kmp_threads[gtid]->th.th_current_task; 1706 if (serialize_immediate) 1707 new_taskdata->td_flags.task_serial = 1; 1708 __kmp_invoke_task(gtid, new_task, current_task); 1709 } 1710 1711 ANNOTATE_HAPPENS_BEFORE(new_task); 1712 return TASK_CURRENT_NOT_QUEUED; 1713 } 1714 1715 // __kmpc_omp_task: Wrapper around __kmp_omp_task to schedule a 1716 // non-thread-switchable task from the parent thread only! 1717 // 1718 // loc_ref: location of original task pragma (ignored) 1719 // gtid: Global Thread ID of encountering thread 1720 // new_task: non-thread-switchable task thunk allocated by 1721 // __kmp_omp_task_alloc() 1722 // Returns: 1723 // TASK_CURRENT_NOT_QUEUED (0) if did not suspend and queue current task to 1724 // be resumed later. 1725 // TASK_CURRENT_QUEUED (1) if suspended and queued the current task to be 1726 // resumed later. 1727 kmp_int32 __kmpc_omp_task(ident_t *loc_ref, kmp_int32 gtid, 1728 kmp_task_t *new_task) { 1729 kmp_int32 res; 1730 KMP_SET_THREAD_STATE_BLOCK(EXPLICIT_TASK); 1731 1732 #if KMP_DEBUG || OMPT_SUPPORT 1733 kmp_taskdata_t *new_taskdata = KMP_TASK_TO_TASKDATA(new_task); 1734 #endif 1735 KA_TRACE(10, ("__kmpc_omp_task(enter): T#%d loc=%p task=%p\n", gtid, loc_ref, 1736 new_taskdata)); 1737 1738 #if OMPT_SUPPORT 1739 kmp_taskdata_t *parent = NULL; 1740 if (UNLIKELY(ompt_enabled.enabled)) { 1741 if (!new_taskdata->td_flags.started) { 1742 OMPT_STORE_RETURN_ADDRESS(gtid); 1743 parent = new_taskdata->td_parent; 1744 if (!parent->ompt_task_info.frame.enter_frame.ptr) { 1745 parent->ompt_task_info.frame.enter_frame.ptr = OMPT_GET_FRAME_ADDRESS(0); 1746 } 1747 if (ompt_enabled.ompt_callback_task_create) { 1748 ompt_data_t task_data = ompt_data_none; 1749 ompt_callbacks.ompt_callback(ompt_callback_task_create)( 1750 parent ? &(parent->ompt_task_info.task_data) : &task_data, 1751 parent ? &(parent->ompt_task_info.frame) : NULL, 1752 &(new_taskdata->ompt_task_info.task_data), 1753 ompt_task_explicit | TASK_TYPE_DETAILS_FORMAT(new_taskdata), 0, 1754 OMPT_LOAD_RETURN_ADDRESS(gtid)); 1755 } 1756 } else { 1757 // We are scheduling the continuation of an UNTIED task. 1758 // Scheduling back to the parent task. 1759 __ompt_task_finish(new_task, 1760 new_taskdata->ompt_task_info.scheduling_parent, 1761 ompt_task_switch); 1762 new_taskdata->ompt_task_info.frame.exit_frame = ompt_data_none; 1763 } 1764 } 1765 #endif 1766 1767 res = __kmp_omp_task(gtid, new_task, true); 1768 1769 KA_TRACE(10, ("__kmpc_omp_task(exit): T#%d returning " 1770 "TASK_CURRENT_NOT_QUEUED: loc=%p task=%p\n", 1771 gtid, loc_ref, new_taskdata)); 1772 #if OMPT_SUPPORT 1773 if (UNLIKELY(ompt_enabled.enabled && parent != NULL)) { 1774 parent->ompt_task_info.frame.enter_frame = ompt_data_none; 1775 } 1776 #endif 1777 return res; 1778 } 1779 1780 // __kmp_omp_taskloop_task: Wrapper around __kmp_omp_task to schedule 1781 // a taskloop task with the correct OMPT return address 1782 // 1783 // loc_ref: location of original task pragma (ignored) 1784 // gtid: Global Thread ID of encountering thread 1785 // new_task: non-thread-switchable task thunk allocated by 1786 // __kmp_omp_task_alloc() 1787 // codeptr_ra: return address for OMPT callback 1788 // Returns: 1789 // TASK_CURRENT_NOT_QUEUED (0) if did not suspend and queue current task to 1790 // be resumed later. 1791 // TASK_CURRENT_QUEUED (1) if suspended and queued the current task to be 1792 // resumed later. 1793 kmp_int32 __kmp_omp_taskloop_task(ident_t *loc_ref, kmp_int32 gtid, 1794 kmp_task_t *new_task, void *codeptr_ra) { 1795 kmp_int32 res; 1796 KMP_SET_THREAD_STATE_BLOCK(EXPLICIT_TASK); 1797 1798 #if KMP_DEBUG || OMPT_SUPPORT 1799 kmp_taskdata_t *new_taskdata = KMP_TASK_TO_TASKDATA(new_task); 1800 #endif 1801 KA_TRACE(10, ("__kmpc_omp_task(enter): T#%d loc=%p task=%p\n", gtid, loc_ref, 1802 new_taskdata)); 1803 1804 #if OMPT_SUPPORT 1805 kmp_taskdata_t *parent = NULL; 1806 if (UNLIKELY(ompt_enabled.enabled && !new_taskdata->td_flags.started)) { 1807 parent = new_taskdata->td_parent; 1808 if (!parent->ompt_task_info.frame.enter_frame.ptr) 1809 parent->ompt_task_info.frame.enter_frame.ptr = OMPT_GET_FRAME_ADDRESS(0); 1810 if (ompt_enabled.ompt_callback_task_create) { 1811 ompt_data_t task_data = ompt_data_none; 1812 ompt_callbacks.ompt_callback(ompt_callback_task_create)( 1813 parent ? &(parent->ompt_task_info.task_data) : &task_data, 1814 parent ? &(parent->ompt_task_info.frame) : NULL, 1815 &(new_taskdata->ompt_task_info.task_data), 1816 ompt_task_explicit | TASK_TYPE_DETAILS_FORMAT(new_taskdata), 0, 1817 codeptr_ra); 1818 } 1819 } 1820 #endif 1821 1822 res = __kmp_omp_task(gtid, new_task, true); 1823 1824 KA_TRACE(10, ("__kmpc_omp_task(exit): T#%d returning " 1825 "TASK_CURRENT_NOT_QUEUED: loc=%p task=%p\n", 1826 gtid, loc_ref, new_taskdata)); 1827 #if OMPT_SUPPORT 1828 if (UNLIKELY(ompt_enabled.enabled && parent != NULL)) { 1829 parent->ompt_task_info.frame.enter_frame = ompt_data_none; 1830 } 1831 #endif 1832 return res; 1833 } 1834 1835 template <bool ompt> 1836 static kmp_int32 __kmpc_omp_taskwait_template(ident_t *loc_ref, kmp_int32 gtid, 1837 void *frame_address, 1838 void *return_address) { 1839 kmp_taskdata_t *taskdata; 1840 kmp_info_t *thread; 1841 int thread_finished = FALSE; 1842 KMP_SET_THREAD_STATE_BLOCK(TASKWAIT); 1843 1844 KA_TRACE(10, ("__kmpc_omp_taskwait(enter): T#%d loc=%p\n", gtid, loc_ref)); 1845 1846 if (__kmp_tasking_mode != tskm_immediate_exec) { 1847 thread = __kmp_threads[gtid]; 1848 taskdata = thread->th.th_current_task; 1849 1850 #if OMPT_SUPPORT && OMPT_OPTIONAL 1851 ompt_data_t *my_task_data; 1852 ompt_data_t *my_parallel_data; 1853 1854 if (ompt) { 1855 my_task_data = &(taskdata->ompt_task_info.task_data); 1856 my_parallel_data = OMPT_CUR_TEAM_DATA(thread); 1857 1858 taskdata->ompt_task_info.frame.enter_frame.ptr = frame_address; 1859 1860 if (ompt_enabled.ompt_callback_sync_region) { 1861 ompt_callbacks.ompt_callback(ompt_callback_sync_region)( 1862 ompt_sync_region_taskwait, ompt_scope_begin, my_parallel_data, 1863 my_task_data, return_address); 1864 } 1865 1866 if (ompt_enabled.ompt_callback_sync_region_wait) { 1867 ompt_callbacks.ompt_callback(ompt_callback_sync_region_wait)( 1868 ompt_sync_region_taskwait, ompt_scope_begin, my_parallel_data, 1869 my_task_data, return_address); 1870 } 1871 } 1872 #endif // OMPT_SUPPORT && OMPT_OPTIONAL 1873 1874 // Debugger: The taskwait is active. Store location and thread encountered the 1875 // taskwait. 1876 #if USE_ITT_BUILD 1877 // Note: These values are used by ITT events as well. 1878 #endif /* USE_ITT_BUILD */ 1879 taskdata->td_taskwait_counter += 1; 1880 taskdata->td_taskwait_ident = loc_ref; 1881 taskdata->td_taskwait_thread = gtid + 1; 1882 1883 #if USE_ITT_BUILD 1884 void *itt_sync_obj = __kmp_itt_taskwait_object(gtid); 1885 if (itt_sync_obj != NULL) 1886 __kmp_itt_taskwait_starting(gtid, itt_sync_obj); 1887 #endif /* USE_ITT_BUILD */ 1888 1889 bool must_wait = 1890 !taskdata->td_flags.team_serial && !taskdata->td_flags.final; 1891 1892 #if OMP_45_ENABLED 1893 must_wait = must_wait || (thread->th.th_task_team != NULL && 1894 thread->th.th_task_team->tt.tt_found_proxy_tasks); 1895 #endif 1896 if (must_wait) { 1897 kmp_flag_32 flag(RCAST(std::atomic<kmp_uint32> *, 1898 &(taskdata->td_incomplete_child_tasks)), 1899 0U); 1900 while (KMP_ATOMIC_LD_ACQ(&taskdata->td_incomplete_child_tasks) != 0) { 1901 flag.execute_tasks(thread, gtid, FALSE, 1902 &thread_finished USE_ITT_BUILD_ARG(itt_sync_obj), 1903 __kmp_task_stealing_constraint); 1904 } 1905 } 1906 #if USE_ITT_BUILD 1907 if (itt_sync_obj != NULL) 1908 __kmp_itt_taskwait_finished(gtid, itt_sync_obj); 1909 #endif /* USE_ITT_BUILD */ 1910 1911 // Debugger: The taskwait is completed. Location remains, but thread is 1912 // negated. 1913 taskdata->td_taskwait_thread = -taskdata->td_taskwait_thread; 1914 1915 #if OMPT_SUPPORT && OMPT_OPTIONAL 1916 if (ompt) { 1917 if (ompt_enabled.ompt_callback_sync_region_wait) { 1918 ompt_callbacks.ompt_callback(ompt_callback_sync_region_wait)( 1919 ompt_sync_region_taskwait, ompt_scope_end, my_parallel_data, 1920 my_task_data, return_address); 1921 } 1922 if (ompt_enabled.ompt_callback_sync_region) { 1923 ompt_callbacks.ompt_callback(ompt_callback_sync_region)( 1924 ompt_sync_region_taskwait, ompt_scope_end, my_parallel_data, 1925 my_task_data, return_address); 1926 } 1927 taskdata->ompt_task_info.frame.enter_frame = ompt_data_none; 1928 } 1929 #endif // OMPT_SUPPORT && OMPT_OPTIONAL 1930 1931 ANNOTATE_HAPPENS_AFTER(taskdata); 1932 } 1933 1934 KA_TRACE(10, ("__kmpc_omp_taskwait(exit): T#%d task %p finished waiting, " 1935 "returning TASK_CURRENT_NOT_QUEUED\n", 1936 gtid, taskdata)); 1937 1938 return TASK_CURRENT_NOT_QUEUED; 1939 } 1940 1941 #if OMPT_SUPPORT && OMPT_OPTIONAL 1942 OMPT_NOINLINE 1943 static kmp_int32 __kmpc_omp_taskwait_ompt(ident_t *loc_ref, kmp_int32 gtid, 1944 void *frame_address, 1945 void *return_address) { 1946 return __kmpc_omp_taskwait_template<true>(loc_ref, gtid, frame_address, 1947 return_address); 1948 } 1949 #endif // OMPT_SUPPORT && OMPT_OPTIONAL 1950 1951 // __kmpc_omp_taskwait: Wait until all tasks generated by the current task are 1952 // complete 1953 kmp_int32 __kmpc_omp_taskwait(ident_t *loc_ref, kmp_int32 gtid) { 1954 #if OMPT_SUPPORT && OMPT_OPTIONAL 1955 if (UNLIKELY(ompt_enabled.enabled)) { 1956 OMPT_STORE_RETURN_ADDRESS(gtid); 1957 return __kmpc_omp_taskwait_ompt(loc_ref, gtid, OMPT_GET_FRAME_ADDRESS(0), 1958 OMPT_LOAD_RETURN_ADDRESS(gtid)); 1959 } 1960 #endif 1961 return __kmpc_omp_taskwait_template<false>(loc_ref, gtid, NULL, NULL); 1962 } 1963 1964 // __kmpc_omp_taskyield: switch to a different task 1965 kmp_int32 __kmpc_omp_taskyield(ident_t *loc_ref, kmp_int32 gtid, int end_part) { 1966 kmp_taskdata_t *taskdata; 1967 kmp_info_t *thread; 1968 int thread_finished = FALSE; 1969 1970 KMP_COUNT_BLOCK(OMP_TASKYIELD); 1971 KMP_SET_THREAD_STATE_BLOCK(TASKYIELD); 1972 1973 KA_TRACE(10, ("__kmpc_omp_taskyield(enter): T#%d loc=%p end_part = %d\n", 1974 gtid, loc_ref, end_part)); 1975 1976 if (__kmp_tasking_mode != tskm_immediate_exec && __kmp_init_parallel) { 1977 thread = __kmp_threads[gtid]; 1978 taskdata = thread->th.th_current_task; 1979 // Should we model this as a task wait or not? 1980 // Debugger: The taskwait is active. Store location and thread encountered the 1981 // taskwait. 1982 #if USE_ITT_BUILD 1983 // Note: These values are used by ITT events as well. 1984 #endif /* USE_ITT_BUILD */ 1985 taskdata->td_taskwait_counter += 1; 1986 taskdata->td_taskwait_ident = loc_ref; 1987 taskdata->td_taskwait_thread = gtid + 1; 1988 1989 #if USE_ITT_BUILD 1990 void *itt_sync_obj = __kmp_itt_taskwait_object(gtid); 1991 if (itt_sync_obj != NULL) 1992 __kmp_itt_taskwait_starting(gtid, itt_sync_obj); 1993 #endif /* USE_ITT_BUILD */ 1994 if (!taskdata->td_flags.team_serial) { 1995 kmp_task_team_t *task_team = thread->th.th_task_team; 1996 if (task_team != NULL) { 1997 if (KMP_TASKING_ENABLED(task_team)) { 1998 #if OMPT_SUPPORT 1999 if (UNLIKELY(ompt_enabled.enabled)) 2000 thread->th.ompt_thread_info.ompt_task_yielded = 1; 2001 #endif 2002 __kmp_execute_tasks_32( 2003 thread, gtid, NULL, FALSE, 2004 &thread_finished USE_ITT_BUILD_ARG(itt_sync_obj), 2005 __kmp_task_stealing_constraint); 2006 #if OMPT_SUPPORT 2007 if (UNLIKELY(ompt_enabled.enabled)) 2008 thread->th.ompt_thread_info.ompt_task_yielded = 0; 2009 #endif 2010 } 2011 } 2012 } 2013 #if USE_ITT_BUILD 2014 if (itt_sync_obj != NULL) 2015 __kmp_itt_taskwait_finished(gtid, itt_sync_obj); 2016 #endif /* USE_ITT_BUILD */ 2017 2018 // Debugger: The taskwait is completed. Location remains, but thread is 2019 // negated. 2020 taskdata->td_taskwait_thread = -taskdata->td_taskwait_thread; 2021 } 2022 2023 KA_TRACE(10, ("__kmpc_omp_taskyield(exit): T#%d task %p resuming, " 2024 "returning TASK_CURRENT_NOT_QUEUED\n", 2025 gtid, taskdata)); 2026 2027 return TASK_CURRENT_NOT_QUEUED; 2028 } 2029 2030 #if OMP_50_ENABLED 2031 // Task Reduction implementation 2032 // 2033 // Note: initial implementation didn't take into account the possibility 2034 // to specify omp_orig for initializer of the UDR (user defined reduction). 2035 // Corrected implementation takes into account the omp_orig object. 2036 // Compiler is free to use old implementation if omp_orig is not specified. 2037 2038 /*! 2039 @ingroup BASIC_TYPES 2040 @{ 2041 */ 2042 2043 /*! 2044 Flags for special info per task reduction item. 2045 */ 2046 typedef struct kmp_taskred_flags { 2047 /*! 1 - use lazy alloc/init (e.g. big objects, #tasks < #threads) */ 2048 unsigned lazy_priv : 1; 2049 unsigned reserved31 : 31; 2050 } kmp_taskred_flags_t; 2051 2052 /*! 2053 Internal struct for reduction data item related info set up by compiler. 2054 */ 2055 typedef struct kmp_task_red_input { 2056 void *reduce_shar; /**< shared between tasks item to reduce into */ 2057 size_t reduce_size; /**< size of data item in bytes */ 2058 // three compiler-generated routines (init, fini are optional): 2059 void *reduce_init; /**< data initialization routine (single parameter) */ 2060 void *reduce_fini; /**< data finalization routine */ 2061 void *reduce_comb; /**< data combiner routine */ 2062 kmp_taskred_flags_t flags; /**< flags for additional info from compiler */ 2063 } kmp_task_red_input_t; 2064 2065 /*! 2066 Internal struct for reduction data item related info saved by the library. 2067 */ 2068 typedef struct kmp_taskred_data { 2069 void *reduce_shar; /**< shared between tasks item to reduce into */ 2070 size_t reduce_size; /**< size of data item */ 2071 kmp_taskred_flags_t flags; /**< flags for additional info from compiler */ 2072 void *reduce_priv; /**< array of thread specific items */ 2073 void *reduce_pend; /**< end of private data for faster comparison op */ 2074 // three compiler-generated routines (init, fini are optional): 2075 void *reduce_comb; /**< data combiner routine */ 2076 void *reduce_init; /**< data initialization routine (two parameters) */ 2077 void *reduce_fini; /**< data finalization routine */ 2078 void *reduce_orig; /**< original item (can be used in UDR initializer) */ 2079 } kmp_taskred_data_t; 2080 2081 /*! 2082 Internal struct for reduction data item related info set up by compiler. 2083 2084 New interface: added reduce_orig field to provide omp_orig for UDR initializer. 2085 */ 2086 typedef struct kmp_taskred_input { 2087 void *reduce_shar; /**< shared between tasks item to reduce into */ 2088 void *reduce_orig; /**< original reduction item used for initialization */ 2089 size_t reduce_size; /**< size of data item */ 2090 // three compiler-generated routines (init, fini are optional): 2091 void *reduce_init; /**< data initialization routine (two parameters) */ 2092 void *reduce_fini; /**< data finalization routine */ 2093 void *reduce_comb; /**< data combiner routine */ 2094 kmp_taskred_flags_t flags; /**< flags for additional info from compiler */ 2095 } kmp_taskred_input_t; 2096 /*! 2097 @} 2098 */ 2099 2100 template <typename T> void __kmp_assign_orig(kmp_taskred_data_t &item, T &src); 2101 template <> 2102 void __kmp_assign_orig<kmp_task_red_input_t>(kmp_taskred_data_t &item, 2103 kmp_task_red_input_t &src) { 2104 item.reduce_orig = NULL; 2105 } 2106 template <> 2107 void __kmp_assign_orig<kmp_taskred_input_t>(kmp_taskred_data_t &item, 2108 kmp_taskred_input_t &src) { 2109 if (src.reduce_orig != NULL) { 2110 item.reduce_orig = src.reduce_orig; 2111 } else { 2112 item.reduce_orig = src.reduce_shar; 2113 } // non-NULL reduce_orig means new interface used 2114 } 2115 2116 template <typename T> void __kmp_call_init(kmp_taskred_data_t &item, int j); 2117 template <> 2118 void __kmp_call_init<kmp_task_red_input_t>(kmp_taskred_data_t &item, 2119 int offset) { 2120 ((void (*)(void *))item.reduce_init)((char *)(item.reduce_priv) + offset); 2121 } 2122 template <> 2123 void __kmp_call_init<kmp_taskred_input_t>(kmp_taskred_data_t &item, 2124 int offset) { 2125 ((void (*)(void *, void *))item.reduce_init)( 2126 (char *)(item.reduce_priv) + offset, item.reduce_orig); 2127 } 2128 2129 template <typename T> 2130 void *__kmp_task_reduction_init(int gtid, int num, T *data) { 2131 kmp_info_t *thread = __kmp_threads[gtid]; 2132 kmp_taskgroup_t *tg = thread->th.th_current_task->td_taskgroup; 2133 kmp_int32 nth = thread->th.th_team_nproc; 2134 kmp_taskred_data_t *arr; 2135 2136 // check input data just in case 2137 KMP_ASSERT(tg != NULL); 2138 KMP_ASSERT(data != NULL); 2139 KMP_ASSERT(num > 0); 2140 if (nth == 1) { 2141 KA_TRACE(10, ("__kmpc_task_reduction_init: T#%d, tg %p, exiting nth=1\n", 2142 gtid, tg)); 2143 return (void *)tg; 2144 } 2145 KA_TRACE(10, ("__kmpc_task_reduction_init: T#%d, taskgroup %p, #items %d\n", 2146 gtid, tg, num)); 2147 arr = (kmp_taskred_data_t *)__kmp_thread_malloc( 2148 thread, num * sizeof(kmp_taskred_data_t)); 2149 for (int i = 0; i < num; ++i) { 2150 size_t size = data[i].reduce_size - 1; 2151 // round the size up to cache line per thread-specific item 2152 size += CACHE_LINE - size % CACHE_LINE; 2153 KMP_ASSERT(data[i].reduce_comb != NULL); // combiner is mandatory 2154 arr[i].reduce_shar = data[i].reduce_shar; 2155 arr[i].reduce_size = size; 2156 arr[i].flags = data[i].flags; 2157 arr[i].reduce_comb = data[i].reduce_comb; 2158 arr[i].reduce_init = data[i].reduce_init; 2159 arr[i].reduce_fini = data[i].reduce_fini; 2160 __kmp_assign_orig<T>(arr[i], data[i]); 2161 if (!arr[i].flags.lazy_priv) { 2162 // allocate cache-line aligned block and fill it with zeros 2163 arr[i].reduce_priv = __kmp_allocate(nth * size); 2164 arr[i].reduce_pend = (char *)(arr[i].reduce_priv) + nth * size; 2165 if (arr[i].reduce_init != NULL) { 2166 // initialize all thread-specific items 2167 for (int j = 0; j < nth; ++j) { 2168 __kmp_call_init<T>(arr[i], j * size); 2169 } 2170 } 2171 } else { 2172 // only allocate space for pointers now, 2173 // objects will be lazily allocated/initialized if/when requested 2174 // note that __kmp_allocate zeroes the allocated memory 2175 arr[i].reduce_priv = __kmp_allocate(nth * sizeof(void *)); 2176 } 2177 } 2178 tg->reduce_data = (void *)arr; 2179 tg->reduce_num_data = num; 2180 return (void *)tg; 2181 } 2182 2183 /*! 2184 @ingroup TASKING 2185 @param gtid Global thread ID 2186 @param num Number of data items to reduce 2187 @param data Array of data for reduction 2188 @return The taskgroup identifier 2189 2190 Initialize task reduction for the taskgroup. 2191 2192 Note: this entry supposes the optional compiler-generated initializer routine 2193 has single parameter - pointer to object to be initialized. That means 2194 the reduction either does not use omp_orig object, or the omp_orig is accessible 2195 without help of the runtime library. 2196 */ 2197 void *__kmpc_task_reduction_init(int gtid, int num, void *data) { 2198 return __kmp_task_reduction_init(gtid, num, (kmp_task_red_input_t *)data); 2199 } 2200 2201 /*! 2202 @ingroup TASKING 2203 @param gtid Global thread ID 2204 @param num Number of data items to reduce 2205 @param data Array of data for reduction 2206 @return The taskgroup identifier 2207 2208 Initialize task reduction for the taskgroup. 2209 2210 Note: this entry supposes the optional compiler-generated initializer routine 2211 has two parameters, pointer to object to be initialized and pointer to omp_orig 2212 */ 2213 void *__kmpc_taskred_init(int gtid, int num, void *data) { 2214 return __kmp_task_reduction_init(gtid, num, (kmp_taskred_input_t *)data); 2215 } 2216 2217 // Copy task reduction data (except for shared pointers). 2218 template <typename T> 2219 void __kmp_task_reduction_init_copy(kmp_info_t *thr, int num, T *data, 2220 kmp_taskgroup_t *tg, void *reduce_data) { 2221 kmp_taskred_data_t *arr; 2222 KA_TRACE(20, ("__kmp_task_reduction_init_copy: Th %p, init taskgroup %p," 2223 " from data %p\n", 2224 thr, tg, reduce_data)); 2225 arr = (kmp_taskred_data_t *)__kmp_thread_malloc( 2226 thr, num * sizeof(kmp_taskred_data_t)); 2227 // threads will share private copies, thunk routines, sizes, flags, etc.: 2228 KMP_MEMCPY(arr, reduce_data, num * sizeof(kmp_taskred_data_t)); 2229 for (int i = 0; i < num; ++i) { 2230 arr[i].reduce_shar = data[i].reduce_shar; // init unique shared pointers 2231 } 2232 tg->reduce_data = (void *)arr; 2233 tg->reduce_num_data = num; 2234 } 2235 2236 /*! 2237 @ingroup TASKING 2238 @param gtid Global thread ID 2239 @param tskgrp The taskgroup ID (optional) 2240 @param data Shared location of the item 2241 @return The pointer to per-thread data 2242 2243 Get thread-specific location of data item 2244 */ 2245 void *__kmpc_task_reduction_get_th_data(int gtid, void *tskgrp, void *data) { 2246 kmp_info_t *thread = __kmp_threads[gtid]; 2247 kmp_int32 nth = thread->th.th_team_nproc; 2248 if (nth == 1) 2249 return data; // nothing to do 2250 2251 kmp_taskgroup_t *tg = (kmp_taskgroup_t *)tskgrp; 2252 if (tg == NULL) 2253 tg = thread->th.th_current_task->td_taskgroup; 2254 KMP_ASSERT(tg != NULL); 2255 kmp_taskred_data_t *arr = (kmp_taskred_data_t *)(tg->reduce_data); 2256 kmp_int32 num = tg->reduce_num_data; 2257 kmp_int32 tid = thread->th.th_info.ds.ds_tid; 2258 2259 KMP_ASSERT(data != NULL); 2260 while (tg != NULL) { 2261 for (int i = 0; i < num; ++i) { 2262 if (!arr[i].flags.lazy_priv) { 2263 if (data == arr[i].reduce_shar || 2264 (data >= arr[i].reduce_priv && data < arr[i].reduce_pend)) 2265 return (char *)(arr[i].reduce_priv) + tid * arr[i].reduce_size; 2266 } else { 2267 // check shared location first 2268 void **p_priv = (void **)(arr[i].reduce_priv); 2269 if (data == arr[i].reduce_shar) 2270 goto found; 2271 // check if we get some thread specific location as parameter 2272 for (int j = 0; j < nth; ++j) 2273 if (data == p_priv[j]) 2274 goto found; 2275 continue; // not found, continue search 2276 found: 2277 if (p_priv[tid] == NULL) { 2278 // allocate thread specific object lazily 2279 p_priv[tid] = __kmp_allocate(arr[i].reduce_size); 2280 if (arr[i].reduce_init != NULL) { 2281 if (arr[i].reduce_orig != NULL) { // new interface 2282 ((void (*)(void *, void *))arr[i].reduce_init)( 2283 p_priv[tid], arr[i].reduce_orig); 2284 } else { // old interface (single parameter) 2285 ((void (*)(void *))arr[i].reduce_init)(p_priv[tid]); 2286 } 2287 } 2288 } 2289 return p_priv[tid]; 2290 } 2291 } 2292 tg = tg->parent; 2293 arr = (kmp_taskred_data_t *)(tg->reduce_data); 2294 num = tg->reduce_num_data; 2295 } 2296 KMP_ASSERT2(0, "Unknown task reduction item"); 2297 return NULL; // ERROR, this line never executed 2298 } 2299 2300 // Finalize task reduction. 2301 // Called from __kmpc_end_taskgroup() 2302 static void __kmp_task_reduction_fini(kmp_info_t *th, kmp_taskgroup_t *tg) { 2303 kmp_int32 nth = th->th.th_team_nproc; 2304 KMP_DEBUG_ASSERT(nth > 1); // should not be called if nth == 1 2305 kmp_taskred_data_t *arr = (kmp_taskred_data_t *)tg->reduce_data; 2306 kmp_int32 num = tg->reduce_num_data; 2307 for (int i = 0; i < num; ++i) { 2308 void *sh_data = arr[i].reduce_shar; 2309 void (*f_fini)(void *) = (void (*)(void *))(arr[i].reduce_fini); 2310 void (*f_comb)(void *, void *) = 2311 (void (*)(void *, void *))(arr[i].reduce_comb); 2312 if (!arr[i].flags.lazy_priv) { 2313 void *pr_data = arr[i].reduce_priv; 2314 size_t size = arr[i].reduce_size; 2315 for (int j = 0; j < nth; ++j) { 2316 void *priv_data = (char *)pr_data + j * size; 2317 f_comb(sh_data, priv_data); // combine results 2318 if (f_fini) 2319 f_fini(priv_data); // finalize if needed 2320 } 2321 } else { 2322 void **pr_data = (void **)(arr[i].reduce_priv); 2323 for (int j = 0; j < nth; ++j) { 2324 if (pr_data[j] != NULL) { 2325 f_comb(sh_data, pr_data[j]); // combine results 2326 if (f_fini) 2327 f_fini(pr_data[j]); // finalize if needed 2328 __kmp_free(pr_data[j]); 2329 } 2330 } 2331 } 2332 __kmp_free(arr[i].reduce_priv); 2333 } 2334 __kmp_thread_free(th, arr); 2335 tg->reduce_data = NULL; 2336 tg->reduce_num_data = 0; 2337 } 2338 2339 // Cleanup task reduction data for parallel or worksharing, 2340 // do not touch task private data other threads still working with. 2341 // Called from __kmpc_end_taskgroup() 2342 static void __kmp_task_reduction_clean(kmp_info_t *th, kmp_taskgroup_t *tg) { 2343 __kmp_thread_free(th, tg->reduce_data); 2344 tg->reduce_data = NULL; 2345 tg->reduce_num_data = 0; 2346 } 2347 2348 template <typename T> 2349 void *__kmp_task_reduction_modifier_init(ident_t *loc, int gtid, int is_ws, 2350 int num, T *data) { 2351 kmp_info_t *thr = __kmp_threads[gtid]; 2352 kmp_int32 nth = thr->th.th_team_nproc; 2353 __kmpc_taskgroup(loc, gtid); // form new taskgroup first 2354 if (nth == 1) { 2355 KA_TRACE(10, 2356 ("__kmpc_reduction_modifier_init: T#%d, tg %p, exiting nth=1\n", 2357 gtid, thr->th.th_current_task->td_taskgroup)); 2358 return (void *)thr->th.th_current_task->td_taskgroup; 2359 } 2360 kmp_team_t *team = thr->th.th_team; 2361 void *reduce_data; 2362 kmp_taskgroup_t *tg; 2363 reduce_data = KMP_ATOMIC_LD_RLX(&team->t.t_tg_reduce_data[is_ws]); 2364 if (reduce_data == NULL && 2365 __kmp_atomic_compare_store(&team->t.t_tg_reduce_data[is_ws], reduce_data, 2366 (void *)1)) { 2367 // single thread enters this block to initialize common reduction data 2368 KMP_DEBUG_ASSERT(reduce_data == NULL); 2369 // first initialize own data, then make a copy other threads can use 2370 tg = (kmp_taskgroup_t *)__kmp_task_reduction_init<T>(gtid, num, data); 2371 reduce_data = __kmp_thread_malloc(thr, num * sizeof(kmp_taskred_data_t)); 2372 KMP_MEMCPY(reduce_data, tg->reduce_data, num * sizeof(kmp_taskred_data_t)); 2373 // fini counters should be 0 at this point 2374 KMP_DEBUG_ASSERT(KMP_ATOMIC_LD_RLX(&team->t.t_tg_fini_counter[0]) == 0); 2375 KMP_DEBUG_ASSERT(KMP_ATOMIC_LD_RLX(&team->t.t_tg_fini_counter[1]) == 0); 2376 KMP_ATOMIC_ST_REL(&team->t.t_tg_reduce_data[is_ws], reduce_data); 2377 } else { 2378 while ( 2379 (reduce_data = KMP_ATOMIC_LD_ACQ(&team->t.t_tg_reduce_data[is_ws])) == 2380 (void *)1) { // wait for task reduction initialization 2381 KMP_CPU_PAUSE(); 2382 } 2383 KMP_DEBUG_ASSERT(reduce_data > (void *)1); // should be valid pointer here 2384 tg = thr->th.th_current_task->td_taskgroup; 2385 __kmp_task_reduction_init_copy<T>(thr, num, data, tg, reduce_data); 2386 } 2387 return tg; 2388 } 2389 2390 /*! 2391 @ingroup TASKING 2392 @param loc Source location info 2393 @param gtid Global thread ID 2394 @param is_ws Is 1 if the reduction is for worksharing, 0 otherwise 2395 @param num Number of data items to reduce 2396 @param data Array of data for reduction 2397 @return The taskgroup identifier 2398 2399 Initialize task reduction for a parallel or worksharing. 2400 2401 Note: this entry supposes the optional compiler-generated initializer routine 2402 has single parameter - pointer to object to be initialized. That means 2403 the reduction either does not use omp_orig object, or the omp_orig is accessible 2404 without help of the runtime library. 2405 */ 2406 void *__kmpc_task_reduction_modifier_init(ident_t *loc, int gtid, int is_ws, 2407 int num, void *data) { 2408 return __kmp_task_reduction_modifier_init(loc, gtid, is_ws, num, 2409 (kmp_task_red_input_t *)data); 2410 } 2411 2412 /*! 2413 @ingroup TASKING 2414 @param loc Source location info 2415 @param gtid Global thread ID 2416 @param is_ws Is 1 if the reduction is for worksharing, 0 otherwise 2417 @param num Number of data items to reduce 2418 @param data Array of data for reduction 2419 @return The taskgroup identifier 2420 2421 Initialize task reduction for a parallel or worksharing. 2422 2423 Note: this entry supposes the optional compiler-generated initializer routine 2424 has two parameters, pointer to object to be initialized and pointer to omp_orig 2425 */ 2426 void *__kmpc_taskred_modifier_init(ident_t *loc, int gtid, int is_ws, int num, 2427 void *data) { 2428 return __kmp_task_reduction_modifier_init(loc, gtid, is_ws, num, 2429 (kmp_taskred_input_t *)data); 2430 } 2431 2432 /*! 2433 @ingroup TASKING 2434 @param loc Source location info 2435 @param gtid Global thread ID 2436 @param is_ws Is 1 if the reduction is for worksharing, 0 otherwise 2437 2438 Finalize task reduction for a parallel or worksharing. 2439 */ 2440 void __kmpc_task_reduction_modifier_fini(ident_t *loc, int gtid, int is_ws) { 2441 __kmpc_end_taskgroup(loc, gtid); 2442 } 2443 #endif 2444 2445 #if OMP_40_ENABLED 2446 // __kmpc_taskgroup: Start a new taskgroup 2447 void __kmpc_taskgroup(ident_t *loc, int gtid) { 2448 kmp_info_t *thread = __kmp_threads[gtid]; 2449 kmp_taskdata_t *taskdata = thread->th.th_current_task; 2450 kmp_taskgroup_t *tg_new = 2451 (kmp_taskgroup_t *)__kmp_thread_malloc(thread, sizeof(kmp_taskgroup_t)); 2452 KA_TRACE(10, ("__kmpc_taskgroup: T#%d loc=%p group=%p\n", gtid, loc, tg_new)); 2453 KMP_ATOMIC_ST_RLX(&tg_new->count, 0); 2454 KMP_ATOMIC_ST_RLX(&tg_new->cancel_request, cancel_noreq); 2455 tg_new->parent = taskdata->td_taskgroup; 2456 #if OMP_50_ENABLED 2457 tg_new->reduce_data = NULL; 2458 tg_new->reduce_num_data = 0; 2459 #endif 2460 taskdata->td_taskgroup = tg_new; 2461 2462 #if OMPT_SUPPORT && OMPT_OPTIONAL 2463 if (UNLIKELY(ompt_enabled.ompt_callback_sync_region)) { 2464 void *codeptr = OMPT_LOAD_RETURN_ADDRESS(gtid); 2465 if (!codeptr) 2466 codeptr = OMPT_GET_RETURN_ADDRESS(0); 2467 kmp_team_t *team = thread->th.th_team; 2468 ompt_data_t my_task_data = taskdata->ompt_task_info.task_data; 2469 // FIXME: I think this is wrong for lwt! 2470 ompt_data_t my_parallel_data = team->t.ompt_team_info.parallel_data; 2471 2472 ompt_callbacks.ompt_callback(ompt_callback_sync_region)( 2473 ompt_sync_region_taskgroup, ompt_scope_begin, &(my_parallel_data), 2474 &(my_task_data), codeptr); 2475 } 2476 #endif 2477 } 2478 2479 // __kmpc_end_taskgroup: Wait until all tasks generated by the current task 2480 // and its descendants are complete 2481 void __kmpc_end_taskgroup(ident_t *loc, int gtid) { 2482 kmp_info_t *thread = __kmp_threads[gtid]; 2483 kmp_taskdata_t *taskdata = thread->th.th_current_task; 2484 kmp_taskgroup_t *taskgroup = taskdata->td_taskgroup; 2485 int thread_finished = FALSE; 2486 2487 #if OMPT_SUPPORT && OMPT_OPTIONAL 2488 kmp_team_t *team; 2489 ompt_data_t my_task_data; 2490 ompt_data_t my_parallel_data; 2491 void *codeptr; 2492 if (UNLIKELY(ompt_enabled.enabled)) { 2493 team = thread->th.th_team; 2494 my_task_data = taskdata->ompt_task_info.task_data; 2495 // FIXME: I think this is wrong for lwt! 2496 my_parallel_data = team->t.ompt_team_info.parallel_data; 2497 codeptr = OMPT_LOAD_RETURN_ADDRESS(gtid); 2498 if (!codeptr) 2499 codeptr = OMPT_GET_RETURN_ADDRESS(0); 2500 } 2501 #endif 2502 2503 KA_TRACE(10, ("__kmpc_end_taskgroup(enter): T#%d loc=%p\n", gtid, loc)); 2504 KMP_DEBUG_ASSERT(taskgroup != NULL); 2505 KMP_SET_THREAD_STATE_BLOCK(TASKGROUP); 2506 2507 if (__kmp_tasking_mode != tskm_immediate_exec) { 2508 // mark task as waiting not on a barrier 2509 taskdata->td_taskwait_counter += 1; 2510 taskdata->td_taskwait_ident = loc; 2511 taskdata->td_taskwait_thread = gtid + 1; 2512 #if USE_ITT_BUILD 2513 // For ITT the taskgroup wait is similar to taskwait until we need to 2514 // distinguish them 2515 void *itt_sync_obj = __kmp_itt_taskwait_object(gtid); 2516 if (itt_sync_obj != NULL) 2517 __kmp_itt_taskwait_starting(gtid, itt_sync_obj); 2518 #endif /* USE_ITT_BUILD */ 2519 2520 #if OMPT_SUPPORT && OMPT_OPTIONAL 2521 if (UNLIKELY(ompt_enabled.ompt_callback_sync_region_wait)) { 2522 ompt_callbacks.ompt_callback(ompt_callback_sync_region_wait)( 2523 ompt_sync_region_taskgroup, ompt_scope_begin, &(my_parallel_data), 2524 &(my_task_data), codeptr); 2525 } 2526 #endif 2527 2528 #if OMP_45_ENABLED 2529 if (!taskdata->td_flags.team_serial || 2530 (thread->th.th_task_team != NULL && 2531 thread->th.th_task_team->tt.tt_found_proxy_tasks)) 2532 #else 2533 if (!taskdata->td_flags.team_serial) 2534 #endif 2535 { 2536 kmp_flag_32 flag(RCAST(std::atomic<kmp_uint32> *, &(taskgroup->count)), 2537 0U); 2538 while (KMP_ATOMIC_LD_ACQ(&taskgroup->count) != 0) { 2539 flag.execute_tasks(thread, gtid, FALSE, 2540 &thread_finished USE_ITT_BUILD_ARG(itt_sync_obj), 2541 __kmp_task_stealing_constraint); 2542 } 2543 } 2544 taskdata->td_taskwait_thread = -taskdata->td_taskwait_thread; // end waiting 2545 2546 #if OMPT_SUPPORT && OMPT_OPTIONAL 2547 if (UNLIKELY(ompt_enabled.ompt_callback_sync_region_wait)) { 2548 ompt_callbacks.ompt_callback(ompt_callback_sync_region_wait)( 2549 ompt_sync_region_taskgroup, ompt_scope_end, &(my_parallel_data), 2550 &(my_task_data), codeptr); 2551 } 2552 #endif 2553 2554 #if USE_ITT_BUILD 2555 if (itt_sync_obj != NULL) 2556 __kmp_itt_taskwait_finished(gtid, itt_sync_obj); 2557 #endif /* USE_ITT_BUILD */ 2558 } 2559 KMP_DEBUG_ASSERT(taskgroup->count == 0); 2560 2561 #if OMP_50_ENABLED 2562 if (taskgroup->reduce_data != NULL) { // need to reduce? 2563 int cnt; 2564 void *reduce_data; 2565 kmp_team_t *t = thread->th.th_team; 2566 kmp_taskred_data_t *arr = (kmp_taskred_data_t *)taskgroup->reduce_data; 2567 // check if <priv> data of the first reduction variable shared for the team 2568 void *priv0 = arr[0].reduce_priv; 2569 if ((reduce_data = KMP_ATOMIC_LD_ACQ(&t->t.t_tg_reduce_data[0])) != NULL && 2570 ((kmp_taskred_data_t *)reduce_data)[0].reduce_priv == priv0) { 2571 // finishing task reduction on parallel 2572 cnt = KMP_ATOMIC_INC(&t->t.t_tg_fini_counter[0]); 2573 if (cnt == thread->th.th_team_nproc - 1) { 2574 // we are the last thread passing __kmpc_reduction_modifier_fini() 2575 // finalize task reduction: 2576 __kmp_task_reduction_fini(thread, taskgroup); 2577 // cleanup fields in the team structure: 2578 // TODO: is relaxed store enough here (whole barrier should follow)? 2579 __kmp_thread_free(thread, reduce_data); 2580 KMP_ATOMIC_ST_REL(&t->t.t_tg_reduce_data[0], NULL); 2581 KMP_ATOMIC_ST_REL(&t->t.t_tg_fini_counter[0], 0); 2582 } else { 2583 // we are not the last thread passing __kmpc_reduction_modifier_fini(), 2584 // so do not finalize reduction, just clean own copy of the data 2585 __kmp_task_reduction_clean(thread, taskgroup); 2586 } 2587 } else if ((reduce_data = KMP_ATOMIC_LD_ACQ(&t->t.t_tg_reduce_data[1])) != 2588 NULL && 2589 ((kmp_taskred_data_t *)reduce_data)[0].reduce_priv == priv0) { 2590 // finishing task reduction on worksharing 2591 cnt = KMP_ATOMIC_INC(&t->t.t_tg_fini_counter[1]); 2592 if (cnt == thread->th.th_team_nproc - 1) { 2593 // we are the last thread passing __kmpc_reduction_modifier_fini() 2594 __kmp_task_reduction_fini(thread, taskgroup); 2595 // cleanup fields in team structure: 2596 // TODO: is relaxed store enough here (whole barrier should follow)? 2597 __kmp_thread_free(thread, reduce_data); 2598 KMP_ATOMIC_ST_REL(&t->t.t_tg_reduce_data[1], NULL); 2599 KMP_ATOMIC_ST_REL(&t->t.t_tg_fini_counter[1], 0); 2600 } else { 2601 // we are not the last thread passing __kmpc_reduction_modifier_fini(), 2602 // so do not finalize reduction, just clean own copy of the data 2603 __kmp_task_reduction_clean(thread, taskgroup); 2604 } 2605 } else { 2606 // finishing task reduction on taskgroup 2607 __kmp_task_reduction_fini(thread, taskgroup); 2608 } 2609 } 2610 #endif 2611 // Restore parent taskgroup for the current task 2612 taskdata->td_taskgroup = taskgroup->parent; 2613 __kmp_thread_free(thread, taskgroup); 2614 2615 KA_TRACE(10, ("__kmpc_end_taskgroup(exit): T#%d task %p finished waiting\n", 2616 gtid, taskdata)); 2617 ANNOTATE_HAPPENS_AFTER(taskdata); 2618 2619 #if OMPT_SUPPORT && OMPT_OPTIONAL 2620 if (UNLIKELY(ompt_enabled.ompt_callback_sync_region)) { 2621 ompt_callbacks.ompt_callback(ompt_callback_sync_region)( 2622 ompt_sync_region_taskgroup, ompt_scope_end, &(my_parallel_data), 2623 &(my_task_data), codeptr); 2624 } 2625 #endif 2626 } 2627 #endif 2628 2629 // __kmp_remove_my_task: remove a task from my own deque 2630 static kmp_task_t *__kmp_remove_my_task(kmp_info_t *thread, kmp_int32 gtid, 2631 kmp_task_team_t *task_team, 2632 kmp_int32 is_constrained) { 2633 kmp_task_t *task; 2634 kmp_taskdata_t *taskdata; 2635 kmp_thread_data_t *thread_data; 2636 kmp_uint32 tail; 2637 2638 KMP_DEBUG_ASSERT(__kmp_tasking_mode != tskm_immediate_exec); 2639 KMP_DEBUG_ASSERT(task_team->tt.tt_threads_data != 2640 NULL); // Caller should check this condition 2641 2642 thread_data = &task_team->tt.tt_threads_data[__kmp_tid_from_gtid(gtid)]; 2643 2644 KA_TRACE(10, ("__kmp_remove_my_task(enter): T#%d ntasks=%d head=%u tail=%u\n", 2645 gtid, thread_data->td.td_deque_ntasks, 2646 thread_data->td.td_deque_head, thread_data->td.td_deque_tail)); 2647 2648 if (TCR_4(thread_data->td.td_deque_ntasks) == 0) { 2649 KA_TRACE(10, 2650 ("__kmp_remove_my_task(exit #1): T#%d No tasks to remove: " 2651 "ntasks=%d head=%u tail=%u\n", 2652 gtid, thread_data->td.td_deque_ntasks, 2653 thread_data->td.td_deque_head, thread_data->td.td_deque_tail)); 2654 return NULL; 2655 } 2656 2657 __kmp_acquire_bootstrap_lock(&thread_data->td.td_deque_lock); 2658 2659 if (TCR_4(thread_data->td.td_deque_ntasks) == 0) { 2660 __kmp_release_bootstrap_lock(&thread_data->td.td_deque_lock); 2661 KA_TRACE(10, 2662 ("__kmp_remove_my_task(exit #2): T#%d No tasks to remove: " 2663 "ntasks=%d head=%u tail=%u\n", 2664 gtid, thread_data->td.td_deque_ntasks, 2665 thread_data->td.td_deque_head, thread_data->td.td_deque_tail)); 2666 return NULL; 2667 } 2668 2669 tail = (thread_data->td.td_deque_tail - 1) & 2670 TASK_DEQUE_MASK(thread_data->td); // Wrap index. 2671 taskdata = thread_data->td.td_deque[tail]; 2672 2673 if (!__kmp_task_is_allowed(gtid, is_constrained, taskdata, 2674 thread->th.th_current_task)) { 2675 // The TSC does not allow to steal victim task 2676 __kmp_release_bootstrap_lock(&thread_data->td.td_deque_lock); 2677 KA_TRACE(10, 2678 ("__kmp_remove_my_task(exit #3): T#%d TSC blocks tail task: " 2679 "ntasks=%d head=%u tail=%u\n", 2680 gtid, thread_data->td.td_deque_ntasks, 2681 thread_data->td.td_deque_head, thread_data->td.td_deque_tail)); 2682 return NULL; 2683 } 2684 2685 thread_data->td.td_deque_tail = tail; 2686 TCW_4(thread_data->td.td_deque_ntasks, thread_data->td.td_deque_ntasks - 1); 2687 2688 __kmp_release_bootstrap_lock(&thread_data->td.td_deque_lock); 2689 2690 KA_TRACE(10, ("__kmp_remove_my_task(exit #4): T#%d task %p removed: " 2691 "ntasks=%d head=%u tail=%u\n", 2692 gtid, taskdata, thread_data->td.td_deque_ntasks, 2693 thread_data->td.td_deque_head, thread_data->td.td_deque_tail)); 2694 2695 task = KMP_TASKDATA_TO_TASK(taskdata); 2696 return task; 2697 } 2698 2699 // __kmp_steal_task: remove a task from another thread's deque 2700 // Assume that calling thread has already checked existence of 2701 // task_team thread_data before calling this routine. 2702 static kmp_task_t *__kmp_steal_task(kmp_info_t *victim_thr, kmp_int32 gtid, 2703 kmp_task_team_t *task_team, 2704 std::atomic<kmp_int32> *unfinished_threads, 2705 int *thread_finished, 2706 kmp_int32 is_constrained) { 2707 kmp_task_t *task; 2708 kmp_taskdata_t *taskdata; 2709 kmp_taskdata_t *current; 2710 kmp_thread_data_t *victim_td, *threads_data; 2711 kmp_int32 target; 2712 kmp_int32 victim_tid; 2713 2714 KMP_DEBUG_ASSERT(__kmp_tasking_mode != tskm_immediate_exec); 2715 2716 threads_data = task_team->tt.tt_threads_data; 2717 KMP_DEBUG_ASSERT(threads_data != NULL); // Caller should check this condition 2718 2719 victim_tid = victim_thr->th.th_info.ds.ds_tid; 2720 victim_td = &threads_data[victim_tid]; 2721 2722 KA_TRACE(10, ("__kmp_steal_task(enter): T#%d try to steal from T#%d: " 2723 "task_team=%p ntasks=%d head=%u tail=%u\n", 2724 gtid, __kmp_gtid_from_thread(victim_thr), task_team, 2725 victim_td->td.td_deque_ntasks, victim_td->td.td_deque_head, 2726 victim_td->td.td_deque_tail)); 2727 2728 if (TCR_4(victim_td->td.td_deque_ntasks) == 0) { 2729 KA_TRACE(10, ("__kmp_steal_task(exit #1): T#%d could not steal from T#%d: " 2730 "task_team=%p ntasks=%d head=%u tail=%u\n", 2731 gtid, __kmp_gtid_from_thread(victim_thr), task_team, 2732 victim_td->td.td_deque_ntasks, victim_td->td.td_deque_head, 2733 victim_td->td.td_deque_tail)); 2734 return NULL; 2735 } 2736 2737 __kmp_acquire_bootstrap_lock(&victim_td->td.td_deque_lock); 2738 2739 int ntasks = TCR_4(victim_td->td.td_deque_ntasks); 2740 // Check again after we acquire the lock 2741 if (ntasks == 0) { 2742 __kmp_release_bootstrap_lock(&victim_td->td.td_deque_lock); 2743 KA_TRACE(10, ("__kmp_steal_task(exit #2): T#%d could not steal from T#%d: " 2744 "task_team=%p ntasks=%d head=%u tail=%u\n", 2745 gtid, __kmp_gtid_from_thread(victim_thr), task_team, ntasks, 2746 victim_td->td.td_deque_head, victim_td->td.td_deque_tail)); 2747 return NULL; 2748 } 2749 2750 KMP_DEBUG_ASSERT(victim_td->td.td_deque != NULL); 2751 current = __kmp_threads[gtid]->th.th_current_task; 2752 taskdata = victim_td->td.td_deque[victim_td->td.td_deque_head]; 2753 if (__kmp_task_is_allowed(gtid, is_constrained, taskdata, current)) { 2754 // Bump head pointer and Wrap. 2755 victim_td->td.td_deque_head = 2756 (victim_td->td.td_deque_head + 1) & TASK_DEQUE_MASK(victim_td->td); 2757 } else { 2758 if (!task_team->tt.tt_untied_task_encountered) { 2759 // The TSC does not allow to steal victim task 2760 __kmp_release_bootstrap_lock(&victim_td->td.td_deque_lock); 2761 KA_TRACE(10, ("__kmp_steal_task(exit #3): T#%d could not steal from " 2762 "T#%d: task_team=%p ntasks=%d head=%u tail=%u\n", 2763 gtid, __kmp_gtid_from_thread(victim_thr), task_team, ntasks, 2764 victim_td->td.td_deque_head, victim_td->td.td_deque_tail)); 2765 return NULL; 2766 } 2767 int i; 2768 // walk through victim's deque trying to steal any task 2769 target = victim_td->td.td_deque_head; 2770 taskdata = NULL; 2771 for (i = 1; i < ntasks; ++i) { 2772 target = (target + 1) & TASK_DEQUE_MASK(victim_td->td); 2773 taskdata = victim_td->td.td_deque[target]; 2774 if (__kmp_task_is_allowed(gtid, is_constrained, taskdata, current)) { 2775 break; // found victim task 2776 } else { 2777 taskdata = NULL; 2778 } 2779 } 2780 if (taskdata == NULL) { 2781 // No appropriate candidate to steal found 2782 __kmp_release_bootstrap_lock(&victim_td->td.td_deque_lock); 2783 KA_TRACE(10, ("__kmp_steal_task(exit #4): T#%d could not steal from " 2784 "T#%d: task_team=%p ntasks=%d head=%u tail=%u\n", 2785 gtid, __kmp_gtid_from_thread(victim_thr), task_team, ntasks, 2786 victim_td->td.td_deque_head, victim_td->td.td_deque_tail)); 2787 return NULL; 2788 } 2789 int prev = target; 2790 for (i = i + 1; i < ntasks; ++i) { 2791 // shift remaining tasks in the deque left by 1 2792 target = (target + 1) & TASK_DEQUE_MASK(victim_td->td); 2793 victim_td->td.td_deque[prev] = victim_td->td.td_deque[target]; 2794 prev = target; 2795 } 2796 KMP_DEBUG_ASSERT( 2797 victim_td->td.td_deque_tail == 2798 (kmp_uint32)((target + 1) & TASK_DEQUE_MASK(victim_td->td))); 2799 victim_td->td.td_deque_tail = target; // tail -= 1 (wrapped)) 2800 } 2801 if (*thread_finished) { 2802 // We need to un-mark this victim as a finished victim. This must be done 2803 // before releasing the lock, or else other threads (starting with the 2804 // master victim) might be prematurely released from the barrier!!! 2805 kmp_int32 count; 2806 2807 count = KMP_ATOMIC_INC(unfinished_threads); 2808 2809 KA_TRACE( 2810 20, 2811 ("__kmp_steal_task: T#%d inc unfinished_threads to %d: task_team=%p\n", 2812 gtid, count + 1, task_team)); 2813 2814 *thread_finished = FALSE; 2815 } 2816 TCW_4(victim_td->td.td_deque_ntasks, ntasks - 1); 2817 2818 __kmp_release_bootstrap_lock(&victim_td->td.td_deque_lock); 2819 2820 KMP_COUNT_BLOCK(TASK_stolen); 2821 KA_TRACE(10, 2822 ("__kmp_steal_task(exit #5): T#%d stole task %p from T#%d: " 2823 "task_team=%p ntasks=%d head=%u tail=%u\n", 2824 gtid, taskdata, __kmp_gtid_from_thread(victim_thr), task_team, 2825 ntasks, victim_td->td.td_deque_head, victim_td->td.td_deque_tail)); 2826 2827 task = KMP_TASKDATA_TO_TASK(taskdata); 2828 return task; 2829 } 2830 2831 // __kmp_execute_tasks_template: Choose and execute tasks until either the 2832 // condition is statisfied (return true) or there are none left (return false). 2833 // 2834 // final_spin is TRUE if this is the spin at the release barrier. 2835 // thread_finished indicates whether the thread is finished executing all 2836 // the tasks it has on its deque, and is at the release barrier. 2837 // spinner is the location on which to spin. 2838 // spinner == NULL means only execute a single task and return. 2839 // checker is the value to check to terminate the spin. 2840 template <class C> 2841 static inline int __kmp_execute_tasks_template( 2842 kmp_info_t *thread, kmp_int32 gtid, C *flag, int final_spin, 2843 int *thread_finished USE_ITT_BUILD_ARG(void *itt_sync_obj), 2844 kmp_int32 is_constrained) { 2845 kmp_task_team_t *task_team = thread->th.th_task_team; 2846 kmp_thread_data_t *threads_data; 2847 kmp_task_t *task; 2848 kmp_info_t *other_thread; 2849 kmp_taskdata_t *current_task = thread->th.th_current_task; 2850 std::atomic<kmp_int32> *unfinished_threads; 2851 kmp_int32 nthreads, victim_tid = -2, use_own_tasks = 1, new_victim = 0, 2852 tid = thread->th.th_info.ds.ds_tid; 2853 2854 KMP_DEBUG_ASSERT(__kmp_tasking_mode != tskm_immediate_exec); 2855 KMP_DEBUG_ASSERT(thread == __kmp_threads[gtid]); 2856 2857 if (task_team == NULL || current_task == NULL) 2858 return FALSE; 2859 2860 KA_TRACE(15, ("__kmp_execute_tasks_template(enter): T#%d final_spin=%d " 2861 "*thread_finished=%d\n", 2862 gtid, final_spin, *thread_finished)); 2863 2864 thread->th.th_reap_state = KMP_NOT_SAFE_TO_REAP; 2865 threads_data = (kmp_thread_data_t *)TCR_PTR(task_team->tt.tt_threads_data); 2866 KMP_DEBUG_ASSERT(threads_data != NULL); 2867 2868 nthreads = task_team->tt.tt_nproc; 2869 unfinished_threads = &(task_team->tt.tt_unfinished_threads); 2870 #if OMP_45_ENABLED 2871 KMP_DEBUG_ASSERT(nthreads > 1 || task_team->tt.tt_found_proxy_tasks); 2872 #else 2873 KMP_DEBUG_ASSERT(nthreads > 1); 2874 #endif 2875 KMP_DEBUG_ASSERT(*unfinished_threads >= 0); 2876 2877 while (1) { // Outer loop keeps trying to find tasks in case of single thread 2878 // getting tasks from target constructs 2879 while (1) { // Inner loop to find a task and execute it 2880 task = NULL; 2881 if (use_own_tasks) { // check on own queue first 2882 task = __kmp_remove_my_task(thread, gtid, task_team, is_constrained); 2883 } 2884 if ((task == NULL) && (nthreads > 1)) { // Steal a task 2885 int asleep = 1; 2886 use_own_tasks = 0; 2887 // Try to steal from the last place I stole from successfully. 2888 if (victim_tid == -2) { // haven't stolen anything yet 2889 victim_tid = threads_data[tid].td.td_deque_last_stolen; 2890 if (victim_tid != 2891 -1) // if we have a last stolen from victim, get the thread 2892 other_thread = threads_data[victim_tid].td.td_thr; 2893 } 2894 if (victim_tid != -1) { // found last victim 2895 asleep = 0; 2896 } else if (!new_victim) { // no recent steals and we haven't already 2897 // used a new victim; select a random thread 2898 do { // Find a different thread to steal work from. 2899 // Pick a random thread. Initial plan was to cycle through all the 2900 // threads, and only return if we tried to steal from every thread, 2901 // and failed. Arch says that's not such a great idea. 2902 victim_tid = __kmp_get_random(thread) % (nthreads - 1); 2903 if (victim_tid >= tid) { 2904 ++victim_tid; // Adjusts random distribution to exclude self 2905 } 2906 // Found a potential victim 2907 other_thread = threads_data[victim_tid].td.td_thr; 2908 // There is a slight chance that __kmp_enable_tasking() did not wake 2909 // up all threads waiting at the barrier. If victim is sleeping, 2910 // then wake it up. Since we were going to pay the cache miss 2911 // penalty for referencing another thread's kmp_info_t struct 2912 // anyway, 2913 // the check shouldn't cost too much performance at this point. In 2914 // extra barrier mode, tasks do not sleep at the separate tasking 2915 // barrier, so this isn't a problem. 2916 asleep = 0; 2917 if ((__kmp_tasking_mode == tskm_task_teams) && 2918 (__kmp_dflt_blocktime != KMP_MAX_BLOCKTIME) && 2919 (TCR_PTR(CCAST(void *, other_thread->th.th_sleep_loc)) != 2920 NULL)) { 2921 asleep = 1; 2922 __kmp_null_resume_wrapper(__kmp_gtid_from_thread(other_thread), 2923 other_thread->th.th_sleep_loc); 2924 // A sleeping thread should not have any tasks on it's queue. 2925 // There is a slight possibility that it resumes, steals a task 2926 // from another thread, which spawns more tasks, all in the time 2927 // that it takes this thread to check => don't write an assertion 2928 // that the victim's queue is empty. Try stealing from a 2929 // different thread. 2930 } 2931 } while (asleep); 2932 } 2933 2934 if (!asleep) { 2935 // We have a victim to try to steal from 2936 task = __kmp_steal_task(other_thread, gtid, task_team, 2937 unfinished_threads, thread_finished, 2938 is_constrained); 2939 } 2940 if (task != NULL) { // set last stolen to victim 2941 if (threads_data[tid].td.td_deque_last_stolen != victim_tid) { 2942 threads_data[tid].td.td_deque_last_stolen = victim_tid; 2943 // The pre-refactored code did not try more than 1 successful new 2944 // vicitm, unless the last one generated more local tasks; 2945 // new_victim keeps track of this 2946 new_victim = 1; 2947 } 2948 } else { // No tasks found; unset last_stolen 2949 KMP_CHECK_UPDATE(threads_data[tid].td.td_deque_last_stolen, -1); 2950 victim_tid = -2; // no successful victim found 2951 } 2952 } 2953 2954 if (task == NULL) // break out of tasking loop 2955 break; 2956 2957 // Found a task; execute it 2958 #if USE_ITT_BUILD && USE_ITT_NOTIFY 2959 if (__itt_sync_create_ptr || KMP_ITT_DEBUG) { 2960 if (itt_sync_obj == NULL) { // we are at fork barrier where we could not 2961 // get the object reliably 2962 itt_sync_obj = __kmp_itt_barrier_object(gtid, bs_forkjoin_barrier); 2963 } 2964 __kmp_itt_task_starting(itt_sync_obj); 2965 } 2966 #endif /* USE_ITT_BUILD && USE_ITT_NOTIFY */ 2967 __kmp_invoke_task(gtid, task, current_task); 2968 #if USE_ITT_BUILD 2969 if (itt_sync_obj != NULL) 2970 __kmp_itt_task_finished(itt_sync_obj); 2971 #endif /* USE_ITT_BUILD */ 2972 // If this thread is only partway through the barrier and the condition is 2973 // met, then return now, so that the barrier gather/release pattern can 2974 // proceed. If this thread is in the last spin loop in the barrier, 2975 // waiting to be released, we know that the termination condition will not 2976 // be satisified, so don't waste any cycles checking it. 2977 if (flag == NULL || (!final_spin && flag->done_check())) { 2978 KA_TRACE( 2979 15, 2980 ("__kmp_execute_tasks_template: T#%d spin condition satisfied\n", 2981 gtid)); 2982 return TRUE; 2983 } 2984 if (thread->th.th_task_team == NULL) { 2985 break; 2986 } 2987 KMP_YIELD(__kmp_library == library_throughput); // Yield before next task 2988 // If execution of a stolen task results in more tasks being placed on our 2989 // run queue, reset use_own_tasks 2990 if (!use_own_tasks && TCR_4(threads_data[tid].td.td_deque_ntasks) != 0) { 2991 KA_TRACE(20, ("__kmp_execute_tasks_template: T#%d stolen task spawned " 2992 "other tasks, restart\n", 2993 gtid)); 2994 use_own_tasks = 1; 2995 new_victim = 0; 2996 } 2997 } 2998 2999 // The task source has been exhausted. If in final spin loop of barrier, check 3000 // if termination condition is satisfied. 3001 #if OMP_45_ENABLED 3002 // The work queue may be empty but there might be proxy tasks still 3003 // executing 3004 if (final_spin && 3005 KMP_ATOMIC_LD_ACQ(¤t_task->td_incomplete_child_tasks) == 0) 3006 #else 3007 if (final_spin) 3008 #endif 3009 { 3010 // First, decrement the #unfinished threads, if that has not already been 3011 // done. This decrement might be to the spin location, and result in the 3012 // termination condition being satisfied. 3013 if (!*thread_finished) { 3014 kmp_int32 count; 3015 3016 count = KMP_ATOMIC_DEC(unfinished_threads) - 1; 3017 KA_TRACE(20, ("__kmp_execute_tasks_template: T#%d dec " 3018 "unfinished_threads to %d task_team=%p\n", 3019 gtid, count, task_team)); 3020 *thread_finished = TRUE; 3021 } 3022 3023 // It is now unsafe to reference thread->th.th_team !!! 3024 // Decrementing task_team->tt.tt_unfinished_threads can allow the master 3025 // thread to pass through the barrier, where it might reset each thread's 3026 // th.th_team field for the next parallel region. If we can steal more 3027 // work, we know that this has not happened yet. 3028 if (flag != NULL && flag->done_check()) { 3029 KA_TRACE( 3030 15, 3031 ("__kmp_execute_tasks_template: T#%d spin condition satisfied\n", 3032 gtid)); 3033 return TRUE; 3034 } 3035 } 3036 3037 // If this thread's task team is NULL, master has recognized that there are 3038 // no more tasks; bail out 3039 if (thread->th.th_task_team == NULL) { 3040 KA_TRACE(15, 3041 ("__kmp_execute_tasks_template: T#%d no more tasks\n", gtid)); 3042 return FALSE; 3043 } 3044 3045 #if OMP_45_ENABLED 3046 // We could be getting tasks from target constructs; if this is the only 3047 // thread, keep trying to execute tasks from own queue 3048 if (nthreads == 1) 3049 use_own_tasks = 1; 3050 else 3051 #endif 3052 { 3053 KA_TRACE(15, 3054 ("__kmp_execute_tasks_template: T#%d can't find work\n", gtid)); 3055 return FALSE; 3056 } 3057 } 3058 } 3059 3060 int __kmp_execute_tasks_32( 3061 kmp_info_t *thread, kmp_int32 gtid, kmp_flag_32 *flag, int final_spin, 3062 int *thread_finished USE_ITT_BUILD_ARG(void *itt_sync_obj), 3063 kmp_int32 is_constrained) { 3064 return __kmp_execute_tasks_template( 3065 thread, gtid, flag, final_spin, 3066 thread_finished USE_ITT_BUILD_ARG(itt_sync_obj), is_constrained); 3067 } 3068 3069 int __kmp_execute_tasks_64( 3070 kmp_info_t *thread, kmp_int32 gtid, kmp_flag_64 *flag, int final_spin, 3071 int *thread_finished USE_ITT_BUILD_ARG(void *itt_sync_obj), 3072 kmp_int32 is_constrained) { 3073 return __kmp_execute_tasks_template( 3074 thread, gtid, flag, final_spin, 3075 thread_finished USE_ITT_BUILD_ARG(itt_sync_obj), is_constrained); 3076 } 3077 3078 int __kmp_execute_tasks_oncore( 3079 kmp_info_t *thread, kmp_int32 gtid, kmp_flag_oncore *flag, int final_spin, 3080 int *thread_finished USE_ITT_BUILD_ARG(void *itt_sync_obj), 3081 kmp_int32 is_constrained) { 3082 return __kmp_execute_tasks_template( 3083 thread, gtid, flag, final_spin, 3084 thread_finished USE_ITT_BUILD_ARG(itt_sync_obj), is_constrained); 3085 } 3086 3087 // __kmp_enable_tasking: Allocate task team and resume threads sleeping at the 3088 // next barrier so they can assist in executing enqueued tasks. 3089 // First thread in allocates the task team atomically. 3090 static void __kmp_enable_tasking(kmp_task_team_t *task_team, 3091 kmp_info_t *this_thr) { 3092 kmp_thread_data_t *threads_data; 3093 int nthreads, i, is_init_thread; 3094 3095 KA_TRACE(10, ("__kmp_enable_tasking(enter): T#%d\n", 3096 __kmp_gtid_from_thread(this_thr))); 3097 3098 KMP_DEBUG_ASSERT(task_team != NULL); 3099 KMP_DEBUG_ASSERT(this_thr->th.th_team != NULL); 3100 3101 nthreads = task_team->tt.tt_nproc; 3102 KMP_DEBUG_ASSERT(nthreads > 0); 3103 KMP_DEBUG_ASSERT(nthreads == this_thr->th.th_team->t.t_nproc); 3104 3105 // Allocate or increase the size of threads_data if necessary 3106 is_init_thread = __kmp_realloc_task_threads_data(this_thr, task_team); 3107 3108 if (!is_init_thread) { 3109 // Some other thread already set up the array. 3110 KA_TRACE( 3111 20, 3112 ("__kmp_enable_tasking(exit): T#%d: threads array already set up.\n", 3113 __kmp_gtid_from_thread(this_thr))); 3114 return; 3115 } 3116 threads_data = (kmp_thread_data_t *)TCR_PTR(task_team->tt.tt_threads_data); 3117 KMP_DEBUG_ASSERT(threads_data != NULL); 3118 3119 if (__kmp_tasking_mode == tskm_task_teams && 3120 (__kmp_dflt_blocktime != KMP_MAX_BLOCKTIME)) { 3121 // Release any threads sleeping at the barrier, so that they can steal 3122 // tasks and execute them. In extra barrier mode, tasks do not sleep 3123 // at the separate tasking barrier, so this isn't a problem. 3124 for (i = 0; i < nthreads; i++) { 3125 volatile void *sleep_loc; 3126 kmp_info_t *thread = threads_data[i].td.td_thr; 3127 3128 if (i == this_thr->th.th_info.ds.ds_tid) { 3129 continue; 3130 } 3131 // Since we haven't locked the thread's suspend mutex lock at this 3132 // point, there is a small window where a thread might be putting 3133 // itself to sleep, but hasn't set the th_sleep_loc field yet. 3134 // To work around this, __kmp_execute_tasks_template() periodically checks 3135 // see if other threads are sleeping (using the same random mechanism that 3136 // is used for task stealing) and awakens them if they are. 3137 if ((sleep_loc = TCR_PTR(CCAST(void *, thread->th.th_sleep_loc))) != 3138 NULL) { 3139 KF_TRACE(50, ("__kmp_enable_tasking: T#%d waking up thread T#%d\n", 3140 __kmp_gtid_from_thread(this_thr), 3141 __kmp_gtid_from_thread(thread))); 3142 __kmp_null_resume_wrapper(__kmp_gtid_from_thread(thread), sleep_loc); 3143 } else { 3144 KF_TRACE(50, ("__kmp_enable_tasking: T#%d don't wake up thread T#%d\n", 3145 __kmp_gtid_from_thread(this_thr), 3146 __kmp_gtid_from_thread(thread))); 3147 } 3148 } 3149 } 3150 3151 KA_TRACE(10, ("__kmp_enable_tasking(exit): T#%d\n", 3152 __kmp_gtid_from_thread(this_thr))); 3153 } 3154 3155 /* // TODO: Check the comment consistency 3156 * Utility routines for "task teams". A task team (kmp_task_t) is kind of 3157 * like a shadow of the kmp_team_t data struct, with a different lifetime. 3158 * After a child * thread checks into a barrier and calls __kmp_release() from 3159 * the particular variant of __kmp_<barrier_kind>_barrier_gather(), it can no 3160 * longer assume that the kmp_team_t structure is intact (at any moment, the 3161 * master thread may exit the barrier code and free the team data structure, 3162 * and return the threads to the thread pool). 3163 * 3164 * This does not work with the the tasking code, as the thread is still 3165 * expected to participate in the execution of any tasks that may have been 3166 * spawned my a member of the team, and the thread still needs access to all 3167 * to each thread in the team, so that it can steal work from it. 3168 * 3169 * Enter the existence of the kmp_task_team_t struct. It employs a reference 3170 * counting mechanims, and is allocated by the master thread before calling 3171 * __kmp_<barrier_kind>_release, and then is release by the last thread to 3172 * exit __kmp_<barrier_kind>_release at the next barrier. I.e. the lifetimes 3173 * of the kmp_task_team_t structs for consecutive barriers can overlap 3174 * (and will, unless the master thread is the last thread to exit the barrier 3175 * release phase, which is not typical). 3176 * 3177 * The existence of such a struct is useful outside the context of tasking, 3178 * but for now, I'm trying to keep it specific to the OMP_30_ENABLED macro, 3179 * so that any performance differences show up when comparing the 2.5 vs. 3.0 3180 * libraries. 3181 * 3182 * We currently use the existence of the threads array as an indicator that 3183 * tasks were spawned since the last barrier. If the structure is to be 3184 * useful outside the context of tasking, then this will have to change, but 3185 * not settting the field minimizes the performance impact of tasking on 3186 * barriers, when no explicit tasks were spawned (pushed, actually). 3187 */ 3188 3189 static kmp_task_team_t *__kmp_free_task_teams = 3190 NULL; // Free list for task_team data structures 3191 // Lock for task team data structures 3192 kmp_bootstrap_lock_t __kmp_task_team_lock = 3193 KMP_BOOTSTRAP_LOCK_INITIALIZER(__kmp_task_team_lock); 3194 3195 // __kmp_alloc_task_deque: 3196 // Allocates a task deque for a particular thread, and initialize the necessary 3197 // data structures relating to the deque. This only happens once per thread 3198 // per task team since task teams are recycled. No lock is needed during 3199 // allocation since each thread allocates its own deque. 3200 static void __kmp_alloc_task_deque(kmp_info_t *thread, 3201 kmp_thread_data_t *thread_data) { 3202 __kmp_init_bootstrap_lock(&thread_data->td.td_deque_lock); 3203 KMP_DEBUG_ASSERT(thread_data->td.td_deque == NULL); 3204 3205 // Initialize last stolen task field to "none" 3206 thread_data->td.td_deque_last_stolen = -1; 3207 3208 KMP_DEBUG_ASSERT(TCR_4(thread_data->td.td_deque_ntasks) == 0); 3209 KMP_DEBUG_ASSERT(thread_data->td.td_deque_head == 0); 3210 KMP_DEBUG_ASSERT(thread_data->td.td_deque_tail == 0); 3211 3212 KE_TRACE( 3213 10, 3214 ("__kmp_alloc_task_deque: T#%d allocating deque[%d] for thread_data %p\n", 3215 __kmp_gtid_from_thread(thread), INITIAL_TASK_DEQUE_SIZE, thread_data)); 3216 // Allocate space for task deque, and zero the deque 3217 // Cannot use __kmp_thread_calloc() because threads not around for 3218 // kmp_reap_task_team( ). 3219 thread_data->td.td_deque = (kmp_taskdata_t **)__kmp_allocate( 3220 INITIAL_TASK_DEQUE_SIZE * sizeof(kmp_taskdata_t *)); 3221 thread_data->td.td_deque_size = INITIAL_TASK_DEQUE_SIZE; 3222 } 3223 3224 // __kmp_free_task_deque: 3225 // Deallocates a task deque for a particular thread. Happens at library 3226 // deallocation so don't need to reset all thread data fields. 3227 static void __kmp_free_task_deque(kmp_thread_data_t *thread_data) { 3228 if (thread_data->td.td_deque != NULL) { 3229 __kmp_acquire_bootstrap_lock(&thread_data->td.td_deque_lock); 3230 TCW_4(thread_data->td.td_deque_ntasks, 0); 3231 __kmp_free(thread_data->td.td_deque); 3232 thread_data->td.td_deque = NULL; 3233 __kmp_release_bootstrap_lock(&thread_data->td.td_deque_lock); 3234 } 3235 3236 #ifdef BUILD_TIED_TASK_STACK 3237 // GEH: Figure out what to do here for td_susp_tied_tasks 3238 if (thread_data->td.td_susp_tied_tasks.ts_entries != TASK_STACK_EMPTY) { 3239 __kmp_free_task_stack(__kmp_thread_from_gtid(gtid), thread_data); 3240 } 3241 #endif // BUILD_TIED_TASK_STACK 3242 } 3243 3244 // __kmp_realloc_task_threads_data: 3245 // Allocates a threads_data array for a task team, either by allocating an 3246 // initial array or enlarging an existing array. Only the first thread to get 3247 // the lock allocs or enlarges the array and re-initializes the array eleemnts. 3248 // That thread returns "TRUE", the rest return "FALSE". 3249 // Assumes that the new array size is given by task_team -> tt.tt_nproc. 3250 // The current size is given by task_team -> tt.tt_max_threads. 3251 static int __kmp_realloc_task_threads_data(kmp_info_t *thread, 3252 kmp_task_team_t *task_team) { 3253 kmp_thread_data_t **threads_data_p; 3254 kmp_int32 nthreads, maxthreads; 3255 int is_init_thread = FALSE; 3256 3257 if (TCR_4(task_team->tt.tt_found_tasks)) { 3258 // Already reallocated and initialized. 3259 return FALSE; 3260 } 3261 3262 threads_data_p = &task_team->tt.tt_threads_data; 3263 nthreads = task_team->tt.tt_nproc; 3264 maxthreads = task_team->tt.tt_max_threads; 3265 3266 // All threads must lock when they encounter the first task of the implicit 3267 // task region to make sure threads_data fields are (re)initialized before 3268 // used. 3269 __kmp_acquire_bootstrap_lock(&task_team->tt.tt_threads_lock); 3270 3271 if (!TCR_4(task_team->tt.tt_found_tasks)) { 3272 // first thread to enable tasking 3273 kmp_team_t *team = thread->th.th_team; 3274 int i; 3275 3276 is_init_thread = TRUE; 3277 if (maxthreads < nthreads) { 3278 3279 if (*threads_data_p != NULL) { 3280 kmp_thread_data_t *old_data = *threads_data_p; 3281 kmp_thread_data_t *new_data = NULL; 3282 3283 KE_TRACE( 3284 10, 3285 ("__kmp_realloc_task_threads_data: T#%d reallocating " 3286 "threads data for task_team %p, new_size = %d, old_size = %d\n", 3287 __kmp_gtid_from_thread(thread), task_team, nthreads, maxthreads)); 3288 // Reallocate threads_data to have more elements than current array 3289 // Cannot use __kmp_thread_realloc() because threads not around for 3290 // kmp_reap_task_team( ). Note all new array entries are initialized 3291 // to zero by __kmp_allocate(). 3292 new_data = (kmp_thread_data_t *)__kmp_allocate( 3293 nthreads * sizeof(kmp_thread_data_t)); 3294 // copy old data to new data 3295 KMP_MEMCPY_S((void *)new_data, nthreads * sizeof(kmp_thread_data_t), 3296 (void *)old_data, maxthreads * sizeof(kmp_thread_data_t)); 3297 3298 #ifdef BUILD_TIED_TASK_STACK 3299 // GEH: Figure out if this is the right thing to do 3300 for (i = maxthreads; i < nthreads; i++) { 3301 kmp_thread_data_t *thread_data = &(*threads_data_p)[i]; 3302 __kmp_init_task_stack(__kmp_gtid_from_thread(thread), thread_data); 3303 } 3304 #endif // BUILD_TIED_TASK_STACK 3305 // Install the new data and free the old data 3306 (*threads_data_p) = new_data; 3307 __kmp_free(old_data); 3308 } else { 3309 KE_TRACE(10, ("__kmp_realloc_task_threads_data: T#%d allocating " 3310 "threads data for task_team %p, size = %d\n", 3311 __kmp_gtid_from_thread(thread), task_team, nthreads)); 3312 // Make the initial allocate for threads_data array, and zero entries 3313 // Cannot use __kmp_thread_calloc() because threads not around for 3314 // kmp_reap_task_team( ). 3315 ANNOTATE_IGNORE_WRITES_BEGIN(); 3316 *threads_data_p = (kmp_thread_data_t *)__kmp_allocate( 3317 nthreads * sizeof(kmp_thread_data_t)); 3318 ANNOTATE_IGNORE_WRITES_END(); 3319 #ifdef BUILD_TIED_TASK_STACK 3320 // GEH: Figure out if this is the right thing to do 3321 for (i = 0; i < nthreads; i++) { 3322 kmp_thread_data_t *thread_data = &(*threads_data_p)[i]; 3323 __kmp_init_task_stack(__kmp_gtid_from_thread(thread), thread_data); 3324 } 3325 #endif // BUILD_TIED_TASK_STACK 3326 } 3327 task_team->tt.tt_max_threads = nthreads; 3328 } else { 3329 // If array has (more than) enough elements, go ahead and use it 3330 KMP_DEBUG_ASSERT(*threads_data_p != NULL); 3331 } 3332 3333 // initialize threads_data pointers back to thread_info structures 3334 for (i = 0; i < nthreads; i++) { 3335 kmp_thread_data_t *thread_data = &(*threads_data_p)[i]; 3336 thread_data->td.td_thr = team->t.t_threads[i]; 3337 3338 if (thread_data->td.td_deque_last_stolen >= nthreads) { 3339 // The last stolen field survives across teams / barrier, and the number 3340 // of threads may have changed. It's possible (likely?) that a new 3341 // parallel region will exhibit the same behavior as previous region. 3342 thread_data->td.td_deque_last_stolen = -1; 3343 } 3344 } 3345 3346 KMP_MB(); 3347 TCW_SYNC_4(task_team->tt.tt_found_tasks, TRUE); 3348 } 3349 3350 __kmp_release_bootstrap_lock(&task_team->tt.tt_threads_lock); 3351 return is_init_thread; 3352 } 3353 3354 // __kmp_free_task_threads_data: 3355 // Deallocates a threads_data array for a task team, including any attached 3356 // tasking deques. Only occurs at library shutdown. 3357 static void __kmp_free_task_threads_data(kmp_task_team_t *task_team) { 3358 __kmp_acquire_bootstrap_lock(&task_team->tt.tt_threads_lock); 3359 if (task_team->tt.tt_threads_data != NULL) { 3360 int i; 3361 for (i = 0; i < task_team->tt.tt_max_threads; i++) { 3362 __kmp_free_task_deque(&task_team->tt.tt_threads_data[i]); 3363 } 3364 __kmp_free(task_team->tt.tt_threads_data); 3365 task_team->tt.tt_threads_data = NULL; 3366 } 3367 __kmp_release_bootstrap_lock(&task_team->tt.tt_threads_lock); 3368 } 3369 3370 // __kmp_allocate_task_team: 3371 // Allocates a task team associated with a specific team, taking it from 3372 // the global task team free list if possible. Also initializes data 3373 // structures. 3374 static kmp_task_team_t *__kmp_allocate_task_team(kmp_info_t *thread, 3375 kmp_team_t *team) { 3376 kmp_task_team_t *task_team = NULL; 3377 int nthreads; 3378 3379 KA_TRACE(20, ("__kmp_allocate_task_team: T#%d entering; team = %p\n", 3380 (thread ? __kmp_gtid_from_thread(thread) : -1), team)); 3381 3382 if (TCR_PTR(__kmp_free_task_teams) != NULL) { 3383 // Take a task team from the task team pool 3384 __kmp_acquire_bootstrap_lock(&__kmp_task_team_lock); 3385 if (__kmp_free_task_teams != NULL) { 3386 task_team = __kmp_free_task_teams; 3387 TCW_PTR(__kmp_free_task_teams, task_team->tt.tt_next); 3388 task_team->tt.tt_next = NULL; 3389 } 3390 __kmp_release_bootstrap_lock(&__kmp_task_team_lock); 3391 } 3392 3393 if (task_team == NULL) { 3394 KE_TRACE(10, ("__kmp_allocate_task_team: T#%d allocating " 3395 "task team for team %p\n", 3396 __kmp_gtid_from_thread(thread), team)); 3397 // Allocate a new task team if one is not available. 3398 // Cannot use __kmp_thread_malloc() because threads not around for 3399 // kmp_reap_task_team( ). 3400 task_team = (kmp_task_team_t *)__kmp_allocate(sizeof(kmp_task_team_t)); 3401 __kmp_init_bootstrap_lock(&task_team->tt.tt_threads_lock); 3402 // AC: __kmp_allocate zeroes returned memory 3403 // task_team -> tt.tt_threads_data = NULL; 3404 // task_team -> tt.tt_max_threads = 0; 3405 // task_team -> tt.tt_next = NULL; 3406 } 3407 3408 TCW_4(task_team->tt.tt_found_tasks, FALSE); 3409 #if OMP_45_ENABLED 3410 TCW_4(task_team->tt.tt_found_proxy_tasks, FALSE); 3411 #endif 3412 task_team->tt.tt_nproc = nthreads = team->t.t_nproc; 3413 3414 KMP_ATOMIC_ST_REL(&task_team->tt.tt_unfinished_threads, nthreads); 3415 TCW_4(task_team->tt.tt_active, TRUE); 3416 3417 KA_TRACE(20, ("__kmp_allocate_task_team: T#%d exiting; task_team = %p " 3418 "unfinished_threads init'd to %d\n", 3419 (thread ? __kmp_gtid_from_thread(thread) : -1), task_team, 3420 KMP_ATOMIC_LD_RLX(&task_team->tt.tt_unfinished_threads))); 3421 return task_team; 3422 } 3423 3424 // __kmp_free_task_team: 3425 // Frees the task team associated with a specific thread, and adds it 3426 // to the global task team free list. 3427 void __kmp_free_task_team(kmp_info_t *thread, kmp_task_team_t *task_team) { 3428 KA_TRACE(20, ("__kmp_free_task_team: T#%d task_team = %p\n", 3429 thread ? __kmp_gtid_from_thread(thread) : -1, task_team)); 3430 3431 // Put task team back on free list 3432 __kmp_acquire_bootstrap_lock(&__kmp_task_team_lock); 3433 3434 KMP_DEBUG_ASSERT(task_team->tt.tt_next == NULL); 3435 task_team->tt.tt_next = __kmp_free_task_teams; 3436 TCW_PTR(__kmp_free_task_teams, task_team); 3437 3438 __kmp_release_bootstrap_lock(&__kmp_task_team_lock); 3439 } 3440 3441 // __kmp_reap_task_teams: 3442 // Free all the task teams on the task team free list. 3443 // Should only be done during library shutdown. 3444 // Cannot do anything that needs a thread structure or gtid since they are 3445 // already gone. 3446 void __kmp_reap_task_teams(void) { 3447 kmp_task_team_t *task_team; 3448 3449 if (TCR_PTR(__kmp_free_task_teams) != NULL) { 3450 // Free all task_teams on the free list 3451 __kmp_acquire_bootstrap_lock(&__kmp_task_team_lock); 3452 while ((task_team = __kmp_free_task_teams) != NULL) { 3453 __kmp_free_task_teams = task_team->tt.tt_next; 3454 task_team->tt.tt_next = NULL; 3455 3456 // Free threads_data if necessary 3457 if (task_team->tt.tt_threads_data != NULL) { 3458 __kmp_free_task_threads_data(task_team); 3459 } 3460 __kmp_free(task_team); 3461 } 3462 __kmp_release_bootstrap_lock(&__kmp_task_team_lock); 3463 } 3464 } 3465 3466 // __kmp_wait_to_unref_task_teams: 3467 // Some threads could still be in the fork barrier release code, possibly 3468 // trying to steal tasks. Wait for each thread to unreference its task team. 3469 void __kmp_wait_to_unref_task_teams(void) { 3470 kmp_info_t *thread; 3471 kmp_uint32 spins; 3472 int done; 3473 3474 KMP_INIT_YIELD(spins); 3475 3476 for (;;) { 3477 done = TRUE; 3478 3479 // TODO: GEH - this may be is wrong because some sync would be necessary 3480 // in case threads are added to the pool during the traversal. Need to 3481 // verify that lock for thread pool is held when calling this routine. 3482 for (thread = CCAST(kmp_info_t *, __kmp_thread_pool); thread != NULL; 3483 thread = thread->th.th_next_pool) { 3484 #if KMP_OS_WINDOWS 3485 DWORD exit_val; 3486 #endif 3487 if (TCR_PTR(thread->th.th_task_team) == NULL) { 3488 KA_TRACE(10, ("__kmp_wait_to_unref_task_team: T#%d task_team == NULL\n", 3489 __kmp_gtid_from_thread(thread))); 3490 continue; 3491 } 3492 #if KMP_OS_WINDOWS 3493 // TODO: GEH - add this check for Linux* OS / OS X* as well? 3494 if (!__kmp_is_thread_alive(thread, &exit_val)) { 3495 thread->th.th_task_team = NULL; 3496 continue; 3497 } 3498 #endif 3499 3500 done = FALSE; // Because th_task_team pointer is not NULL for this thread 3501 3502 KA_TRACE(10, ("__kmp_wait_to_unref_task_team: Waiting for T#%d to " 3503 "unreference task_team\n", 3504 __kmp_gtid_from_thread(thread))); 3505 3506 if (__kmp_dflt_blocktime != KMP_MAX_BLOCKTIME) { 3507 volatile void *sleep_loc; 3508 // If the thread is sleeping, awaken it. 3509 if ((sleep_loc = TCR_PTR(CCAST(void *, thread->th.th_sleep_loc))) != 3510 NULL) { 3511 KA_TRACE( 3512 10, 3513 ("__kmp_wait_to_unref_task_team: T#%d waking up thread T#%d\n", 3514 __kmp_gtid_from_thread(thread), __kmp_gtid_from_thread(thread))); 3515 __kmp_null_resume_wrapper(__kmp_gtid_from_thread(thread), sleep_loc); 3516 } 3517 } 3518 } 3519 if (done) { 3520 break; 3521 } 3522 3523 // If oversubscribed or have waited a bit, yield. 3524 KMP_YIELD_OVERSUB_ELSE_SPIN(spins); 3525 } 3526 } 3527 3528 // __kmp_task_team_setup: Create a task_team for the current team, but use 3529 // an already created, unused one if it already exists. 3530 void __kmp_task_team_setup(kmp_info_t *this_thr, kmp_team_t *team, int always) { 3531 KMP_DEBUG_ASSERT(__kmp_tasking_mode != tskm_immediate_exec); 3532 3533 // If this task_team hasn't been created yet, allocate it. It will be used in 3534 // the region after the next. 3535 // If it exists, it is the current task team and shouldn't be touched yet as 3536 // it may still be in use. 3537 if (team->t.t_task_team[this_thr->th.th_task_state] == NULL && 3538 (always || team->t.t_nproc > 1)) { 3539 team->t.t_task_team[this_thr->th.th_task_state] = 3540 __kmp_allocate_task_team(this_thr, team); 3541 KA_TRACE(20, ("__kmp_task_team_setup: Master T#%d created new task_team %p " 3542 "for team %d at parity=%d\n", 3543 __kmp_gtid_from_thread(this_thr), 3544 team->t.t_task_team[this_thr->th.th_task_state], 3545 ((team != NULL) ? team->t.t_id : -1), 3546 this_thr->th.th_task_state)); 3547 } 3548 3549 // After threads exit the release, they will call sync, and then point to this 3550 // other task_team; make sure it is allocated and properly initialized. As 3551 // threads spin in the barrier release phase, they will continue to use the 3552 // previous task_team struct(above), until they receive the signal to stop 3553 // checking for tasks (they can't safely reference the kmp_team_t struct, 3554 // which could be reallocated by the master thread). No task teams are formed 3555 // for serialized teams. 3556 if (team->t.t_nproc > 1) { 3557 int other_team = 1 - this_thr->th.th_task_state; 3558 if (team->t.t_task_team[other_team] == NULL) { // setup other team as well 3559 team->t.t_task_team[other_team] = 3560 __kmp_allocate_task_team(this_thr, team); 3561 KA_TRACE(20, ("__kmp_task_team_setup: Master T#%d created second new " 3562 "task_team %p for team %d at parity=%d\n", 3563 __kmp_gtid_from_thread(this_thr), 3564 team->t.t_task_team[other_team], 3565 ((team != NULL) ? team->t.t_id : -1), other_team)); 3566 } else { // Leave the old task team struct in place for the upcoming region; 3567 // adjust as needed 3568 kmp_task_team_t *task_team = team->t.t_task_team[other_team]; 3569 if (!task_team->tt.tt_active || 3570 team->t.t_nproc != task_team->tt.tt_nproc) { 3571 TCW_4(task_team->tt.tt_nproc, team->t.t_nproc); 3572 TCW_4(task_team->tt.tt_found_tasks, FALSE); 3573 #if OMP_45_ENABLED 3574 TCW_4(task_team->tt.tt_found_proxy_tasks, FALSE); 3575 #endif 3576 KMP_ATOMIC_ST_REL(&task_team->tt.tt_unfinished_threads, 3577 team->t.t_nproc); 3578 TCW_4(task_team->tt.tt_active, TRUE); 3579 } 3580 // if team size has changed, the first thread to enable tasking will 3581 // realloc threads_data if necessary 3582 KA_TRACE(20, ("__kmp_task_team_setup: Master T#%d reset next task_team " 3583 "%p for team %d at parity=%d\n", 3584 __kmp_gtid_from_thread(this_thr), 3585 team->t.t_task_team[other_team], 3586 ((team != NULL) ? team->t.t_id : -1), other_team)); 3587 } 3588 } 3589 } 3590 3591 // __kmp_task_team_sync: Propagation of task team data from team to threads 3592 // which happens just after the release phase of a team barrier. This may be 3593 // called by any thread, but only for teams with # threads > 1. 3594 void __kmp_task_team_sync(kmp_info_t *this_thr, kmp_team_t *team) { 3595 KMP_DEBUG_ASSERT(__kmp_tasking_mode != tskm_immediate_exec); 3596 3597 // Toggle the th_task_state field, to switch which task_team this thread 3598 // refers to 3599 this_thr->th.th_task_state = 1 - this_thr->th.th_task_state; 3600 // It is now safe to propagate the task team pointer from the team struct to 3601 // the current thread. 3602 TCW_PTR(this_thr->th.th_task_team, 3603 team->t.t_task_team[this_thr->th.th_task_state]); 3604 KA_TRACE(20, 3605 ("__kmp_task_team_sync: Thread T#%d task team switched to task_team " 3606 "%p from Team #%d (parity=%d)\n", 3607 __kmp_gtid_from_thread(this_thr), this_thr->th.th_task_team, 3608 ((team != NULL) ? team->t.t_id : -1), this_thr->th.th_task_state)); 3609 } 3610 3611 // __kmp_task_team_wait: Master thread waits for outstanding tasks after the 3612 // barrier gather phase. Only called by master thread if #threads in team > 1 or 3613 // if proxy tasks were created. 3614 // 3615 // wait is a flag that defaults to 1 (see kmp.h), but waiting can be turned off 3616 // by passing in 0 optionally as the last argument. When wait is zero, master 3617 // thread does not wait for unfinished_threads to reach 0. 3618 void __kmp_task_team_wait( 3619 kmp_info_t *this_thr, 3620 kmp_team_t *team USE_ITT_BUILD_ARG(void *itt_sync_obj), int wait) { 3621 kmp_task_team_t *task_team = team->t.t_task_team[this_thr->th.th_task_state]; 3622 3623 KMP_DEBUG_ASSERT(__kmp_tasking_mode != tskm_immediate_exec); 3624 KMP_DEBUG_ASSERT(task_team == this_thr->th.th_task_team); 3625 3626 if ((task_team != NULL) && KMP_TASKING_ENABLED(task_team)) { 3627 if (wait) { 3628 KA_TRACE(20, ("__kmp_task_team_wait: Master T#%d waiting for all tasks " 3629 "(for unfinished_threads to reach 0) on task_team = %p\n", 3630 __kmp_gtid_from_thread(this_thr), task_team)); 3631 // Worker threads may have dropped through to release phase, but could 3632 // still be executing tasks. Wait here for tasks to complete. To avoid 3633 // memory contention, only master thread checks termination condition. 3634 kmp_flag_32 flag(RCAST(std::atomic<kmp_uint32> *, 3635 &task_team->tt.tt_unfinished_threads), 3636 0U); 3637 flag.wait(this_thr, TRUE USE_ITT_BUILD_ARG(itt_sync_obj)); 3638 } 3639 // Deactivate the old task team, so that the worker threads will stop 3640 // referencing it while spinning. 3641 KA_TRACE( 3642 20, 3643 ("__kmp_task_team_wait: Master T#%d deactivating task_team %p: " 3644 "setting active to false, setting local and team's pointer to NULL\n", 3645 __kmp_gtid_from_thread(this_thr), task_team)); 3646 #if OMP_45_ENABLED 3647 KMP_DEBUG_ASSERT(task_team->tt.tt_nproc > 1 || 3648 task_team->tt.tt_found_proxy_tasks == TRUE); 3649 TCW_SYNC_4(task_team->tt.tt_found_proxy_tasks, FALSE); 3650 #else 3651 KMP_DEBUG_ASSERT(task_team->tt.tt_nproc > 1); 3652 #endif 3653 KMP_CHECK_UPDATE(task_team->tt.tt_untied_task_encountered, 0); 3654 TCW_SYNC_4(task_team->tt.tt_active, FALSE); 3655 KMP_MB(); 3656 3657 TCW_PTR(this_thr->th.th_task_team, NULL); 3658 } 3659 } 3660 3661 // __kmp_tasking_barrier: 3662 // This routine may only called when __kmp_tasking_mode == tskm_extra_barrier. 3663 // Internal function to execute all tasks prior to a regular barrier or a join 3664 // barrier. It is a full barrier itself, which unfortunately turns regular 3665 // barriers into double barriers and join barriers into 1 1/2 barriers. 3666 void __kmp_tasking_barrier(kmp_team_t *team, kmp_info_t *thread, int gtid) { 3667 std::atomic<kmp_uint32> *spin = RCAST( 3668 std::atomic<kmp_uint32> *, 3669 &team->t.t_task_team[thread->th.th_task_state]->tt.tt_unfinished_threads); 3670 int flag = FALSE; 3671 KMP_DEBUG_ASSERT(__kmp_tasking_mode == tskm_extra_barrier); 3672 3673 #if USE_ITT_BUILD 3674 KMP_FSYNC_SPIN_INIT(spin, NULL); 3675 #endif /* USE_ITT_BUILD */ 3676 kmp_flag_32 spin_flag(spin, 0U); 3677 while (!spin_flag.execute_tasks(thread, gtid, TRUE, 3678 &flag USE_ITT_BUILD_ARG(NULL), 0)) { 3679 #if USE_ITT_BUILD 3680 // TODO: What about itt_sync_obj?? 3681 KMP_FSYNC_SPIN_PREPARE(RCAST(void *, spin)); 3682 #endif /* USE_ITT_BUILD */ 3683 3684 if (TCR_4(__kmp_global.g.g_done)) { 3685 if (__kmp_global.g.g_abort) 3686 __kmp_abort_thread(); 3687 break; 3688 } 3689 KMP_YIELD(TRUE); 3690 } 3691 #if USE_ITT_BUILD 3692 KMP_FSYNC_SPIN_ACQUIRED(RCAST(void *, spin)); 3693 #endif /* USE_ITT_BUILD */ 3694 } 3695 3696 #if OMP_45_ENABLED 3697 3698 // __kmp_give_task puts a task into a given thread queue if: 3699 // - the queue for that thread was created 3700 // - there's space in that queue 3701 // Because of this, __kmp_push_task needs to check if there's space after 3702 // getting the lock 3703 static bool __kmp_give_task(kmp_info_t *thread, kmp_int32 tid, kmp_task_t *task, 3704 kmp_int32 pass) { 3705 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 3706 kmp_task_team_t *task_team = taskdata->td_task_team; 3707 3708 KA_TRACE(20, ("__kmp_give_task: trying to give task %p to thread %d.\n", 3709 taskdata, tid)); 3710 3711 // If task_team is NULL something went really bad... 3712 KMP_DEBUG_ASSERT(task_team != NULL); 3713 3714 bool result = false; 3715 kmp_thread_data_t *thread_data = &task_team->tt.tt_threads_data[tid]; 3716 3717 if (thread_data->td.td_deque == NULL) { 3718 // There's no queue in this thread, go find another one 3719 // We're guaranteed that at least one thread has a queue 3720 KA_TRACE(30, 3721 ("__kmp_give_task: thread %d has no queue while giving task %p.\n", 3722 tid, taskdata)); 3723 return result; 3724 } 3725 3726 if (TCR_4(thread_data->td.td_deque_ntasks) >= 3727 TASK_DEQUE_SIZE(thread_data->td)) { 3728 KA_TRACE( 3729 30, 3730 ("__kmp_give_task: queue is full while giving task %p to thread %d.\n", 3731 taskdata, tid)); 3732 3733 // if this deque is bigger than the pass ratio give a chance to another 3734 // thread 3735 if (TASK_DEQUE_SIZE(thread_data->td) / INITIAL_TASK_DEQUE_SIZE >= pass) 3736 return result; 3737 3738 __kmp_acquire_bootstrap_lock(&thread_data->td.td_deque_lock); 3739 __kmp_realloc_task_deque(thread, thread_data); 3740 3741 } else { 3742 3743 __kmp_acquire_bootstrap_lock(&thread_data->td.td_deque_lock); 3744 3745 if (TCR_4(thread_data->td.td_deque_ntasks) >= 3746 TASK_DEQUE_SIZE(thread_data->td)) { 3747 KA_TRACE(30, ("__kmp_give_task: queue is full while giving task %p to " 3748 "thread %d.\n", 3749 taskdata, tid)); 3750 3751 // if this deque is bigger than the pass ratio give a chance to another 3752 // thread 3753 if (TASK_DEQUE_SIZE(thread_data->td) / INITIAL_TASK_DEQUE_SIZE >= pass) 3754 goto release_and_exit; 3755 3756 __kmp_realloc_task_deque(thread, thread_data); 3757 } 3758 } 3759 3760 // lock is held here, and there is space in the deque 3761 3762 thread_data->td.td_deque[thread_data->td.td_deque_tail] = taskdata; 3763 // Wrap index. 3764 thread_data->td.td_deque_tail = 3765 (thread_data->td.td_deque_tail + 1) & TASK_DEQUE_MASK(thread_data->td); 3766 TCW_4(thread_data->td.td_deque_ntasks, 3767 TCR_4(thread_data->td.td_deque_ntasks) + 1); 3768 3769 result = true; 3770 KA_TRACE(30, ("__kmp_give_task: successfully gave task %p to thread %d.\n", 3771 taskdata, tid)); 3772 3773 release_and_exit: 3774 __kmp_release_bootstrap_lock(&thread_data->td.td_deque_lock); 3775 3776 return result; 3777 } 3778 3779 /* The finish of the proxy tasks is divided in two pieces: 3780 - the top half is the one that can be done from a thread outside the team 3781 - the bottom half must be run from a thread within the team 3782 3783 In order to run the bottom half the task gets queued back into one of the 3784 threads of the team. Once the td_incomplete_child_task counter of the parent 3785 is decremented the threads can leave the barriers. So, the bottom half needs 3786 to be queued before the counter is decremented. The top half is therefore 3787 divided in two parts: 3788 - things that can be run before queuing the bottom half 3789 - things that must be run after queuing the bottom half 3790 3791 This creates a second race as the bottom half can free the task before the 3792 second top half is executed. To avoid this we use the 3793 td_incomplete_child_task of the proxy task to synchronize the top and bottom 3794 half. */ 3795 static void __kmp_first_top_half_finish_proxy(kmp_taskdata_t *taskdata) { 3796 KMP_DEBUG_ASSERT(taskdata->td_flags.tasktype == TASK_EXPLICIT); 3797 KMP_DEBUG_ASSERT(taskdata->td_flags.proxy == TASK_PROXY); 3798 KMP_DEBUG_ASSERT(taskdata->td_flags.complete == 0); 3799 KMP_DEBUG_ASSERT(taskdata->td_flags.freed == 0); 3800 3801 taskdata->td_flags.complete = 1; // mark the task as completed 3802 3803 if (taskdata->td_taskgroup) 3804 KMP_ATOMIC_DEC(&taskdata->td_taskgroup->count); 3805 3806 // Create an imaginary children for this task so the bottom half cannot 3807 // release the task before we have completed the second top half 3808 KMP_ATOMIC_INC(&taskdata->td_incomplete_child_tasks); 3809 } 3810 3811 static void __kmp_second_top_half_finish_proxy(kmp_taskdata_t *taskdata) { 3812 kmp_int32 children = 0; 3813 3814 // Predecrement simulated by "- 1" calculation 3815 children = 3816 KMP_ATOMIC_DEC(&taskdata->td_parent->td_incomplete_child_tasks) - 1; 3817 KMP_DEBUG_ASSERT(children >= 0); 3818 3819 // Remove the imaginary children 3820 KMP_ATOMIC_DEC(&taskdata->td_incomplete_child_tasks); 3821 } 3822 3823 static void __kmp_bottom_half_finish_proxy(kmp_int32 gtid, kmp_task_t *ptask) { 3824 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(ptask); 3825 kmp_info_t *thread = __kmp_threads[gtid]; 3826 3827 KMP_DEBUG_ASSERT(taskdata->td_flags.proxy == TASK_PROXY); 3828 KMP_DEBUG_ASSERT(taskdata->td_flags.complete == 3829 1); // top half must run before bottom half 3830 3831 // We need to wait to make sure the top half is finished 3832 // Spinning here should be ok as this should happen quickly 3833 while (KMP_ATOMIC_LD_ACQ(&taskdata->td_incomplete_child_tasks) > 0) 3834 ; 3835 3836 __kmp_release_deps(gtid, taskdata); 3837 __kmp_free_task_and_ancestors(gtid, taskdata, thread); 3838 } 3839 3840 /*! 3841 @ingroup TASKING 3842 @param gtid Global Thread ID of encountering thread 3843 @param ptask Task which execution is completed 3844 3845 Execute the completation of a proxy task from a thread of that is part of the 3846 team. Run first and bottom halves directly. 3847 */ 3848 void __kmpc_proxy_task_completed(kmp_int32 gtid, kmp_task_t *ptask) { 3849 KMP_DEBUG_ASSERT(ptask != NULL); 3850 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(ptask); 3851 KA_TRACE( 3852 10, ("__kmp_proxy_task_completed(enter): T#%d proxy task %p completing\n", 3853 gtid, taskdata)); 3854 3855 KMP_DEBUG_ASSERT(taskdata->td_flags.proxy == TASK_PROXY); 3856 3857 __kmp_first_top_half_finish_proxy(taskdata); 3858 __kmp_second_top_half_finish_proxy(taskdata); 3859 __kmp_bottom_half_finish_proxy(gtid, ptask); 3860 3861 KA_TRACE(10, 3862 ("__kmp_proxy_task_completed(exit): T#%d proxy task %p completing\n", 3863 gtid, taskdata)); 3864 } 3865 3866 /*! 3867 @ingroup TASKING 3868 @param ptask Task which execution is completed 3869 3870 Execute the completation of a proxy task from a thread that could not belong to 3871 the team. 3872 */ 3873 void __kmpc_proxy_task_completed_ooo(kmp_task_t *ptask) { 3874 KMP_DEBUG_ASSERT(ptask != NULL); 3875 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(ptask); 3876 3877 KA_TRACE( 3878 10, 3879 ("__kmp_proxy_task_completed_ooo(enter): proxy task completing ooo %p\n", 3880 taskdata)); 3881 3882 KMP_DEBUG_ASSERT(taskdata->td_flags.proxy == TASK_PROXY); 3883 3884 __kmp_first_top_half_finish_proxy(taskdata); 3885 3886 // Enqueue task to complete bottom half completion from a thread within the 3887 // corresponding team 3888 kmp_team_t *team = taskdata->td_team; 3889 kmp_int32 nthreads = team->t.t_nproc; 3890 kmp_info_t *thread; 3891 3892 // This should be similar to start_k = __kmp_get_random( thread ) % nthreads 3893 // but we cannot use __kmp_get_random here 3894 kmp_int32 start_k = 0; 3895 kmp_int32 pass = 1; 3896 kmp_int32 k = start_k; 3897 3898 do { 3899 // For now we're just linearly trying to find a thread 3900 thread = team->t.t_threads[k]; 3901 k = (k + 1) % nthreads; 3902 3903 // we did a full pass through all the threads 3904 if (k == start_k) 3905 pass = pass << 1; 3906 3907 } while (!__kmp_give_task(thread, k, ptask, pass)); 3908 3909 __kmp_second_top_half_finish_proxy(taskdata); 3910 3911 KA_TRACE( 3912 10, 3913 ("__kmp_proxy_task_completed_ooo(exit): proxy task completing ooo %p\n", 3914 taskdata)); 3915 } 3916 3917 // __kmp_task_dup_alloc: Allocate the taskdata and make a copy of source task 3918 // for taskloop 3919 // 3920 // thread: allocating thread 3921 // task_src: pointer to source task to be duplicated 3922 // returns: a pointer to the allocated kmp_task_t structure (task). 3923 kmp_task_t *__kmp_task_dup_alloc(kmp_info_t *thread, kmp_task_t *task_src) { 3924 kmp_task_t *task; 3925 kmp_taskdata_t *taskdata; 3926 kmp_taskdata_t *taskdata_src; 3927 kmp_taskdata_t *parent_task = thread->th.th_current_task; 3928 size_t shareds_offset; 3929 size_t task_size; 3930 3931 KA_TRACE(10, ("__kmp_task_dup_alloc(enter): Th %p, source task %p\n", thread, 3932 task_src)); 3933 taskdata_src = KMP_TASK_TO_TASKDATA(task_src); 3934 KMP_DEBUG_ASSERT(taskdata_src->td_flags.proxy == 3935 TASK_FULL); // it should not be proxy task 3936 KMP_DEBUG_ASSERT(taskdata_src->td_flags.tasktype == TASK_EXPLICIT); 3937 task_size = taskdata_src->td_size_alloc; 3938 3939 // Allocate a kmp_taskdata_t block and a kmp_task_t block. 3940 KA_TRACE(30, ("__kmp_task_dup_alloc: Th %p, malloc size %ld\n", thread, 3941 task_size)); 3942 #if USE_FAST_MEMORY 3943 taskdata = (kmp_taskdata_t *)__kmp_fast_allocate(thread, task_size); 3944 #else 3945 taskdata = (kmp_taskdata_t *)__kmp_thread_malloc(thread, task_size); 3946 #endif /* USE_FAST_MEMORY */ 3947 KMP_MEMCPY(taskdata, taskdata_src, task_size); 3948 3949 task = KMP_TASKDATA_TO_TASK(taskdata); 3950 3951 // Initialize new task (only specific fields not affected by memcpy) 3952 taskdata->td_task_id = KMP_GEN_TASK_ID(); 3953 if (task->shareds != NULL) { // need setup shareds pointer 3954 shareds_offset = (char *)task_src->shareds - (char *)taskdata_src; 3955 task->shareds = &((char *)taskdata)[shareds_offset]; 3956 KMP_DEBUG_ASSERT((((kmp_uintptr_t)task->shareds) & (sizeof(void *) - 1)) == 3957 0); 3958 } 3959 taskdata->td_alloc_thread = thread; 3960 taskdata->td_parent = parent_task; 3961 taskdata->td_taskgroup = 3962 parent_task 3963 ->td_taskgroup; // task inherits the taskgroup from the parent task 3964 3965 // Only need to keep track of child task counts if team parallel and tasking 3966 // not serialized 3967 if (!(taskdata->td_flags.team_serial || taskdata->td_flags.tasking_ser)) { 3968 KMP_ATOMIC_INC(&parent_task->td_incomplete_child_tasks); 3969 if (parent_task->td_taskgroup) 3970 KMP_ATOMIC_INC(&parent_task->td_taskgroup->count); 3971 // Only need to keep track of allocated child tasks for explicit tasks since 3972 // implicit not deallocated 3973 if (taskdata->td_parent->td_flags.tasktype == TASK_EXPLICIT) 3974 KMP_ATOMIC_INC(&taskdata->td_parent->td_allocated_child_tasks); 3975 } 3976 3977 KA_TRACE(20, 3978 ("__kmp_task_dup_alloc(exit): Th %p, created task %p, parent=%p\n", 3979 thread, taskdata, taskdata->td_parent)); 3980 #if OMPT_SUPPORT 3981 if (UNLIKELY(ompt_enabled.enabled)) 3982 __ompt_task_init(taskdata, thread->th.th_info.ds.ds_gtid); 3983 #endif 3984 return task; 3985 } 3986 3987 // Routine optionally generated by the compiler for setting the lastprivate flag 3988 // and calling needed constructors for private/firstprivate objects 3989 // (used to form taskloop tasks from pattern task) 3990 // Parameters: dest task, src task, lastprivate flag. 3991 typedef void (*p_task_dup_t)(kmp_task_t *, kmp_task_t *, kmp_int32); 3992 3993 KMP_BUILD_ASSERT(sizeof(long) == 4 || sizeof(long) == 8); 3994 3995 // class to encapsulate manipulating loop bounds in a taskloop task. 3996 // this abstracts away the Intel vs GOMP taskloop interface for setting/getting 3997 // the loop bound variables. 3998 class kmp_taskloop_bounds_t { 3999 kmp_task_t *task; 4000 const kmp_taskdata_t *taskdata; 4001 size_t lower_offset; 4002 size_t upper_offset; 4003 4004 public: 4005 kmp_taskloop_bounds_t(kmp_task_t *_task, kmp_uint64 *lb, kmp_uint64 *ub) 4006 : task(_task), taskdata(KMP_TASK_TO_TASKDATA(task)), 4007 lower_offset((char *)lb - (char *)task), 4008 upper_offset((char *)ub - (char *)task) { 4009 KMP_DEBUG_ASSERT((char *)lb > (char *)_task); 4010 KMP_DEBUG_ASSERT((char *)ub > (char *)_task); 4011 } 4012 kmp_taskloop_bounds_t(kmp_task_t *_task, const kmp_taskloop_bounds_t &bounds) 4013 : task(_task), taskdata(KMP_TASK_TO_TASKDATA(_task)), 4014 lower_offset(bounds.lower_offset), upper_offset(bounds.upper_offset) {} 4015 size_t get_lower_offset() const { return lower_offset; } 4016 size_t get_upper_offset() const { return upper_offset; } 4017 kmp_uint64 get_lb() const { 4018 kmp_int64 retval; 4019 #if defined(KMP_GOMP_COMPAT) 4020 // Intel task just returns the lower bound normally 4021 if (!taskdata->td_flags.native) { 4022 retval = *(kmp_int64 *)((char *)task + lower_offset); 4023 } else { 4024 // GOMP task has to take into account the sizeof(long) 4025 if (taskdata->td_size_loop_bounds == 4) { 4026 kmp_int32 *lb = RCAST(kmp_int32 *, task->shareds); 4027 retval = (kmp_int64)*lb; 4028 } else { 4029 kmp_int64 *lb = RCAST(kmp_int64 *, task->shareds); 4030 retval = (kmp_int64)*lb; 4031 } 4032 } 4033 #else 4034 retval = *(kmp_int64 *)((char *)task + lower_offset); 4035 #endif // defined(KMP_GOMP_COMPAT) 4036 return retval; 4037 } 4038 kmp_uint64 get_ub() const { 4039 kmp_int64 retval; 4040 #if defined(KMP_GOMP_COMPAT) 4041 // Intel task just returns the upper bound normally 4042 if (!taskdata->td_flags.native) { 4043 retval = *(kmp_int64 *)((char *)task + upper_offset); 4044 } else { 4045 // GOMP task has to take into account the sizeof(long) 4046 if (taskdata->td_size_loop_bounds == 4) { 4047 kmp_int32 *ub = RCAST(kmp_int32 *, task->shareds) + 1; 4048 retval = (kmp_int64)*ub; 4049 } else { 4050 kmp_int64 *ub = RCAST(kmp_int64 *, task->shareds) + 1; 4051 retval = (kmp_int64)*ub; 4052 } 4053 } 4054 #else 4055 retval = *(kmp_int64 *)((char *)task + upper_offset); 4056 #endif // defined(KMP_GOMP_COMPAT) 4057 return retval; 4058 } 4059 void set_lb(kmp_uint64 lb) { 4060 #if defined(KMP_GOMP_COMPAT) 4061 // Intel task just sets the lower bound normally 4062 if (!taskdata->td_flags.native) { 4063 *(kmp_uint64 *)((char *)task + lower_offset) = lb; 4064 } else { 4065 // GOMP task has to take into account the sizeof(long) 4066 if (taskdata->td_size_loop_bounds == 4) { 4067 kmp_uint32 *lower = RCAST(kmp_uint32 *, task->shareds); 4068 *lower = (kmp_uint32)lb; 4069 } else { 4070 kmp_uint64 *lower = RCAST(kmp_uint64 *, task->shareds); 4071 *lower = (kmp_uint64)lb; 4072 } 4073 } 4074 #else 4075 *(kmp_uint64 *)((char *)task + lower_offset) = lb; 4076 #endif // defined(KMP_GOMP_COMPAT) 4077 } 4078 void set_ub(kmp_uint64 ub) { 4079 #if defined(KMP_GOMP_COMPAT) 4080 // Intel task just sets the upper bound normally 4081 if (!taskdata->td_flags.native) { 4082 *(kmp_uint64 *)((char *)task + upper_offset) = ub; 4083 } else { 4084 // GOMP task has to take into account the sizeof(long) 4085 if (taskdata->td_size_loop_bounds == 4) { 4086 kmp_uint32 *upper = RCAST(kmp_uint32 *, task->shareds) + 1; 4087 *upper = (kmp_uint32)ub; 4088 } else { 4089 kmp_uint64 *upper = RCAST(kmp_uint64 *, task->shareds) + 1; 4090 *upper = (kmp_uint64)ub; 4091 } 4092 } 4093 #else 4094 *(kmp_uint64 *)((char *)task + upper_offset) = ub; 4095 #endif // defined(KMP_GOMP_COMPAT) 4096 } 4097 }; 4098 4099 // __kmp_taskloop_linear: Start tasks of the taskloop linearly 4100 // 4101 // loc Source location information 4102 // gtid Global thread ID 4103 // task Pattern task, exposes the loop iteration range 4104 // lb Pointer to loop lower bound in task structure 4105 // ub Pointer to loop upper bound in task structure 4106 // st Loop stride 4107 // ub_glob Global upper bound (used for lastprivate check) 4108 // num_tasks Number of tasks to execute 4109 // grainsize Number of loop iterations per task 4110 // extras Number of chunks with grainsize+1 iterations 4111 // tc Iterations count 4112 // task_dup Tasks duplication routine 4113 // codeptr_ra Return address for OMPT events 4114 void __kmp_taskloop_linear(ident_t *loc, int gtid, kmp_task_t *task, 4115 kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, 4116 kmp_uint64 ub_glob, kmp_uint64 num_tasks, 4117 kmp_uint64 grainsize, kmp_uint64 extras, 4118 kmp_uint64 tc, 4119 #if OMPT_SUPPORT 4120 void *codeptr_ra, 4121 #endif 4122 void *task_dup) { 4123 KMP_COUNT_BLOCK(OMP_TASKLOOP); 4124 KMP_TIME_PARTITIONED_BLOCK(OMP_taskloop_scheduling); 4125 p_task_dup_t ptask_dup = (p_task_dup_t)task_dup; 4126 // compiler provides global bounds here 4127 kmp_taskloop_bounds_t task_bounds(task, lb, ub); 4128 kmp_uint64 lower = task_bounds.get_lb(); 4129 kmp_uint64 upper = task_bounds.get_ub(); 4130 kmp_uint64 i; 4131 kmp_info_t *thread = __kmp_threads[gtid]; 4132 kmp_taskdata_t *current_task = thread->th.th_current_task; 4133 kmp_task_t *next_task; 4134 kmp_int32 lastpriv = 0; 4135 4136 KMP_DEBUG_ASSERT(tc == num_tasks * grainsize + extras); 4137 KMP_DEBUG_ASSERT(num_tasks > extras); 4138 KMP_DEBUG_ASSERT(num_tasks > 0); 4139 KA_TRACE(20, ("__kmp_taskloop_linear: T#%d: %lld tasks, grainsize %lld, " 4140 "extras %lld, i=%lld,%lld(%d)%lld, dup %p\n", 4141 gtid, num_tasks, grainsize, extras, lower, upper, ub_glob, st, 4142 task_dup)); 4143 4144 // Launch num_tasks tasks, assign grainsize iterations each task 4145 for (i = 0; i < num_tasks; ++i) { 4146 kmp_uint64 chunk_minus_1; 4147 if (extras == 0) { 4148 chunk_minus_1 = grainsize - 1; 4149 } else { 4150 chunk_minus_1 = grainsize; 4151 --extras; // first extras iterations get bigger chunk (grainsize+1) 4152 } 4153 upper = lower + st * chunk_minus_1; 4154 if (i == num_tasks - 1) { 4155 // schedule the last task, set lastprivate flag if needed 4156 if (st == 1) { // most common case 4157 KMP_DEBUG_ASSERT(upper == *ub); 4158 if (upper == ub_glob) 4159 lastpriv = 1; 4160 } else if (st > 0) { // positive loop stride 4161 KMP_DEBUG_ASSERT((kmp_uint64)st > *ub - upper); 4162 if ((kmp_uint64)st > ub_glob - upper) 4163 lastpriv = 1; 4164 } else { // negative loop stride 4165 KMP_DEBUG_ASSERT(upper + st < *ub); 4166 if (upper - ub_glob < (kmp_uint64)(-st)) 4167 lastpriv = 1; 4168 } 4169 } 4170 next_task = __kmp_task_dup_alloc(thread, task); // allocate new task 4171 kmp_taskdata_t *next_taskdata = KMP_TASK_TO_TASKDATA(next_task); 4172 kmp_taskloop_bounds_t next_task_bounds = 4173 kmp_taskloop_bounds_t(next_task, task_bounds); 4174 4175 // adjust task-specific bounds 4176 next_task_bounds.set_lb(lower); 4177 if (next_taskdata->td_flags.native) { 4178 next_task_bounds.set_ub(upper + (st > 0 ? 1 : -1)); 4179 } else { 4180 next_task_bounds.set_ub(upper); 4181 } 4182 if (ptask_dup != NULL) // set lastprivate flag, construct fistprivates, etc. 4183 ptask_dup(next_task, task, lastpriv); 4184 KA_TRACE(40, 4185 ("__kmp_taskloop_linear: T#%d; task #%llu: task %p: lower %lld, " 4186 "upper %lld stride %lld, (offsets %p %p)\n", 4187 gtid, i, next_task, lower, upper, st, 4188 next_task_bounds.get_lower_offset(), 4189 next_task_bounds.get_upper_offset())); 4190 #if OMPT_SUPPORT 4191 __kmp_omp_taskloop_task(NULL, gtid, next_task, 4192 codeptr_ra); // schedule new task 4193 #else 4194 __kmp_omp_task(gtid, next_task, true); // schedule new task 4195 #endif 4196 lower = upper + st; // adjust lower bound for the next iteration 4197 } 4198 // free the pattern task and exit 4199 __kmp_task_start(gtid, task, current_task); // make internal bookkeeping 4200 // do not execute the pattern task, just do internal bookkeeping 4201 __kmp_task_finish<false>(gtid, task, current_task); 4202 } 4203 4204 // Structure to keep taskloop parameters for auxiliary task 4205 // kept in the shareds of the task structure. 4206 typedef struct __taskloop_params { 4207 kmp_task_t *task; 4208 kmp_uint64 *lb; 4209 kmp_uint64 *ub; 4210 void *task_dup; 4211 kmp_int64 st; 4212 kmp_uint64 ub_glob; 4213 kmp_uint64 num_tasks; 4214 kmp_uint64 grainsize; 4215 kmp_uint64 extras; 4216 kmp_uint64 tc; 4217 kmp_uint64 num_t_min; 4218 #if OMPT_SUPPORT 4219 void *codeptr_ra; 4220 #endif 4221 } __taskloop_params_t; 4222 4223 void __kmp_taskloop_recur(ident_t *, int, kmp_task_t *, kmp_uint64 *, 4224 kmp_uint64 *, kmp_int64, kmp_uint64, kmp_uint64, 4225 kmp_uint64, kmp_uint64, kmp_uint64, kmp_uint64, 4226 #if OMPT_SUPPORT 4227 void *, 4228 #endif 4229 void *); 4230 4231 // Execute part of the the taskloop submitted as a task. 4232 int __kmp_taskloop_task(int gtid, void *ptask) { 4233 __taskloop_params_t *p = 4234 (__taskloop_params_t *)((kmp_task_t *)ptask)->shareds; 4235 kmp_task_t *task = p->task; 4236 kmp_uint64 *lb = p->lb; 4237 kmp_uint64 *ub = p->ub; 4238 void *task_dup = p->task_dup; 4239 // p_task_dup_t ptask_dup = (p_task_dup_t)task_dup; 4240 kmp_int64 st = p->st; 4241 kmp_uint64 ub_glob = p->ub_glob; 4242 kmp_uint64 num_tasks = p->num_tasks; 4243 kmp_uint64 grainsize = p->grainsize; 4244 kmp_uint64 extras = p->extras; 4245 kmp_uint64 tc = p->tc; 4246 kmp_uint64 num_t_min = p->num_t_min; 4247 #if OMPT_SUPPORT 4248 void *codeptr_ra = p->codeptr_ra; 4249 #endif 4250 #if KMP_DEBUG 4251 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 4252 KMP_DEBUG_ASSERT(task != NULL); 4253 KA_TRACE(20, ("__kmp_taskloop_task: T#%d, task %p: %lld tasks, grainsize" 4254 " %lld, extras %lld, i=%lld,%lld(%d), dup %p\n", 4255 gtid, taskdata, num_tasks, grainsize, extras, *lb, *ub, st, 4256 task_dup)); 4257 #endif 4258 KMP_DEBUG_ASSERT(num_tasks * 2 + 1 > num_t_min); 4259 if (num_tasks > num_t_min) 4260 __kmp_taskloop_recur(NULL, gtid, task, lb, ub, st, ub_glob, num_tasks, 4261 grainsize, extras, tc, num_t_min, 4262 #if OMPT_SUPPORT 4263 codeptr_ra, 4264 #endif 4265 task_dup); 4266 else 4267 __kmp_taskloop_linear(NULL, gtid, task, lb, ub, st, ub_glob, num_tasks, 4268 grainsize, extras, tc, 4269 #if OMPT_SUPPORT 4270 codeptr_ra, 4271 #endif 4272 task_dup); 4273 4274 KA_TRACE(40, ("__kmp_taskloop_task(exit): T#%d\n", gtid)); 4275 return 0; 4276 } 4277 4278 // Schedule part of the the taskloop as a task, 4279 // execute the rest of the the taskloop. 4280 // 4281 // loc Source location information 4282 // gtid Global thread ID 4283 // task Pattern task, exposes the loop iteration range 4284 // lb Pointer to loop lower bound in task structure 4285 // ub Pointer to loop upper bound in task structure 4286 // st Loop stride 4287 // ub_glob Global upper bound (used for lastprivate check) 4288 // num_tasks Number of tasks to execute 4289 // grainsize Number of loop iterations per task 4290 // extras Number of chunks with grainsize+1 iterations 4291 // tc Iterations count 4292 // num_t_min Threashold to launch tasks recursively 4293 // task_dup Tasks duplication routine 4294 // codeptr_ra Return address for OMPT events 4295 void __kmp_taskloop_recur(ident_t *loc, int gtid, kmp_task_t *task, 4296 kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, 4297 kmp_uint64 ub_glob, kmp_uint64 num_tasks, 4298 kmp_uint64 grainsize, kmp_uint64 extras, 4299 kmp_uint64 tc, kmp_uint64 num_t_min, 4300 #if OMPT_SUPPORT 4301 void *codeptr_ra, 4302 #endif 4303 void *task_dup) { 4304 #if KMP_DEBUG 4305 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 4306 KMP_DEBUG_ASSERT(task != NULL); 4307 KMP_DEBUG_ASSERT(num_tasks > num_t_min); 4308 KA_TRACE(20, ("__kmp_taskloop_recur: T#%d, task %p: %lld tasks, grainsize" 4309 " %lld, extras %lld, i=%lld,%lld(%d), dup %p\n", 4310 gtid, taskdata, num_tasks, grainsize, extras, *lb, *ub, st, 4311 task_dup)); 4312 #endif 4313 p_task_dup_t ptask_dup = (p_task_dup_t)task_dup; 4314 kmp_uint64 lower = *lb; 4315 kmp_info_t *thread = __kmp_threads[gtid]; 4316 // kmp_taskdata_t *current_task = thread->th.th_current_task; 4317 kmp_task_t *next_task; 4318 size_t lower_offset = 4319 (char *)lb - (char *)task; // remember offset of lb in the task structure 4320 size_t upper_offset = 4321 (char *)ub - (char *)task; // remember offset of ub in the task structure 4322 4323 KMP_DEBUG_ASSERT(tc == num_tasks * grainsize + extras); 4324 KMP_DEBUG_ASSERT(num_tasks > extras); 4325 KMP_DEBUG_ASSERT(num_tasks > 0); 4326 4327 // split the loop in two halves 4328 kmp_uint64 lb1, ub0, tc0, tc1, ext0, ext1; 4329 kmp_uint64 gr_size0 = grainsize; 4330 kmp_uint64 n_tsk0 = num_tasks >> 1; // num_tasks/2 to execute 4331 kmp_uint64 n_tsk1 = num_tasks - n_tsk0; // to schedule as a task 4332 if (n_tsk0 <= extras) { 4333 gr_size0++; // integrate extras into grainsize 4334 ext0 = 0; // no extra iters in 1st half 4335 ext1 = extras - n_tsk0; // remaining extras 4336 tc0 = gr_size0 * n_tsk0; 4337 tc1 = tc - tc0; 4338 } else { // n_tsk0 > extras 4339 ext1 = 0; // no extra iters in 2nd half 4340 ext0 = extras; 4341 tc1 = grainsize * n_tsk1; 4342 tc0 = tc - tc1; 4343 } 4344 ub0 = lower + st * (tc0 - 1); 4345 lb1 = ub0 + st; 4346 4347 // create pattern task for 2nd half of the loop 4348 next_task = __kmp_task_dup_alloc(thread, task); // duplicate the task 4349 // adjust lower bound (upper bound is not changed) for the 2nd half 4350 *(kmp_uint64 *)((char *)next_task + lower_offset) = lb1; 4351 if (ptask_dup != NULL) // construct fistprivates, etc. 4352 ptask_dup(next_task, task, 0); 4353 *ub = ub0; // adjust upper bound for the 1st half 4354 4355 // create auxiliary task for 2nd half of the loop 4356 kmp_task_t *new_task = 4357 __kmpc_omp_task_alloc(loc, gtid, 1, 3 * sizeof(void *), 4358 sizeof(__taskloop_params_t), &__kmp_taskloop_task); 4359 __taskloop_params_t *p = (__taskloop_params_t *)new_task->shareds; 4360 p->task = next_task; 4361 p->lb = (kmp_uint64 *)((char *)next_task + lower_offset); 4362 p->ub = (kmp_uint64 *)((char *)next_task + upper_offset); 4363 p->task_dup = task_dup; 4364 p->st = st; 4365 p->ub_glob = ub_glob; 4366 p->num_tasks = n_tsk1; 4367 p->grainsize = grainsize; 4368 p->extras = ext1; 4369 p->tc = tc1; 4370 p->num_t_min = num_t_min; 4371 #if OMPT_SUPPORT 4372 p->codeptr_ra = codeptr_ra; 4373 #endif 4374 4375 #if OMPT_SUPPORT 4376 // schedule new task with correct return address for OMPT events 4377 __kmp_omp_taskloop_task(NULL, gtid, new_task, codeptr_ra); 4378 #else 4379 __kmp_omp_task(gtid, new_task, true); // schedule new task 4380 #endif 4381 4382 // execute the 1st half of current subrange 4383 if (n_tsk0 > num_t_min) 4384 __kmp_taskloop_recur(loc, gtid, task, lb, ub, st, ub_glob, n_tsk0, gr_size0, 4385 ext0, tc0, num_t_min, 4386 #if OMPT_SUPPORT 4387 codeptr_ra, 4388 #endif 4389 task_dup); 4390 else 4391 __kmp_taskloop_linear(loc, gtid, task, lb, ub, st, ub_glob, n_tsk0, 4392 gr_size0, ext0, tc0, 4393 #if OMPT_SUPPORT 4394 codeptr_ra, 4395 #endif 4396 task_dup); 4397 4398 KA_TRACE(40, ("__kmpc_taskloop_recur(exit): T#%d\n", gtid)); 4399 } 4400 4401 /*! 4402 @ingroup TASKING 4403 @param loc Source location information 4404 @param gtid Global thread ID 4405 @param task Task structure 4406 @param if_val Value of the if clause 4407 @param lb Pointer to loop lower bound in task structure 4408 @param ub Pointer to loop upper bound in task structure 4409 @param st Loop stride 4410 @param nogroup Flag, 1 if no taskgroup needs to be added, 0 otherwise 4411 @param sched Schedule specified 0/1/2 for none/grainsize/num_tasks 4412 @param grainsize Schedule value if specified 4413 @param task_dup Tasks duplication routine 4414 4415 Execute the taskloop construct. 4416 */ 4417 void __kmpc_taskloop(ident_t *loc, int gtid, kmp_task_t *task, int if_val, 4418 kmp_uint64 *lb, kmp_uint64 *ub, kmp_int64 st, int nogroup, 4419 int sched, kmp_uint64 grainsize, void *task_dup) { 4420 kmp_taskdata_t *taskdata = KMP_TASK_TO_TASKDATA(task); 4421 KMP_DEBUG_ASSERT(task != NULL); 4422 4423 if (nogroup == 0) { 4424 #if OMPT_SUPPORT && OMPT_OPTIONAL 4425 OMPT_STORE_RETURN_ADDRESS(gtid); 4426 #endif 4427 __kmpc_taskgroup(loc, gtid); 4428 } 4429 4430 // ========================================================================= 4431 // calculate loop parameters 4432 kmp_taskloop_bounds_t task_bounds(task, lb, ub); 4433 kmp_uint64 tc; 4434 // compiler provides global bounds here 4435 kmp_uint64 lower = task_bounds.get_lb(); 4436 kmp_uint64 upper = task_bounds.get_ub(); 4437 kmp_uint64 ub_glob = upper; // global upper used to calc lastprivate flag 4438 kmp_uint64 num_tasks = 0, extras = 0; 4439 kmp_uint64 num_tasks_min = __kmp_taskloop_min_tasks; 4440 kmp_info_t *thread = __kmp_threads[gtid]; 4441 kmp_taskdata_t *current_task = thread->th.th_current_task; 4442 4443 KA_TRACE(20, ("__kmpc_taskloop: T#%d, task %p, lb %lld, ub %lld, st %lld, " 4444 "grain %llu(%d), dup %p\n", 4445 gtid, taskdata, lower, upper, st, grainsize, sched, task_dup)); 4446 4447 // compute trip count 4448 if (st == 1) { // most common case 4449 tc = upper - lower + 1; 4450 } else if (st < 0) { 4451 tc = (lower - upper) / (-st) + 1; 4452 } else { // st > 0 4453 tc = (upper - lower) / st + 1; 4454 } 4455 if (tc == 0) { 4456 KA_TRACE(20, ("__kmpc_taskloop(exit): T#%d zero-trip loop\n", gtid)); 4457 // free the pattern task and exit 4458 __kmp_task_start(gtid, task, current_task); 4459 // do not execute anything for zero-trip loop 4460 __kmp_task_finish<false>(gtid, task, current_task); 4461 return; 4462 } 4463 4464 #if OMPT_SUPPORT && OMPT_OPTIONAL 4465 ompt_team_info_t *team_info = __ompt_get_teaminfo(0, NULL); 4466 ompt_task_info_t *task_info = __ompt_get_task_info_object(0); 4467 if (ompt_enabled.ompt_callback_work) { 4468 ompt_callbacks.ompt_callback(ompt_callback_work)( 4469 ompt_work_taskloop, ompt_scope_begin, &(team_info->parallel_data), 4470 &(task_info->task_data), tc, OMPT_GET_RETURN_ADDRESS(0)); 4471 } 4472 #endif 4473 4474 if (num_tasks_min == 0) 4475 // TODO: can we choose better default heuristic? 4476 num_tasks_min = 4477 KMP_MIN(thread->th.th_team_nproc * 10, INITIAL_TASK_DEQUE_SIZE); 4478 4479 // compute num_tasks/grainsize based on the input provided 4480 switch (sched) { 4481 case 0: // no schedule clause specified, we can choose the default 4482 // let's try to schedule (team_size*10) tasks 4483 grainsize = thread->th.th_team_nproc * 10; 4484 KMP_FALLTHROUGH(); 4485 case 2: // num_tasks provided 4486 if (grainsize > tc) { 4487 num_tasks = tc; // too big num_tasks requested, adjust values 4488 grainsize = 1; 4489 extras = 0; 4490 } else { 4491 num_tasks = grainsize; 4492 grainsize = tc / num_tasks; 4493 extras = tc % num_tasks; 4494 } 4495 break; 4496 case 1: // grainsize provided 4497 if (grainsize > tc) { 4498 num_tasks = 1; // too big grainsize requested, adjust values 4499 grainsize = tc; 4500 extras = 0; 4501 } else { 4502 num_tasks = tc / grainsize; 4503 // adjust grainsize for balanced distribution of iterations 4504 grainsize = tc / num_tasks; 4505 extras = tc % num_tasks; 4506 } 4507 break; 4508 default: 4509 KMP_ASSERT2(0, "unknown scheduling of taskloop"); 4510 } 4511 KMP_DEBUG_ASSERT(tc == num_tasks * grainsize + extras); 4512 KMP_DEBUG_ASSERT(num_tasks > extras); 4513 KMP_DEBUG_ASSERT(num_tasks > 0); 4514 // ========================================================================= 4515 4516 // check if clause value first 4517 // Also require GOMP_taskloop to reduce to linear (taskdata->td_flags.native) 4518 if (if_val == 0) { // if(0) specified, mark task as serial 4519 taskdata->td_flags.task_serial = 1; 4520 taskdata->td_flags.tiedness = TASK_TIED; // AC: serial task cannot be untied 4521 // always start serial tasks linearly 4522 __kmp_taskloop_linear(loc, gtid, task, lb, ub, st, ub_glob, num_tasks, 4523 grainsize, extras, tc, 4524 #if OMPT_SUPPORT 4525 OMPT_GET_RETURN_ADDRESS(0), 4526 #endif 4527 task_dup); 4528 // !taskdata->td_flags.native => currently force linear spawning of tasks 4529 // for GOMP_taskloop 4530 } else if (num_tasks > num_tasks_min && !taskdata->td_flags.native) { 4531 KA_TRACE(20, ("__kmpc_taskloop: T#%d, go recursive: tc %llu, #tasks %llu" 4532 "(%lld), grain %llu, extras %llu\n", 4533 gtid, tc, num_tasks, num_tasks_min, grainsize, extras)); 4534 __kmp_taskloop_recur(loc, gtid, task, lb, ub, st, ub_glob, num_tasks, 4535 grainsize, extras, tc, num_tasks_min, 4536 #if OMPT_SUPPORT 4537 OMPT_GET_RETURN_ADDRESS(0), 4538 #endif 4539 task_dup); 4540 } else { 4541 KA_TRACE(20, ("__kmpc_taskloop: T#%d, go linear: tc %llu, #tasks %llu" 4542 "(%lld), grain %llu, extras %llu\n", 4543 gtid, tc, num_tasks, num_tasks_min, grainsize, extras)); 4544 __kmp_taskloop_linear(loc, gtid, task, lb, ub, st, ub_glob, num_tasks, 4545 grainsize, extras, tc, 4546 #if OMPT_SUPPORT 4547 OMPT_GET_RETURN_ADDRESS(0), 4548 #endif 4549 task_dup); 4550 } 4551 4552 #if OMPT_SUPPORT && OMPT_OPTIONAL 4553 if (ompt_enabled.ompt_callback_work) { 4554 ompt_callbacks.ompt_callback(ompt_callback_work)( 4555 ompt_work_taskloop, ompt_scope_end, &(team_info->parallel_data), 4556 &(task_info->task_data), tc, OMPT_GET_RETURN_ADDRESS(0)); 4557 } 4558 #endif 4559 4560 if (nogroup == 0) { 4561 #if OMPT_SUPPORT && OMPT_OPTIONAL 4562 OMPT_STORE_RETURN_ADDRESS(gtid); 4563 #endif 4564 __kmpc_end_taskgroup(loc, gtid); 4565 } 4566 KA_TRACE(20, ("__kmpc_taskloop(exit): T#%d\n", gtid)); 4567 } 4568 4569 #endif 4570