LLVM OpenMP
kmp_runtime.cpp
Go to the documentation of this file.
1/*
2 * kmp_runtime.cpp -- KPTS runtime support library
3 */
4
5//===----------------------------------------------------------------------===//
6//
7// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
8// See https://llvm.org/LICENSE.txt for license information.
9// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
10//
11//===----------------------------------------------------------------------===//
12
13#include "kmp.h"
14#include "kmp_affinity.h"
15#include "kmp_atomic.h"
16#include "kmp_device_env.h"
17#include "kmp_environment.h"
18#include "kmp_error.h"
19#include "kmp_i18n.h"
20#include "kmp_io.h"
21#include "kmp_itt.h"
22#include "kmp_settings.h"
23#include "kmp_stats.h"
24#include "kmp_str.h"
25#include "kmp_wait_release.h"
26#include "kmp_wrapper_getpid.h"
27#include "kmp_dispatch.h"
28#include "kmp_utils.h"
29#if KMP_USE_HIER_SCHED
30#include "kmp_dispatch_hier.h"
31#endif
32
33#if OMPT_SUPPORT
34#include "ompt-specific.h"
35#endif
36#if OMPD_SUPPORT
37#include "ompd-specific.h"
38#endif
39
40#if OMP_PROFILING_SUPPORT
41#include "llvm/Support/TimeProfiler.h"
42static char *ProfileTraceFile = nullptr;
43#endif
44
45/* these are temporary issues to be dealt with */
46#define KMP_USE_PRCTL 0
47
48#if KMP_OS_WINDOWS
49#include <process.h>
50#endif
51
52#ifndef KMP_USE_SHM
53// Windows and WASI do not need these include files as they don't use shared
54// memory.
55#else
56#include <sys/mman.h>
57#include <sys/stat.h>
58#include <fcntl.h>
59#define SHM_SIZE 1024
60#endif
61
62#if defined(KMP_GOMP_COMPAT)
63char const __kmp_version_alt_comp[] =
64 KMP_VERSION_PREFIX "alternative compiler support: yes";
65#endif /* defined(KMP_GOMP_COMPAT) */
66
68 KMP_VERSION_PREFIX "API version: 5.0 (201611)";
69
70#ifdef KMP_DEBUG
71char const __kmp_version_lock[] =
72 KMP_VERSION_PREFIX "lock type: run time selectable";
73#endif /* KMP_DEBUG */
74
75#define KMP_MIN(x, y) ((x) < (y) ? (x) : (y))
76
77/* ------------------------------------------------------------------------ */
78
79#if KMP_USE_MONITOR
81#endif
82
83/* Forward declarations */
84
85void __kmp_cleanup(void);
86
87static void __kmp_initialize_info(kmp_info_t *, kmp_team_t *, int tid,
88 int gtid);
89static void __kmp_initialize_team(kmp_team_t *team, int new_nproc,
90 kmp_internal_control_t *new_icvs,
91 ident_t *loc);
92#if KMP_AFFINITY_SUPPORTED
93static void __kmp_partition_places(kmp_team_t *team,
94 int update_master_only = 0);
95#endif
96static void __kmp_do_serial_initialize(void);
97#if ENABLE_LIBOMPTARGET
98static void __kmp_target_init(void);
99#endif // ENABLE_LIBOMPTARGET
100void __kmp_fork_barrier(int gtid, int tid);
101void __kmp_join_barrier(int gtid);
102void __kmp_setup_icv_copy(kmp_team_t *team, int new_nproc,
104
105#ifdef USE_LOAD_BALANCE
106static int __kmp_load_balance_nproc(kmp_root_t *root, int set_nproc);
107#endif
108
109static int __kmp_expand_threads(int nNeed);
110#if KMP_OS_WINDOWS
111static int __kmp_unregister_root_other_thread(int gtid);
112#endif
113static void __kmp_reap_thread(kmp_info_t *thread, int is_root);
115
116void __kmp_resize_dist_barrier(kmp_team_t *team, int old_nthreads,
117 int new_nthreads);
118void __kmp_add_threads_to_team(kmp_team_t *team, int new_nthreads);
119
121 int level) {
122 kmp_nested_nthreads_t *new_nested_nth =
124 sizeof(kmp_nested_nthreads_t));
125 int new_size = level + thr->th.th_set_nested_nth_sz;
126 new_nested_nth->nth = (int *)KMP_INTERNAL_MALLOC(new_size * sizeof(int));
127 for (int i = 0; i < level + 1; ++i)
128 new_nested_nth->nth[i] = 0;
129 for (int i = level + 1, j = 1; i < new_size; ++i, ++j)
130 new_nested_nth->nth[i] = thr->th.th_set_nested_nth[j];
131 new_nested_nth->size = new_nested_nth->used = new_size;
132 return new_nested_nth;
133}
134
135/* Calculate the identifier of the current thread */
136/* fast (and somewhat portable) way to get unique identifier of executing
137 thread. Returns KMP_GTID_DNE if we haven't been assigned a gtid. */
139 int i;
140 kmp_info_t **other_threads;
141 size_t stack_data;
142 char *stack_addr;
143 size_t stack_size;
144 char *stack_base;
145
146 KA_TRACE(
147 1000,
148 ("*** __kmp_get_global_thread_id: entering, nproc=%d all_nproc=%d\n",
150
151 /* JPH - to handle the case where __kmpc_end(0) is called immediately prior to
152 a parallel region, made it return KMP_GTID_DNE to force serial_initialize
153 by caller. Had to handle KMP_GTID_DNE at all call-sites, or else guarantee
154 __kmp_init_gtid for this to work. */
155
157 return KMP_GTID_DNE;
158
159#ifdef KMP_TDATA_GTID
160 if (TCR_4(__kmp_gtid_mode) >= 3) {
161 KA_TRACE(1000, ("*** __kmp_get_global_thread_id: using TDATA\n"));
162 return __kmp_gtid;
163 }
164#endif
165 if (TCR_4(__kmp_gtid_mode) >= 2) {
166 KA_TRACE(1000, ("*** __kmp_get_global_thread_id: using keyed TLS\n"));
168 }
169 KA_TRACE(1000, ("*** __kmp_get_global_thread_id: using internal alg.\n"));
170
171 stack_addr = (char *)&stack_data;
172 other_threads = __kmp_threads;
173
174 /* ATT: The code below is a source of potential bugs due to unsynchronized
175 access to __kmp_threads array. For example:
176 1. Current thread loads other_threads[i] to thr and checks it, it is
177 non-NULL.
178 2. Current thread is suspended by OS.
179 3. Another thread unregisters and finishes (debug versions of free()
180 may fill memory with something like 0xEF).
181 4. Current thread is resumed.
182 5. Current thread reads junk from *thr.
183 TODO: Fix it. --ln */
184
185 for (i = 0; i < __kmp_threads_capacity; i++) {
186
187 kmp_info_t *thr = (kmp_info_t *)TCR_SYNC_PTR(other_threads[i]);
188 if (!thr)
189 continue;
190
191 stack_size = (size_t)TCR_PTR(thr->th.th_info.ds.ds_stacksize);
192 stack_base = (char *)TCR_PTR(thr->th.th_info.ds.ds_stackbase);
193
194 /* stack grows down -- search through all of the active threads */
195
196 if (stack_addr <= stack_base) {
197 size_t stack_diff = stack_base - stack_addr;
198
199 if (stack_diff <= stack_size) {
200 /* The only way we can be closer than the allocated */
201 /* stack size is if we are running on this thread. */
202 // __kmp_gtid_get_specific can return negative value because this
203 // function can be called by thread destructor. However, before the
204 // thread destructor is called, the value of the corresponding
205 // thread-specific data will be reset to NULL.
208 return i;
209 }
210 }
211 }
212
213 /* get specific to try and determine our gtid */
214 KA_TRACE(1000,
215 ("*** __kmp_get_global_thread_id: internal alg. failed to find "
216 "thread, using TLS\n"));
218
219 /*fprintf( stderr, "=== %d\n", i ); */ /* GROO */
220
221 /* if we havn't been assigned a gtid, then return code */
222 if (i < 0)
223 return i;
224
225 // other_threads[i] can be nullptr at this point because the corresponding
226 // thread could have already been destructed. It can happen when this function
227 // is called in end library routine.
228 if (!TCR_SYNC_PTR(other_threads[i]))
229 return i;
230
231 /* dynamically updated stack window for uber threads to avoid get_specific
232 call */
233 if (!TCR_4(other_threads[i]->th.th_info.ds.ds_stackgrow)) {
234 KMP_FATAL(StackOverflow, i);
235 }
236
237 stack_base = (char *)other_threads[i]->th.th_info.ds.ds_stackbase;
238 if (stack_addr > stack_base) {
239 TCW_PTR(other_threads[i]->th.th_info.ds.ds_stackbase, stack_addr);
240 TCW_PTR(other_threads[i]->th.th_info.ds.ds_stacksize,
241 other_threads[i]->th.th_info.ds.ds_stacksize + stack_addr -
242 stack_base);
243 } else {
244 TCW_PTR(other_threads[i]->th.th_info.ds.ds_stacksize,
245 stack_base - stack_addr);
246 }
247
248 /* Reprint stack bounds for ubermaster since they have been refined */
249 if (__kmp_storage_map) {
250 char *stack_end = (char *)other_threads[i]->th.th_info.ds.ds_stackbase;
251 char *stack_beg = stack_end - other_threads[i]->th.th_info.ds.ds_stacksize;
252 __kmp_print_storage_map_gtid(i, stack_beg, stack_end,
253 other_threads[i]->th.th_info.ds.ds_stacksize,
254 "th_%d stack (refinement)", i);
255 }
256 return i;
257}
258
260 int gtid;
261
262 if (!__kmp_init_serial) {
263 gtid = KMP_GTID_DNE;
264 } else
265#ifdef KMP_TDATA_GTID
266 if (TCR_4(__kmp_gtid_mode) >= 3) {
267 KA_TRACE(1000, ("*** __kmp_get_global_thread_id_reg: using TDATA\n"));
268 gtid = __kmp_gtid;
269 } else
270#endif
271 if (TCR_4(__kmp_gtid_mode) >= 2) {
272 KA_TRACE(1000, ("*** __kmp_get_global_thread_id_reg: using keyed TLS\n"));
274 } else {
275 KA_TRACE(1000,
276 ("*** __kmp_get_global_thread_id_reg: using internal alg.\n"));
278 }
279
280 /* we must be a new uber master sibling thread */
281 if (gtid == KMP_GTID_DNE) {
282 KA_TRACE(10,
283 ("__kmp_get_global_thread_id_reg: Encountered new root thread. "
284 "Registering a new gtid.\n"));
286 if (!__kmp_init_serial) {
289 } else {
291 }
293 /*__kmp_printf( "+++ %d\n", gtid ); */ /* GROO */
294 }
295
296 KMP_DEBUG_ASSERT(gtid >= 0);
297
298 return gtid;
299}
300
301/* caller must hold forkjoin_lock */
303 int f;
304 char *stack_beg = NULL;
305 char *stack_end = NULL;
306 int gtid;
307
308 KA_TRACE(10, ("__kmp_check_stack_overlap: called\n"));
309 if (__kmp_storage_map) {
310 stack_end = (char *)th->th.th_info.ds.ds_stackbase;
311 stack_beg = stack_end - th->th.th_info.ds.ds_stacksize;
312
313 gtid = __kmp_gtid_from_thread(th);
314
315 if (gtid == KMP_GTID_MONITOR) {
317 gtid, stack_beg, stack_end, th->th.th_info.ds.ds_stacksize,
318 "th_%s stack (%s)", "mon",
319 (th->th.th_info.ds.ds_stackgrow) ? "initial" : "actual");
320 } else {
322 gtid, stack_beg, stack_end, th->th.th_info.ds.ds_stacksize,
323 "th_%d stack (%s)", gtid,
324 (th->th.th_info.ds.ds_stackgrow) ? "initial" : "actual");
325 }
326 }
327
328 /* No point in checking ubermaster threads since they use refinement and
329 * cannot overlap */
330 gtid = __kmp_gtid_from_thread(th);
331 if (__kmp_env_checks == TRUE && !KMP_UBER_GTID(gtid)) {
332 KA_TRACE(10,
333 ("__kmp_check_stack_overlap: performing extensive checking\n"));
334 if (stack_beg == NULL) {
335 stack_end = (char *)th->th.th_info.ds.ds_stackbase;
336 stack_beg = stack_end - th->th.th_info.ds.ds_stacksize;
337 }
338
339 for (f = 0; f < __kmp_threads_capacity; f++) {
341
342 if (f_th && f_th != th) {
343 char *other_stack_end =
344 (char *)TCR_PTR(f_th->th.th_info.ds.ds_stackbase);
345 char *other_stack_beg =
346 other_stack_end - (size_t)TCR_PTR(f_th->th.th_info.ds.ds_stacksize);
347 if ((stack_beg > other_stack_beg && stack_beg < other_stack_end) ||
348 (stack_end > other_stack_beg && stack_end < other_stack_end)) {
349
350 /* Print the other stack values before the abort */
353 -1, other_stack_beg, other_stack_end,
354 (size_t)TCR_PTR(f_th->th.th_info.ds.ds_stacksize),
355 "th_%d stack (overlapped)", __kmp_gtid_from_thread(f_th));
356
357 __kmp_fatal(KMP_MSG(StackOverlap), KMP_HNT(ChangeStackLimit),
359 }
360 }
361 }
362 }
363 KA_TRACE(10, ("__kmp_check_stack_overlap: returning\n"));
364}
365
366/* ------------------------------------------------------------------------ */
367
369 static int done = FALSE;
370
371 while (!done) {
373 }
374}
375
376#define MAX_MESSAGE 512
377
378void __kmp_print_storage_map_gtid(int gtid, void *p1, void *p2, size_t size,
379 char const *format, ...) {
380 char buffer[MAX_MESSAGE];
381 va_list ap;
382
383 va_start(ap, format);
384 KMP_SNPRINTF(buffer, sizeof(buffer), "OMP storage map: %p %p%8lu %s\n", p1,
385 p2, (unsigned long)size, format);
387 __kmp_vprintf(kmp_err, buffer, ap);
388#if KMP_PRINT_DATA_PLACEMENT
389 int node;
390 if (gtid >= 0) {
391 if (p1 <= p2 && (char *)p2 - (char *)p1 == size) {
393 node = __kmp_get_host_node(p1);
394 if (node < 0) /* doesn't work, so don't try this next time */
396 else {
397 char *last;
398 int lastNode;
399 int localProc = __kmp_get_cpu_from_gtid(gtid);
400
401 const int page_size = KMP_GET_PAGE_SIZE();
402
403 p1 = (void *)((size_t)p1 & ~((size_t)page_size - 1));
404 p2 = (void *)(((size_t)p2 - 1) & ~((size_t)page_size - 1));
405 if (localProc >= 0)
406 __kmp_printf_no_lock(" GTID %d localNode %d\n", gtid,
407 localProc >> 1);
408 else
409 __kmp_printf_no_lock(" GTID %d\n", gtid);
410#if KMP_USE_PRCTL
411 /* The more elaborate format is disabled for now because of the prctl
412 * hanging bug. */
413 do {
414 last = p1;
415 lastNode = node;
416 /* This loop collates adjacent pages with the same host node. */
417 do {
418 (char *)p1 += page_size;
419 } while (p1 <= p2 && (node = __kmp_get_host_node(p1)) == lastNode);
420 __kmp_printf_no_lock(" %p-%p memNode %d\n", last, (char *)p1 - 1,
421 lastNode);
422 } while (p1 <= p2);
423#else
424 __kmp_printf_no_lock(" %p-%p memNode %d\n", p1,
425 (char *)p1 + (page_size - 1),
426 __kmp_get_host_node(p1));
427 if (p1 < p2) {
428 __kmp_printf_no_lock(" %p-%p memNode %d\n", p2,
429 (char *)p2 + (page_size - 1),
430 __kmp_get_host_node(p2));
431 }
432#endif
433 }
434 }
435 } else
436 __kmp_printf_no_lock(" %s\n", KMP_I18N_STR(StorageMapWarning));
437 }
438#endif /* KMP_PRINT_DATA_PLACEMENT */
440
441 va_end(ap);
442}
443
444void __kmp_warn(char const *format, ...) {
445 char buffer[MAX_MESSAGE];
446 va_list ap;
447
449 return;
450 }
451
452 va_start(ap, format);
453
454 KMP_SNPRINTF(buffer, sizeof(buffer), "OMP warning: %s\n", format);
456 __kmp_vprintf(kmp_err, buffer, ap);
458
459 va_end(ap);
460}
461
463 // A failed assertion or fatal error raised from inside the abort path itself
464 // re-enters this function on the same thread. __kmp_exit_lock is not
465 // recursive, so re-acquiring it below would hang the process instead of
466 // terminating it. Terminate directly on re-entry.
467 static KMP_THREAD_LOCAL bool aborting = false;
468 if (aborting)
469 abort();
470 aborting = true;
471
472 // Later threads may stall here, but that's ok because abort() will kill them.
474
475 if (__kmp_debug_buf) {
477 }
478
479#if KMP_OS_WINDOWS
480 // Let other threads know of abnormal termination and prevent deadlock
481 // if abort happened during library initialization or shutdown
482 __kmp_global.g.g_abort = SIGABRT;
483
484 /* On Windows* OS by default abort() causes pop-up error box, which stalls
485 nightly testing. Unfortunately, we cannot reliably suppress pop-up error
486 boxes. _set_abort_behavior() works well, but this function is not
487 available in VS7 (this is not problem for DLL, but it is a problem for
488 static OpenMP RTL). SetErrorMode (and so, timelimit utility) does not
489 help, at least in some versions of MS C RTL.
490
491 It seems following sequence is the only way to simulate abort() and
492 avoid pop-up error box. */
493 raise(SIGABRT);
494 _exit(3); // Just in case, if signal ignored, exit anyway.
495#else
497 abort();
498#endif
499
502
503} // __kmp_abort_process
504
506 // TODO: Eliminate g_abort global variable and this function.
507 // In case of abort just call abort(), it will kill all the threads.
509} // __kmp_abort_thread
510
511/* Print out the storage map for the major kmp_info_t thread data structures
512 that are allocated together. */
513
514static void __kmp_print_thread_storage_map(kmp_info_t *thr, int gtid) {
515 __kmp_print_storage_map_gtid(gtid, thr, thr + 1, sizeof(kmp_info_t), "th_%d",
516 gtid);
517
518 __kmp_print_storage_map_gtid(gtid, &thr->th.th_info, &thr->th.th_team,
519 sizeof(kmp_desc_t), "th_%d.th_info", gtid);
520
521 __kmp_print_storage_map_gtid(gtid, &thr->th.th_local, &thr->th.th_pri_head,
522 sizeof(kmp_local_t), "th_%d.th_local", gtid);
523
525 gtid, &thr->th.th_bar[0], &thr->th.th_bar[bs_last_barrier],
526 sizeof(kmp_balign_t) * bs_last_barrier, "th_%d.th_bar", gtid);
527
528 __kmp_print_storage_map_gtid(gtid, &thr->th.th_bar[bs_plain_barrier],
529 &thr->th.th_bar[bs_plain_barrier + 1],
530 sizeof(kmp_balign_t), "th_%d.th_bar[plain]",
531 gtid);
532
534 &thr->th.th_bar[bs_forkjoin_barrier + 1],
535 sizeof(kmp_balign_t), "th_%d.th_bar[forkjoin]",
536 gtid);
537
538#if KMP_FAST_REDUCTION_BARRIER
540 &thr->th.th_bar[bs_reduction_barrier + 1],
541 sizeof(kmp_balign_t), "th_%d.th_bar[reduction]",
542 gtid);
543#endif // KMP_FAST_REDUCTION_BARRIER
544}
545
546/* Print out the storage map for the major kmp_team_t team data structures
547 that are allocated together. */
548
549static void __kmp_print_team_storage_map(const char *header, kmp_team_t *team,
550 int team_id, int num_thr) {
551 int num_disp_buff = team->t.t_max_nproc > 1 ? __kmp_dispatch_num_buffers : 2;
552 __kmp_print_storage_map_gtid(-1, team, team + 1, sizeof(kmp_team_t), "%s_%d",
553 header, team_id);
554
555 __kmp_print_storage_map_gtid(-1, &team->t.t_bar[0],
556 &team->t.t_bar[bs_last_barrier],
558 "%s_%d.t_bar", header, team_id);
559
561 &team->t.t_bar[bs_plain_barrier + 1],
562 sizeof(kmp_balign_team_t), "%s_%d.t_bar[plain]",
563 header, team_id);
564
566 &team->t.t_bar[bs_forkjoin_barrier + 1],
567 sizeof(kmp_balign_team_t),
568 "%s_%d.t_bar[forkjoin]", header, team_id);
569
570#if KMP_FAST_REDUCTION_BARRIER
572 &team->t.t_bar[bs_reduction_barrier + 1],
573 sizeof(kmp_balign_team_t),
574 "%s_%d.t_bar[reduction]", header, team_id);
575#endif // KMP_FAST_REDUCTION_BARRIER
576
578 -1, &team->t.t_dispatch[0], &team->t.t_dispatch[num_thr],
579 sizeof(kmp_disp_t) * num_thr, "%s_%d.t_dispatch", header, team_id);
580
582 -1, &team->t.t_threads[0], &team->t.t_threads[num_thr],
583 sizeof(kmp_info_t *) * num_thr, "%s_%d.t_threads", header, team_id);
584
585 __kmp_print_storage_map_gtid(-1, &team->t.t_disp_buffer[0],
586 &team->t.t_disp_buffer[num_disp_buff],
587 sizeof(dispatch_shared_info_t) * num_disp_buff,
588 "%s_%d.t_disp_buffer", header, team_id);
589}
590
599
600/* ------------------------------------------------------------------------ */
601
602#if ENABLE_LIBOMPTARGET
603static void __kmp_init_omptarget() {
604 __kmp_init_target_task();
605}
606#endif
607
608/* ------------------------------------------------------------------------ */
609
610#if KMP_DYNAMIC_LIB
611#if KMP_OS_WINDOWS
612
613BOOL WINAPI DllMain(HINSTANCE hInstDLL, DWORD fdwReason, LPVOID lpReserved) {
614 //__kmp_acquire_bootstrap_lock( &__kmp_initz_lock );
615
616 switch (fdwReason) {
617
618 case DLL_PROCESS_ATTACH:
619 KA_TRACE(10, ("DllMain: PROCESS_ATTACH\n"));
620
621 return TRUE;
622
623 case DLL_PROCESS_DETACH:
624 KA_TRACE(10, ("DllMain: PROCESS_DETACH T#%d\n", __kmp_gtid_get_specific()));
625
626 // According to Windows* documentation for DllMain entry point:
627 // for DLL_PROCESS_DETACH, lpReserved is used for telling the difference:
628 // lpReserved == NULL when FreeLibrary() is called,
629 // lpReserved != NULL when the process is terminated.
630 // When FreeLibrary() is called, worker threads remain alive. So the
631 // runtime's state is consistent and executing proper shutdown is OK.
632 // When the process is terminated, worker threads have exited or been
633 // forcefully terminated by the OS and only the shutdown thread remains.
634 // This can leave the runtime in an inconsistent state.
635 // Hence, only attempt proper cleanup when FreeLibrary() is called.
636 // Otherwise, rely on OS to reclaim resources.
637 if (lpReserved == NULL)
639
640 return TRUE;
641
642 case DLL_THREAD_ATTACH:
643 KA_TRACE(10, ("DllMain: THREAD_ATTACH\n"));
644
645 /* if we want to register new siblings all the time here call
646 * __kmp_get_gtid(); */
647 return TRUE;
648
649 case DLL_THREAD_DETACH:
650 KA_TRACE(10, ("DllMain: THREAD_DETACH T#%d\n", __kmp_gtid_get_specific()));
651
653 return TRUE;
654 }
655
656 return TRUE;
657}
658
659#endif /* KMP_OS_WINDOWS */
660#endif /* KMP_DYNAMIC_LIB */
661
662/* __kmp_parallel_deo -- Wait until it's our turn. */
663void __kmp_parallel_deo(int *gtid_ref, int *cid_ref, ident_t *loc_ref) {
664 int gtid = *gtid_ref;
665#ifdef BUILD_PARALLEL_ORDERED
666 kmp_team_t *team = __kmp_team_from_gtid(gtid);
667#endif /* BUILD_PARALLEL_ORDERED */
668
670 if (__kmp_threads[gtid]->th.th_root->r.r_active)
671#if KMP_USE_DYNAMIC_LOCK
672 __kmp_push_sync(gtid, ct_ordered_in_parallel, loc_ref, NULL, 0);
673#else
674 __kmp_push_sync(gtid, ct_ordered_in_parallel, loc_ref, NULL);
675#endif
676 }
677#ifdef BUILD_PARALLEL_ORDERED
678 if (!team->t.t_serialized) {
679 KMP_MB();
680 KMP_WAIT(&team->t.t_ordered.dt.t_value, __kmp_tid_from_gtid(gtid), KMP_EQ,
681 NULL);
682 KMP_MB();
683 }
684#endif /* BUILD_PARALLEL_ORDERED */
685}
686
687/* __kmp_parallel_dxo -- Signal the next task. */
688void __kmp_parallel_dxo(int *gtid_ref, int *cid_ref, ident_t *loc_ref) {
689 int gtid = *gtid_ref;
690#ifdef BUILD_PARALLEL_ORDERED
691 int tid = __kmp_tid_from_gtid(gtid);
692 kmp_team_t *team = __kmp_team_from_gtid(gtid);
693#endif /* BUILD_PARALLEL_ORDERED */
694
696 if (__kmp_threads[gtid]->th.th_root->r.r_active)
698 }
699#ifdef BUILD_PARALLEL_ORDERED
700 if (!team->t.t_serialized) {
701 KMP_MB(); /* Flush all pending memory write invalidates. */
702
703 /* use the tid of the next thread in this team */
704 /* TODO replace with general release procedure */
705 team->t.t_ordered.dt.t_value = ((tid + 1) % team->t.t_nproc);
706
707 KMP_MB(); /* Flush all pending memory write invalidates. */
708 }
709#endif /* BUILD_PARALLEL_ORDERED */
710}
711
712/* ------------------------------------------------------------------------ */
713/* The BARRIER for a SINGLE process section is always explicit */
714
715int __kmp_enter_single(int gtid, ident_t *id_ref, int push_ws) {
716 int status;
717 kmp_info_t *th;
718 kmp_team_t *team;
719
723
724 th = __kmp_threads[gtid];
725 team = th->th.th_team;
726 status = 0;
727
728 th->th.th_ident = id_ref;
729
730 if (team->t.t_serialized) {
731 status = 1;
732 } else {
733 kmp_int32 old_this = th->th.th_local.this_construct;
734
735 ++th->th.th_local.this_construct;
736 /* try to set team count to thread count--success means thread got the
737 single block */
738 /* TODO: Should this be acquire or release? */
739 if (team->t.t_construct == old_this) {
740 status = __kmp_atomic_compare_store_acq(&team->t.t_construct, old_this,
741 th->th.th_local.this_construct);
742 }
743#if USE_ITT_BUILD
744 if (__itt_metadata_add_ptr && __kmp_forkjoin_frames_mode == 3 &&
745 KMP_MASTER_GTID(gtid) && th->th.th_teams_microtask == NULL &&
746 team->t.t_active_level == 1) {
747 // Only report metadata by primary thread of active team at level 1
748 __kmp_itt_metadata_single(id_ref);
749 }
750#endif /* USE_ITT_BUILD */
751 }
752
754 if (status && push_ws) {
755 __kmp_push_workshare(gtid, ct_psingle, id_ref);
756 } else {
757 __kmp_check_workshare(gtid, ct_psingle, id_ref);
758 }
759 }
760#if USE_ITT_BUILD
761 if (status) {
762 __kmp_itt_single_start(gtid);
763 }
764#endif /* USE_ITT_BUILD */
765 return status;
766}
767
768void __kmp_exit_single(int gtid) {
769#if USE_ITT_BUILD
770 __kmp_itt_single_end(gtid);
771#endif /* USE_ITT_BUILD */
773 __kmp_pop_workshare(gtid, ct_psingle, NULL);
774}
775
776/* determine if we can go parallel or must use a serialized parallel region and
777 * how many threads we can use
778 * set_nproc is the number of threads requested for the team
779 * returns 0 if we should serialize or only use one thread,
780 * otherwise the number of threads to use
781 * The forkjoin lock is held by the caller. */
782static int __kmp_reserve_threads(kmp_root_t *root, kmp_team_t *parent_team,
783 int master_tid, int set_nthreads,
784 int enter_teams) {
785 int capacity;
786 int new_nthreads;
788 KMP_DEBUG_ASSERT(root && parent_team);
789 kmp_info_t *this_thr = parent_team->t.t_threads[master_tid];
790
791 // If dyn-var is set, dynamically adjust the number of desired threads,
792 // according to the method specified by dynamic_mode.
793 new_nthreads = set_nthreads;
794 if (!get__dynamic_2(parent_team, master_tid)) {
795 ;
796 }
797#ifdef USE_LOAD_BALANCE
798 else if (__kmp_global.g.g_dynamic_mode == dynamic_load_balance) {
799 new_nthreads = __kmp_load_balance_nproc(root, set_nthreads);
800 if (new_nthreads == 1) {
801 KC_TRACE(10, ("__kmp_reserve_threads: T#%d load balance reduced "
802 "reservation to 1 thread\n",
803 master_tid));
804 return 1;
805 }
806 if (new_nthreads < set_nthreads) {
807 KC_TRACE(10, ("__kmp_reserve_threads: T#%d load balance reduced "
808 "reservation to %d threads\n",
809 master_tid, new_nthreads));
810 }
811 }
812#endif /* USE_LOAD_BALANCE */
813 else if (__kmp_global.g.g_dynamic_mode == dynamic_thread_limit) {
814 new_nthreads = __kmp_avail_proc - __kmp_nth +
815 (root->r.r_active ? 1 : root->r.r_hot_team->t.t_nproc);
816 if (new_nthreads <= 1) {
817 KC_TRACE(10, ("__kmp_reserve_threads: T#%d thread limit reduced "
818 "reservation to 1 thread\n",
819 master_tid));
820 return 1;
821 }
822 if (new_nthreads < set_nthreads) {
823 KC_TRACE(10, ("__kmp_reserve_threads: T#%d thread limit reduced "
824 "reservation to %d threads\n",
825 master_tid, new_nthreads));
826 } else {
827 new_nthreads = set_nthreads;
828 }
829 } else if (__kmp_global.g.g_dynamic_mode == dynamic_random) {
830 if (set_nthreads > 2) {
831 new_nthreads = __kmp_get_random(parent_team->t.t_threads[master_tid]);
832 new_nthreads = (new_nthreads % set_nthreads) + 1;
833 if (new_nthreads == 1) {
834 KC_TRACE(10, ("__kmp_reserve_threads: T#%d dynamic random reduced "
835 "reservation to 1 thread\n",
836 master_tid));
837 return 1;
838 }
839 if (new_nthreads < set_nthreads) {
840 KC_TRACE(10, ("__kmp_reserve_threads: T#%d dynamic random reduced "
841 "reservation to %d threads\n",
842 master_tid, new_nthreads));
843 }
844 }
845 } else {
846 KMP_ASSERT(0);
847 }
848
849 // Respect KMP_ALL_THREADS/KMP_DEVICE_THREAD_LIMIT.
850 if (__kmp_nth + new_nthreads -
851 (root->r.r_active ? 1 : root->r.r_hot_team->t.t_nproc) >
853 int tl_nthreads = __kmp_max_nth - __kmp_nth +
854 (root->r.r_active ? 1 : root->r.r_hot_team->t.t_nproc);
855 if (tl_nthreads <= 0) {
856 tl_nthreads = 1;
857 }
858
859 // If dyn-var is false, emit a 1-time warning.
860 if (!get__dynamic_2(parent_team, master_tid) && (!__kmp_reserve_warn)) {
863 KMP_MSG(CantFormThrTeam, set_nthreads, tl_nthreads),
864 KMP_HNT(Unset_ALL_THREADS), __kmp_msg_null);
865 }
866 if (tl_nthreads == 1) {
867 KC_TRACE(10, ("__kmp_reserve_threads: T#%d KMP_DEVICE_THREAD_LIMIT "
868 "reduced reservation to 1 thread\n",
869 master_tid));
870 return 1;
871 }
872 KC_TRACE(10, ("__kmp_reserve_threads: T#%d KMP_DEVICE_THREAD_LIMIT reduced "
873 "reservation to %d threads\n",
874 master_tid, tl_nthreads));
875 new_nthreads = tl_nthreads;
876 }
877
878 // Respect OMP_THREAD_LIMIT
879 int cg_nthreads = this_thr->th.th_cg_roots->cg_nthreads;
880 int max_cg_threads = this_thr->th.th_cg_roots->cg_thread_limit;
881 if (cg_nthreads + new_nthreads -
882 (root->r.r_active ? 1 : root->r.r_hot_team->t.t_nproc) >
883 max_cg_threads) {
884 int tl_nthreads = max_cg_threads - cg_nthreads +
885 (root->r.r_active ? 1 : root->r.r_hot_team->t.t_nproc);
886 if (tl_nthreads <= 0) {
887 tl_nthreads = 1;
888 }
889
890 // If dyn-var is false, emit a 1-time warning.
891 if (!get__dynamic_2(parent_team, master_tid) && (!__kmp_reserve_warn)) {
894 KMP_MSG(CantFormThrTeam, set_nthreads, tl_nthreads),
895 KMP_HNT(Unset_ALL_THREADS), __kmp_msg_null);
896 }
897 if (tl_nthreads == 1) {
898 KC_TRACE(10, ("__kmp_reserve_threads: T#%d OMP_THREAD_LIMIT "
899 "reduced reservation to 1 thread\n",
900 master_tid));
901 return 1;
902 }
903 KC_TRACE(10, ("__kmp_reserve_threads: T#%d OMP_THREAD_LIMIT reduced "
904 "reservation to %d threads\n",
905 master_tid, tl_nthreads));
906 new_nthreads = tl_nthreads;
907 }
908
909 // Check if the threads array is large enough, or needs expanding.
910 // See comment in __kmp_register_root() about the adjustment if
911 // __kmp_threads[0] == NULL.
912 capacity = __kmp_threads_capacity;
913 if (TCR_PTR(__kmp_threads[0]) == NULL) {
914 --capacity;
915 }
916 // If it is not for initializing the hidden helper team, we need to take
917 // __kmp_hidden_helper_threads_num out of the capacity because it is included
918 // in __kmp_threads_capacity.
921 }
922 if (__kmp_nth + new_nthreads -
923 (root->r.r_active ? 1 : root->r.r_hot_team->t.t_nproc) >
924 capacity) {
925 // Expand the threads array.
926 int slotsRequired = __kmp_nth + new_nthreads -
927 (root->r.r_active ? 1 : root->r.r_hot_team->t.t_nproc) -
928 capacity;
929 int slotsAdded = __kmp_expand_threads(slotsRequired);
930 if (slotsAdded < slotsRequired) {
931 // The threads array was not expanded enough.
932 new_nthreads -= (slotsRequired - slotsAdded);
933 KMP_ASSERT(new_nthreads >= 1);
934
935 // If dyn-var is false, emit a 1-time warning.
936 if (!get__dynamic_2(parent_team, master_tid) && (!__kmp_reserve_warn)) {
938 if (__kmp_tp_cached) {
940 KMP_MSG(CantFormThrTeam, set_nthreads, new_nthreads),
941 KMP_HNT(Set_ALL_THREADPRIVATE, __kmp_tp_capacity),
942 KMP_HNT(PossibleSystemLimitOnThreads), __kmp_msg_null);
943 } else {
945 KMP_MSG(CantFormThrTeam, set_nthreads, new_nthreads),
946 KMP_HNT(SystemLimitOnThreads), __kmp_msg_null);
947 }
948 }
949 }
950 }
951
952#ifdef KMP_DEBUG
953 if (new_nthreads == 1) {
954 KC_TRACE(10,
955 ("__kmp_reserve_threads: T#%d serializing team after reclaiming "
956 "dead roots and rechecking; requested %d threads\n",
957 __kmp_get_gtid(), set_nthreads));
958 } else {
959 KC_TRACE(10, ("__kmp_reserve_threads: T#%d allocating %d threads; requested"
960 " %d threads\n",
961 __kmp_get_gtid(), new_nthreads, set_nthreads));
962 }
963#endif // KMP_DEBUG
964
965 if (this_thr->th.th_nt_strict && new_nthreads < set_nthreads) {
966 __kmpc_error(this_thr->th.th_nt_loc, this_thr->th.th_nt_sev,
967 this_thr->th.th_nt_msg);
968 }
969 return new_nthreads;
970}
971
972/* Allocate threads from the thread pool and assign them to the new team. We are
973 assured that there are enough threads available, because we checked on that
974 earlier within critical section forkjoin */
976 kmp_info_t *master_th, int master_gtid,
977 int fork_teams_workers) {
978 int i;
979 int use_hot_team;
980
981 KA_TRACE(10, ("__kmp_fork_team_threads: new_nprocs = %d\n", team->t.t_nproc));
982 KMP_DEBUG_ASSERT(master_gtid == __kmp_get_gtid());
983 KMP_MB();
984
985 /* first, let's setup the primary thread */
986 master_th->th.th_info.ds.ds_tid = 0;
987 master_th->th.th_team = team;
988 master_th->th.th_team_nproc = team->t.t_nproc;
989 master_th->th.th_team_master = master_th;
990 master_th->th.th_team_serialized = FALSE;
991 master_th->th.th_dispatch = &team->t.t_dispatch[0];
992
993 /* make sure we are not the optimized hot team */
994 use_hot_team = 0;
995 kmp_hot_team_ptr_t *hot_teams = master_th->th.th_hot_teams;
996 if (hot_teams) { // hot teams array is not allocated if
997 // KMP_HOT_TEAMS_MAX_LEVEL=0
998 int level = team->t.t_active_level - 1; // index in array of hot teams
999 if (master_th->th.th_teams_microtask) { // are we inside the teams?
1000 if (master_th->th.th_teams_size.nteams > 1) {
1001 ++level; // level was not increased in teams construct for
1002 // team_of_masters
1003 }
1004 if (team->t.t_pkfn != (microtask_t)__kmp_teams_master &&
1005 master_th->th.th_teams_level == team->t.t_level) {
1006 ++level; // level was not increased in teams construct for
1007 // team_of_workers before the parallel
1008 } // team->t.t_level will be increased inside parallel
1009 }
1011 if (hot_teams[level].hot_team) {
1012 // hot team has already been allocated for given level
1013 KMP_DEBUG_ASSERT(hot_teams[level].hot_team == team);
1014 use_hot_team = 1; // the team is ready to use
1015 } else {
1016 use_hot_team = 0; // AC: threads are not allocated yet
1017 hot_teams[level].hot_team = team; // remember new hot team
1018 hot_teams[level].hot_team_nth = team->t.t_nproc;
1019 }
1020 } else {
1021 use_hot_team = 0;
1022 }
1023 }
1024 if (!use_hot_team) {
1025
1026 /* install the primary thread */
1027 team->t.t_threads[0] = master_th;
1028 __kmp_initialize_info(master_th, team, 0, master_gtid);
1029
1030 /* now, install the worker threads */
1031 for (i = 1; i < team->t.t_nproc; i++) {
1032
1033 /* fork or reallocate a new thread and install it in team */
1034 kmp_info_t *thr = __kmp_allocate_thread(root, team, i);
1035 team->t.t_threads[i] = thr;
1036 KMP_DEBUG_ASSERT(thr);
1037 KMP_DEBUG_ASSERT(thr->th.th_team == team);
1038 /* align team and thread arrived states */
1039 KA_TRACE(20, ("__kmp_fork_team_threads: T#%d(%d:%d) init arrived "
1040 "T#%d(%d:%d) join =%llu, plain=%llu\n",
1041 __kmp_gtid_from_tid(0, team), team->t.t_id, 0,
1042 __kmp_gtid_from_tid(i, team), team->t.t_id, i,
1043 team->t.t_bar[bs_forkjoin_barrier].b_arrived,
1044 team->t.t_bar[bs_plain_barrier].b_arrived));
1045 thr->th.th_teams_microtask = master_th->th.th_teams_microtask;
1046 thr->th.th_teams_level = master_th->th.th_teams_level;
1047 thr->th.th_teams_size = master_th->th.th_teams_size;
1048 { // Initialize threads' barrier data.
1049 int b;
1050 kmp_balign_t *balign = team->t.t_threads[i]->th.th_bar;
1051 for (b = 0; b < bs_last_barrier; ++b) {
1052 balign[b].bb.b_arrived = team->t.t_bar[b].b_arrived;
1053 KMP_DEBUG_ASSERT(balign[b].bb.wait_flag != KMP_BARRIER_PARENT_FLAG);
1054#if USE_DEBUGGER
1055 balign[b].bb.b_worker_arrived = team->t.t_bar[b].b_team_arrived;
1056#endif
1057 }
1058 }
1059 }
1060
1061#if KMP_AFFINITY_SUPPORTED
1062 // Do not partition the places list for teams construct workers who
1063 // haven't actually been forked to do real work yet. This partitioning
1064 // will take place in the parallel region nested within the teams construct.
1065 if (!fork_teams_workers) {
1066 __kmp_partition_places(team);
1067 }
1068#endif
1069
1070 if (team->t.t_nproc > 1 &&
1072 team->t.b->update_num_threads(team->t.t_nproc);
1073 __kmp_add_threads_to_team(team, team->t.t_nproc);
1074 }
1075 }
1076
1077 // Take care of primary thread's task state
1079 if (use_hot_team) {
1080 KMP_DEBUG_ASSERT_TASKTEAM_INVARIANT(team->t.t_parent, master_th);
1081 KA_TRACE(
1082 20,
1083 ("__kmp_fork_team_threads: Primary T#%d pushing task_team %p / team "
1084 "%p, new task_team %p / team %p\n",
1085 __kmp_gtid_from_thread(master_th), master_th->th.th_task_team,
1086 team->t.t_parent, team->t.t_task_team[master_th->th.th_task_state],
1087 team));
1088
1089 // Store primary thread's current task state on new team
1090 KMP_CHECK_UPDATE(team->t.t_primary_task_state,
1091 master_th->th.th_task_state);
1092
1093 // Restore primary thread's task state to hot team's state
1094 // by using thread 1's task state
1095 if (team->t.t_nproc > 1) {
1096 KMP_DEBUG_ASSERT(team->t.t_threads[1]->th.th_task_state == 0 ||
1097 team->t.t_threads[1]->th.th_task_state == 1);
1098 KMP_CHECK_UPDATE(master_th->th.th_task_state,
1099 team->t.t_threads[1]->th.th_task_state);
1100 } else {
1101 master_th->th.th_task_state = 0;
1102 }
1103 } else {
1104 // Store primary thread's current task_state on new team
1105 KMP_CHECK_UPDATE(team->t.t_primary_task_state,
1106 master_th->th.th_task_state);
1107 // Are not using hot team, so set task state to 0.
1108 master_th->th.th_task_state = 0;
1109 }
1110 }
1111
1112 if (__kmp_display_affinity && team->t.t_display_affinity != 1) {
1113 for (i = 0; i < team->t.t_nproc; i++) {
1114 kmp_info_t *thr = team->t.t_threads[i];
1115 if (thr->th.th_prev_num_threads != team->t.t_nproc ||
1116 thr->th.th_prev_level != team->t.t_level) {
1117 team->t.t_display_affinity = 1;
1118 break;
1119 }
1120 }
1121 }
1122
1123 KMP_MB();
1124}
1125
1126#if KMP_ARCH_X86 || KMP_ARCH_X86_64
1127// Propagate any changes to the floating point control registers out to the team
1128// We try to avoid unnecessary writes to the relevant cache line in the team
1129// structure, so we don't make changes unless they are needed.
1130inline static void propagateFPControl(kmp_team_t *team) {
1131 if (__kmp_inherit_fp_control) {
1132 kmp_int16 x87_fpu_control_word;
1133 kmp_uint32 mxcsr;
1134
1135 // Get primary thread's values of FPU control flags (both X87 and vector)
1136 __kmp_store_x87_fpu_control_word(&x87_fpu_control_word);
1137 __kmp_store_mxcsr(&mxcsr);
1138 mxcsr &= KMP_X86_MXCSR_MASK;
1139
1140 // There is no point looking at t_fp_control_saved here.
1141 // If it is TRUE, we still have to update the values if they are different
1142 // from those we now have. If it is FALSE we didn't save anything yet, but
1143 // our objective is the same. We have to ensure that the values in the team
1144 // are the same as those we have.
1145 // So, this code achieves what we need whether or not t_fp_control_saved is
1146 // true. By checking whether the value needs updating we avoid unnecessary
1147 // writes that would put the cache-line into a written state, causing all
1148 // threads in the team to have to read it again.
1149 KMP_CHECK_UPDATE(team->t.t_x87_fpu_control_word, x87_fpu_control_word);
1150 KMP_CHECK_UPDATE(team->t.t_mxcsr, mxcsr);
1151 // Although we don't use this value, other code in the runtime wants to know
1152 // whether it should restore them. So we must ensure it is correct.
1153 KMP_CHECK_UPDATE(team->t.t_fp_control_saved, TRUE);
1154 } else {
1155 // Similarly here. Don't write to this cache-line in the team structure
1156 // unless we have to.
1157 KMP_CHECK_UPDATE(team->t.t_fp_control_saved, FALSE);
1158 }
1159}
1160
1161// Do the opposite, setting the hardware registers to the updated values from
1162// the team.
1163inline static void updateHWFPControl(kmp_team_t *team) {
1164 if (__kmp_inherit_fp_control && team->t.t_fp_control_saved) {
1165 // Only reset the fp control regs if they have been changed in the team.
1166 // the parallel region that we are exiting.
1167 kmp_int16 x87_fpu_control_word;
1168 kmp_uint32 mxcsr;
1169 __kmp_store_x87_fpu_control_word(&x87_fpu_control_word);
1170 __kmp_store_mxcsr(&mxcsr);
1171 mxcsr &= KMP_X86_MXCSR_MASK;
1172
1173 if (team->t.t_x87_fpu_control_word != x87_fpu_control_word) {
1174 __kmp_clear_x87_fpu_status_word();
1175 __kmp_load_x87_fpu_control_word(&team->t.t_x87_fpu_control_word);
1176 }
1177
1178 if (team->t.t_mxcsr != mxcsr) {
1179 __kmp_load_mxcsr(&team->t.t_mxcsr);
1180 }
1181 }
1182}
1183#else
1184#define propagateFPControl(x) ((void)0)
1185#define updateHWFPControl(x) ((void)0)
1186#endif /* KMP_ARCH_X86 || KMP_ARCH_X86_64 */
1187
1188static void __kmp_alloc_argv_entries(int argc, kmp_team_t *team,
1189 int realloc); // forward declaration
1190
1191/* Run a parallel region that has been serialized, so runs only in a team of the
1192 single primary thread. */
1194 kmp_info_t *this_thr;
1195 kmp_team_t *serial_team;
1196
1197 KC_TRACE(10, ("__kmpc_serialized_parallel: called by T#%d\n", global_tid));
1198
1199 /* Skip all this code for autopar serialized loops since it results in
1200 unacceptable overhead */
1201 if (loc != NULL && (loc->flags & KMP_IDENT_AUTOPAR))
1202 return;
1203
1207
1208 this_thr = __kmp_threads[global_tid];
1209 serial_team = this_thr->th.th_serial_team;
1210
1211 /* utilize the serialized team held by this thread */
1212 KMP_DEBUG_ASSERT(serial_team);
1213 KMP_MB();
1214
1215 kmp_proc_bind_t proc_bind = this_thr->th.th_set_proc_bind;
1216 if (this_thr->th.th_current_task->td_icvs.proc_bind == proc_bind_false) {
1217 proc_bind = proc_bind_false;
1218 } else if (proc_bind == proc_bind_default) {
1219 // No proc_bind clause was specified, so use the current value
1220 // of proc-bind-var for this parallel region.
1221 proc_bind = this_thr->th.th_current_task->td_icvs.proc_bind;
1222 }
1223 // Reset for next parallel region
1224 this_thr->th.th_set_proc_bind = proc_bind_default;
1225
1226 // OpenMP 6.0 12.1.2 requires the num_threads 'strict' modifier to also have
1227 // effect when parallel execution is disabled by a corresponding if clause
1228 // attached to the parallel directive.
1229 if (this_thr->th.th_nt_strict && this_thr->th.th_set_nproc > 1)
1230 __kmpc_error(this_thr->th.th_nt_loc, this_thr->th.th_nt_sev,
1231 this_thr->th.th_nt_msg);
1232 // Reset num_threads for next parallel region
1233 this_thr->th.th_set_nproc = 0;
1234
1235#if OMPT_SUPPORT
1236 ompt_data_t ompt_parallel_data = ompt_data_none;
1237 void *codeptr = OMPT_LOAD_RETURN_ADDRESS(global_tid);
1238 if (ompt_enabled.enabled &&
1239 this_thr->th.ompt_thread_info.state != ompt_state_overhead) {
1240
1241 ompt_task_info_t *parent_task_info;
1242 parent_task_info = OMPT_CUR_TASK_INFO(this_thr);
1243
1244 parent_task_info->frame.enter_frame.ptr = OMPT_GET_FRAME_ADDRESS(0);
1245 if (ompt_enabled.ompt_callback_parallel_begin) {
1246 int team_size = 1;
1247
1248 ompt_callbacks.ompt_callback(ompt_callback_parallel_begin)(
1249 &(parent_task_info->task_data), &(parent_task_info->frame),
1250 &ompt_parallel_data, team_size,
1251 ompt_parallel_invoker_program | ompt_parallel_team, codeptr);
1252 }
1253 }
1254#endif // OMPT_SUPPORT
1255
1256 if (this_thr->th.th_team != serial_team) {
1257 // Nested level will be an index in the nested nthreads array
1258 int level = this_thr->th.th_team->t.t_level;
1259
1260 if (serial_team->t.t_serialized) {
1261 /* this serial team was already used
1262 TODO increase performance by making this locks more specific */
1263 kmp_team_t *new_team;
1264
1266
1267 new_team = __kmp_allocate_team(
1268 this_thr->th.th_root, 1, 1,
1269#if OMPT_SUPPORT
1270 ompt_parallel_data,
1271#endif
1272 proc_bind, &this_thr->th.th_current_task->td_icvs, 0, NULL);
1274 KMP_ASSERT(new_team);
1275
1276 /* setup new serialized team and install it */
1277 new_team->t.t_threads[0] = this_thr;
1278 new_team->t.t_parent = this_thr->th.th_team;
1279 serial_team = new_team;
1280 this_thr->th.th_serial_team = serial_team;
1281
1282 KF_TRACE(
1283 10,
1284 ("__kmpc_serialized_parallel: T#%d allocated new serial team %p\n",
1285 global_tid, serial_team));
1286
1287 /* TODO the above breaks the requirement that if we run out of resources,
1288 then we can still guarantee that serialized teams are ok, since we may
1289 need to allocate a new one */
1290 } else {
1291 KF_TRACE(
1292 10,
1293 ("__kmpc_serialized_parallel: T#%d reusing cached serial team %p\n",
1294 global_tid, serial_team));
1295 }
1296
1297 /* we have to initialize this serial team */
1298 KMP_DEBUG_ASSERT(serial_team->t.t_threads);
1299 KMP_DEBUG_ASSERT(serial_team->t.t_threads[0] == this_thr);
1300 KMP_DEBUG_ASSERT(this_thr->th.th_team != serial_team);
1301 serial_team->t.t_ident = loc;
1302 serial_team->t.t_serialized = 1;
1303 serial_team->t.t_nproc = 1;
1304 serial_team->t.t_parent = this_thr->th.th_team;
1305 if (this_thr->th.th_team->t.t_nested_nth)
1306 serial_team->t.t_nested_nth = this_thr->th.th_team->t.t_nested_nth;
1307 else
1308 serial_team->t.t_nested_nth = &__kmp_nested_nth;
1309 // Save previous team's task state on serial team structure
1310 serial_team->t.t_primary_task_state = this_thr->th.th_task_state;
1311 serial_team->t.t_sched.sched = this_thr->th.th_team->t.t_sched.sched;
1312 this_thr->th.th_team = serial_team;
1313 serial_team->t.t_master_tid = this_thr->th.th_info.ds.ds_tid;
1314
1315 KF_TRACE(10, ("__kmpc_serialized_parallel: T#%d curtask=%p\n", global_tid,
1316 this_thr->th.th_current_task));
1317 KMP_ASSERT(this_thr->th.th_current_task->td_flags.executing == 1);
1318 this_thr->th.th_current_task->td_flags.executing = 0;
1319
1320 __kmp_push_current_task_to_thread(this_thr, serial_team, 0);
1321
1322 /* TODO: GEH: do ICVs work for nested serialized teams? Don't we need an
1323 implicit task for each serialized task represented by
1324 team->t.t_serialized? */
1325 copy_icvs(&this_thr->th.th_current_task->td_icvs,
1326 &this_thr->th.th_current_task->td_parent->td_icvs);
1327
1328 // Thread value exists in the nested nthreads array for the next nested
1329 // level
1331 if (this_thr->th.th_team->t.t_nested_nth)
1332 nested_nth = this_thr->th.th_team->t.t_nested_nth;
1333 if (nested_nth->used && (level + 1 < nested_nth->used)) {
1334 this_thr->th.th_current_task->td_icvs.nproc = nested_nth->nth[level + 1];
1335 }
1336
1337 if (__kmp_nested_proc_bind.used &&
1338 (level + 1 < __kmp_nested_proc_bind.used)) {
1339 this_thr->th.th_current_task->td_icvs.proc_bind =
1340 __kmp_nested_proc_bind.bind_types[level + 1];
1341 }
1342
1343#if USE_DEBUGGER
1344 serial_team->t.t_pkfn = (microtask_t)(~0); // For the debugger.
1345#endif
1346 this_thr->th.th_info.ds.ds_tid = 0;
1347
1348 /* set thread cache values */
1349 this_thr->th.th_team_nproc = 1;
1350 this_thr->th.th_team_master = this_thr;
1351 this_thr->th.th_team_serialized = 1;
1352 this_thr->th.th_task_team = NULL;
1353 this_thr->th.th_task_state = 0;
1354
1355 serial_team->t.t_level = serial_team->t.t_parent->t.t_level + 1;
1356 serial_team->t.t_active_level = serial_team->t.t_parent->t.t_active_level;
1357 serial_team->t.t_def_allocator = this_thr->th.th_def_allocator; // save
1358
1359 propagateFPControl(serial_team);
1360
1361 /* check if we need to allocate dispatch buffers stack */
1362 KMP_DEBUG_ASSERT(serial_team->t.t_dispatch);
1363 if (!serial_team->t.t_dispatch->th_disp_buffer) {
1364 serial_team->t.t_dispatch->th_disp_buffer =
1366 sizeof(dispatch_private_info_t));
1367 }
1368 this_thr->th.th_dispatch = serial_team->t.t_dispatch;
1369
1370 KMP_MB();
1371
1372 } else {
1373 /* this serialized team is already being used,
1374 * that's fine, just add another nested level */
1375 KMP_DEBUG_ASSERT(this_thr->th.th_team == serial_team);
1376 KMP_DEBUG_ASSERT(serial_team->t.t_threads);
1377 KMP_DEBUG_ASSERT(serial_team->t.t_threads[0] == this_thr);
1378 ++serial_team->t.t_serialized;
1379 this_thr->th.th_team_serialized = serial_team->t.t_serialized;
1380
1381 // Nested level will be an index in the nested nthreads array
1382 int level = this_thr->th.th_team->t.t_level;
1383 // Thread value exists in the nested nthreads array for the next nested
1384 // level
1385
1387 if (serial_team->t.t_nested_nth)
1388 nested_nth = serial_team->t.t_nested_nth;
1389 if (nested_nth->used && (level + 1 < nested_nth->used)) {
1390 this_thr->th.th_current_task->td_icvs.nproc = nested_nth->nth[level + 1];
1391 }
1392
1393 serial_team->t.t_level++;
1394 KF_TRACE(10, ("__kmpc_serialized_parallel: T#%d increasing nesting level "
1395 "of serial team %p to %d\n",
1396 global_tid, serial_team, serial_team->t.t_level));
1397
1398 /* allocate/push dispatch buffers stack */
1399 KMP_DEBUG_ASSERT(serial_team->t.t_dispatch);
1400 {
1401 dispatch_private_info_t *disp_buffer =
1403 sizeof(dispatch_private_info_t));
1404 disp_buffer->next = serial_team->t.t_dispatch->th_disp_buffer;
1405 serial_team->t.t_dispatch->th_disp_buffer = disp_buffer;
1406 }
1407 this_thr->th.th_dispatch = serial_team->t.t_dispatch;
1408
1409 /* allocate/push task team stack */
1410 __kmp_push_task_team_node(this_thr, serial_team);
1411
1412 KMP_MB();
1413 }
1414 KMP_CHECK_UPDATE(serial_team->t.t_cancel_request, cancel_noreq);
1415
1416 // Perform the display affinity functionality for
1417 // serialized parallel regions
1419 if (this_thr->th.th_prev_level != serial_team->t.t_level ||
1420 this_thr->th.th_prev_num_threads != 1) {
1421 // NULL means use the affinity-format-var ICV
1422 __kmp_aux_display_affinity(global_tid, NULL);
1423 this_thr->th.th_prev_level = serial_team->t.t_level;
1424 this_thr->th.th_prev_num_threads = 1;
1425 }
1426 }
1427
1429 __kmp_push_parallel(global_tid, NULL);
1430#if OMPT_SUPPORT
1431 serial_team->t.ompt_team_info.master_return_address = codeptr;
1432 if (ompt_enabled.enabled &&
1433 this_thr->th.ompt_thread_info.state != ompt_state_overhead) {
1434 OMPT_CUR_TASK_INFO(this_thr)->frame.exit_frame.ptr =
1436
1437 ompt_lw_taskteam_t lw_taskteam;
1438 __ompt_lw_taskteam_init(&lw_taskteam, this_thr, global_tid,
1439 &ompt_parallel_data, codeptr);
1440
1441 __ompt_lw_taskteam_link(&lw_taskteam, this_thr, 1);
1442 // don't use lw_taskteam after linking. content was swaped
1443
1444 /* OMPT implicit task begin */
1445 if (ompt_enabled.ompt_callback_implicit_task) {
1446 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
1447 ompt_scope_begin, OMPT_CUR_TEAM_DATA(this_thr),
1448 OMPT_CUR_TASK_DATA(this_thr), 1, __kmp_tid_from_gtid(global_tid),
1449 ompt_task_implicit); // TODO: Can this be ompt_task_initial?
1450 OMPT_CUR_TASK_INFO(this_thr)->thread_num =
1451 __kmp_tid_from_gtid(global_tid);
1452 }
1453
1454 /* OMPT state */
1455 this_thr->th.ompt_thread_info.state = ompt_state_work_parallel;
1456 OMPT_CUR_TASK_INFO(this_thr)->frame.exit_frame.ptr =
1458 }
1459#endif
1460}
1461
1462// Test if this fork is for a team closely nested in a teams construct
1463static inline bool __kmp_is_fork_in_teams(kmp_info_t *master_th,
1465 int teams_level, kmp_va_list ap) {
1466 return (master_th->th.th_teams_microtask && ap &&
1467 microtask != (microtask_t)__kmp_teams_master && level == teams_level);
1468}
1469
1470// Test if this fork is for the teams construct, i.e. to form the outer league
1471// of teams
1472static inline bool __kmp_is_entering_teams(int active_level, int level,
1473 int teams_level, kmp_va_list ap) {
1474 return ((ap == NULL && active_level == 0) ||
1475 (ap && teams_level > 0 && teams_level == level));
1476}
1477
1478// AC: This is start of parallel that is nested inside teams construct.
1479// The team is actual (hot), all workers are ready at the fork barrier.
1480// No lock needed to initialize the team a bit, then free workers.
1481static inline int
1483 kmp_int32 argc, kmp_info_t *master_th, kmp_root_t *root,
1484 enum fork_context_e call_context, microtask_t microtask,
1485 launch_t invoker, int master_set_numthreads, int level,
1486#if OMPT_SUPPORT
1487 ompt_data_t ompt_parallel_data, void *return_address,
1488#endif
1489 kmp_va_list ap) {
1490 void **argv;
1491 int i;
1492
1493 parent_team->t.t_ident = loc;
1494 __kmp_alloc_argv_entries(argc, parent_team, TRUE);
1495 parent_team->t.t_argc = argc;
1496 argv = (void **)parent_team->t.t_argv;
1497 for (i = argc - 1; i >= 0; --i) {
1498 *argv++ = va_arg(kmp_va_deref(ap), void *);
1499 }
1500 // Increment our nested depth levels, but not increase the serialization
1501 if (parent_team == master_th->th.th_serial_team) {
1502 // AC: we are in serialized parallel
1504 KMP_DEBUG_ASSERT(parent_team->t.t_serialized > 1);
1505
1506 if (call_context == fork_context_gnu) {
1507 // AC: need to decrement t_serialized for enquiry functions to work
1508 // correctly, will restore at join time
1509 parent_team->t.t_serialized--;
1510 return TRUE;
1511 }
1512
1513#if OMPD_SUPPORT
1514 parent_team->t.t_pkfn = microtask;
1515#endif
1516
1517#if OMPT_SUPPORT
1518 void *dummy;
1519 void **exit_frame_p;
1520 ompt_data_t *implicit_task_data;
1521 ompt_lw_taskteam_t lw_taskteam;
1522
1523 if (ompt_enabled.enabled) {
1524 __ompt_lw_taskteam_init(&lw_taskteam, master_th, gtid,
1525 &ompt_parallel_data, return_address);
1526 exit_frame_p = &(lw_taskteam.ompt_task_info.frame.exit_frame.ptr);
1527
1528 __ompt_lw_taskteam_link(&lw_taskteam, master_th, 0);
1529 // Don't use lw_taskteam after linking. Content was swapped.
1530
1531 /* OMPT implicit task begin */
1532 implicit_task_data = OMPT_CUR_TASK_DATA(master_th);
1533 if (ompt_enabled.ompt_callback_implicit_task) {
1534 OMPT_CUR_TASK_INFO(master_th)->thread_num = __kmp_tid_from_gtid(gtid);
1535 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
1536 ompt_scope_begin, OMPT_CUR_TEAM_DATA(master_th), implicit_task_data,
1537 1, OMPT_CUR_TASK_INFO(master_th)->thread_num, ompt_task_implicit);
1538 }
1539
1540 /* OMPT state */
1541 master_th->th.ompt_thread_info.state = ompt_state_work_parallel;
1542 } else {
1543 exit_frame_p = &dummy;
1544 }
1545#endif
1546
1547 // AC: need to decrement t_serialized for enquiry functions to work
1548 // correctly, will restore at join time
1549 parent_team->t.t_serialized--;
1550
1551 {
1552 KMP_TIME_PARTITIONED_BLOCK(OMP_parallel);
1553 KMP_SET_THREAD_STATE_BLOCK(IMPLICIT_TASK);
1554 __kmp_invoke_microtask(microtask, gtid, 0, argc, parent_team->t.t_argv
1555#if OMPT_SUPPORT
1556 ,
1557 exit_frame_p
1558#endif
1559 );
1560 }
1561
1562#if OMPT_SUPPORT
1563 if (ompt_enabled.enabled) {
1564 *exit_frame_p = NULL;
1565 OMPT_CUR_TASK_INFO(master_th)->frame.exit_frame = ompt_data_none;
1566 if (ompt_enabled.ompt_callback_implicit_task) {
1567 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
1568 ompt_scope_end, NULL, implicit_task_data, 1,
1569 OMPT_CUR_TASK_INFO(master_th)->thread_num, ompt_task_implicit);
1570 }
1571 ompt_parallel_data = *OMPT_CUR_TEAM_DATA(master_th);
1572 __ompt_lw_taskteam_unlink(master_th);
1573 if (ompt_enabled.ompt_callback_parallel_end) {
1574 ompt_callbacks.ompt_callback(ompt_callback_parallel_end)(
1575 &ompt_parallel_data, OMPT_CUR_TASK_DATA(master_th),
1576 OMPT_INVOKER(call_context) | ompt_parallel_team, return_address);
1577 }
1578 master_th->th.ompt_thread_info.state = ompt_state_overhead;
1579 }
1580#endif
1581 return TRUE;
1582 }
1583
1584 parent_team->t.t_pkfn = microtask;
1585 parent_team->t.t_invoke = invoker;
1586 KMP_ATOMIC_INC(&root->r.r_in_parallel);
1587 parent_team->t.t_active_level++;
1588 parent_team->t.t_level++;
1589 parent_team->t.t_def_allocator = master_th->th.th_def_allocator; // save
1590
1591 // If the threads allocated to the team are less than the thread limit, update
1592 // the thread limit here. th_teams_size.nth is specific to this team nested
1593 // in a teams construct, the team is fully created, and we're about to do
1594 // the actual fork. Best to do this here so that the subsequent uses below
1595 // and in the join have the correct value.
1596 master_th->th.th_teams_size.nth = parent_team->t.t_nproc;
1597
1598#if OMPT_SUPPORT
1599 if (ompt_enabled.enabled) {
1600 ompt_lw_taskteam_t lw_taskteam;
1601 __ompt_lw_taskteam_init(&lw_taskteam, master_th, gtid, &ompt_parallel_data,
1602 return_address);
1603 __ompt_lw_taskteam_link(&lw_taskteam, master_th, 1, true);
1604 }
1605#endif
1606
1607 /* Change number of threads in the team if requested */
1608 if (master_set_numthreads) { // The parallel has num_threads clause
1609 if (master_set_numthreads <= master_th->th.th_teams_size.nth) {
1610 // AC: only can reduce number of threads dynamically, can't increase
1611 kmp_info_t **other_threads = parent_team->t.t_threads;
1612 // NOTE: if using distributed barrier, we need to run this code block
1613 // even when the team size appears not to have changed from the max.
1614 int old_proc = master_th->th.th_teams_size.nth;
1616 __kmp_resize_dist_barrier(parent_team, old_proc, master_set_numthreads);
1617 __kmp_add_threads_to_team(parent_team, master_set_numthreads);
1618 }
1619 parent_team->t.t_nproc = master_set_numthreads;
1620 for (i = 0; i < master_set_numthreads; ++i) {
1621 other_threads[i]->th.th_team_nproc = master_set_numthreads;
1622 }
1623 }
1624 // Keep extra threads hot in the team for possible next parallels
1625 master_th->th.th_set_nproc = 0;
1626 }
1627
1628#if USE_DEBUGGER
1629 if (__kmp_debugging) { // Let debugger override number of threads.
1630 int nth = __kmp_omp_num_threads(loc);
1631 if (nth > 0) { // 0 means debugger doesn't want to change num threads
1632 master_set_numthreads = nth;
1633 }
1634 }
1635#endif
1636
1637 // Figure out the proc_bind policy for the nested parallel within teams
1638 kmp_proc_bind_t proc_bind = master_th->th.th_set_proc_bind;
1639 // proc_bind_default means don't update
1640 kmp_proc_bind_t proc_bind_icv = proc_bind_default;
1641 if (master_th->th.th_current_task->td_icvs.proc_bind == proc_bind_false) {
1642 proc_bind = proc_bind_false;
1643 } else {
1644 // No proc_bind clause specified; use current proc-bind-var
1645 if (proc_bind == proc_bind_default) {
1646 proc_bind = master_th->th.th_current_task->td_icvs.proc_bind;
1647 }
1648 /* else: The proc_bind policy was specified explicitly on parallel clause.
1649 This overrides proc-bind-var for this parallel region, but does not
1650 change proc-bind-var. */
1651 // Figure the value of proc-bind-var for the child threads.
1652 if ((level + 1 < __kmp_nested_proc_bind.used) &&
1653 (__kmp_nested_proc_bind.bind_types[level + 1] !=
1654 master_th->th.th_current_task->td_icvs.proc_bind)) {
1655 proc_bind_icv = __kmp_nested_proc_bind.bind_types[level + 1];
1656 }
1657 }
1658 KMP_CHECK_UPDATE(parent_team->t.t_proc_bind, proc_bind);
1659 // Need to change the bind-var ICV to correct value for each implicit task
1660 if (proc_bind_icv != proc_bind_default &&
1661 master_th->th.th_current_task->td_icvs.proc_bind != proc_bind_icv) {
1662 kmp_info_t **other_threads = parent_team->t.t_threads;
1663 for (i = 0; i < master_th->th.th_team_nproc; ++i) {
1664 other_threads[i]->th.th_current_task->td_icvs.proc_bind = proc_bind_icv;
1665 }
1666 }
1667 // Reset for next parallel region
1668 master_th->th.th_set_proc_bind = proc_bind_default;
1669
1670#if USE_ITT_BUILD && USE_ITT_NOTIFY
1671 if (((__itt_frame_submit_v3_ptr && __itt_get_timestamp_ptr) ||
1672 KMP_ITT_DEBUG) &&
1673 __kmp_forkjoin_frames_mode == 3 &&
1674 parent_team->t.t_active_level == 1 // only report frames at level 1
1675 && master_th->th.th_teams_size.nteams == 1) {
1676 kmp_uint64 tmp_time = __itt_get_timestamp();
1677 master_th->th.th_frame_time = tmp_time;
1678 parent_team->t.t_region_time = tmp_time;
1679 }
1680 if (__itt_stack_caller_create_ptr) {
1681 KMP_DEBUG_ASSERT(parent_team->t.t_stack_id == NULL);
1682 // create new stack stitching id before entering fork barrier
1683 parent_team->t.t_stack_id = __kmp_itt_stack_caller_create();
1684 }
1685#endif /* USE_ITT_BUILD && USE_ITT_NOTIFY */
1686#if KMP_AFFINITY_SUPPORTED
1687 __kmp_partition_places(parent_team);
1688#endif
1689
1690 KF_TRACE(10, ("__kmp_fork_in_teams: before internal fork: root=%p, team=%p, "
1691 "master_th=%p, gtid=%d\n",
1692 root, parent_team, master_th, gtid));
1693 __kmp_internal_fork(loc, gtid, parent_team);
1694 KF_TRACE(10, ("__kmp_fork_in_teams: after internal fork: root=%p, team=%p, "
1695 "master_th=%p, gtid=%d\n",
1696 root, parent_team, master_th, gtid));
1697
1698 if (call_context == fork_context_gnu)
1699 return TRUE;
1700
1701 /* Invoke microtask for PRIMARY thread */
1702 KA_TRACE(20, ("__kmp_fork_in_teams: T#%d(%d:0) invoke microtask = %p\n", gtid,
1703 parent_team->t.t_id, parent_team->t.t_pkfn));
1704
1705 if (!parent_team->t.t_invoke(gtid)) {
1706 KMP_ASSERT2(0, "cannot invoke microtask for PRIMARY thread");
1707 }
1708 KA_TRACE(20, ("__kmp_fork_in_teams: T#%d(%d:0) done microtask = %p\n", gtid,
1709 parent_team->t.t_id, parent_team->t.t_pkfn));
1710 KMP_MB(); /* Flush all pending memory write invalidates. */
1711
1712 KA_TRACE(20, ("__kmp_fork_in_teams: parallel exit T#%d\n", gtid));
1713
1714 return TRUE;
1715}
1716
1717// Create a serialized parallel region
1718static inline int
1719__kmp_serial_fork_call(ident_t *loc, int gtid, enum fork_context_e call_context,
1720 kmp_int32 argc, microtask_t microtask, launch_t invoker,
1721 kmp_info_t *master_th, kmp_team_t *parent_team,
1722#if OMPT_SUPPORT
1723 ompt_data_t *ompt_parallel_data, void **return_address,
1724 ompt_data_t **parent_task_data,
1725#endif
1726 kmp_va_list ap) {
1727 kmp_team_t *team;
1728 int i;
1729 void **argv;
1730
1731/* josh todo: hypothetical question: what do we do for OS X*? */
1732#if KMP_OS_LINUX && \
1733 (KMP_ARCH_X86 || KMP_ARCH_X86_64 || KMP_ARCH_ARM || KMP_ARCH_AARCH64)
1734 SimpleVLA<void *> args(argc);
1735#else
1736 void **args = (void **)KMP_ALLOCA(argc * sizeof(void *));
1737#endif /* KMP_OS_LINUX && ( KMP_ARCH_X86 || KMP_ARCH_X86_64 || KMP_ARCH_ARM || \
1738 KMP_ARCH_AARCH64) */
1739
1740 KA_TRACE(
1741 20, ("__kmp_serial_fork_call: T#%d serializing parallel region\n", gtid));
1742
1744
1745#if OMPD_SUPPORT
1746 master_th->th.th_serial_team->t.t_pkfn = microtask;
1747#endif
1748
1749 if (call_context == fork_context_intel) {
1750 /* TODO this sucks, use the compiler itself to pass args! :) */
1751 master_th->th.th_serial_team->t.t_ident = loc;
1752 if (!ap) {
1753 // revert change made in __kmpc_serialized_parallel()
1754 master_th->th.th_serial_team->t.t_level--;
1755// Get args from parent team for teams construct
1756
1757#if OMPT_SUPPORT
1758 void *dummy;
1759 void **exit_frame_p;
1760 ompt_task_info_t *task_info;
1761 ompt_lw_taskteam_t lw_taskteam;
1762
1763 if (ompt_enabled.enabled) {
1764 __ompt_lw_taskteam_init(&lw_taskteam, master_th, gtid,
1765 ompt_parallel_data, *return_address);
1766
1767 __ompt_lw_taskteam_link(&lw_taskteam, master_th, 0);
1768 // don't use lw_taskteam after linking. content was swaped
1769 task_info = OMPT_CUR_TASK_INFO(master_th);
1770 exit_frame_p = &(task_info->frame.exit_frame.ptr);
1771 if (ompt_enabled.ompt_callback_implicit_task) {
1772 OMPT_CUR_TASK_INFO(master_th)->thread_num = __kmp_tid_from_gtid(gtid);
1773 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
1774 ompt_scope_begin, OMPT_CUR_TEAM_DATA(master_th),
1775 &(task_info->task_data), 1,
1776 OMPT_CUR_TASK_INFO(master_th)->thread_num, ompt_task_implicit);
1777 }
1778
1779 /* OMPT state */
1780 master_th->th.ompt_thread_info.state = ompt_state_work_parallel;
1781 } else {
1782 exit_frame_p = &dummy;
1783 }
1784#endif
1785
1786 {
1787 KMP_TIME_PARTITIONED_BLOCK(OMP_parallel);
1788 KMP_SET_THREAD_STATE_BLOCK(IMPLICIT_TASK);
1789 __kmp_invoke_microtask(microtask, gtid, 0, argc, parent_team->t.t_argv
1790#if OMPT_SUPPORT
1791 ,
1792 exit_frame_p
1793#endif
1794 );
1795 }
1796
1797#if OMPT_SUPPORT
1798 if (ompt_enabled.enabled) {
1799 *exit_frame_p = NULL;
1800 if (ompt_enabled.ompt_callback_implicit_task) {
1801 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
1802 ompt_scope_end, NULL, &(task_info->task_data), 1,
1803 OMPT_CUR_TASK_INFO(master_th)->thread_num, ompt_task_implicit);
1804 }
1805 *ompt_parallel_data = *OMPT_CUR_TEAM_DATA(master_th);
1806 __ompt_lw_taskteam_unlink(master_th);
1807 if (ompt_enabled.ompt_callback_parallel_end) {
1808 ompt_callbacks.ompt_callback(ompt_callback_parallel_end)(
1809 ompt_parallel_data, *parent_task_data,
1810 OMPT_INVOKER(call_context) | ompt_parallel_team, *return_address);
1811 }
1812 master_th->th.ompt_thread_info.state = ompt_state_overhead;
1813 }
1814#endif
1815 } else if (microtask == (microtask_t)__kmp_teams_master) {
1816 KMP_DEBUG_ASSERT(master_th->th.th_team == master_th->th.th_serial_team);
1817 team = master_th->th.th_team;
1818 // team->t.t_pkfn = microtask;
1819 team->t.t_invoke = invoker;
1820 __kmp_alloc_argv_entries(argc, team, TRUE);
1821 team->t.t_argc = argc;
1822 argv = (void **)team->t.t_argv;
1823 for (i = argc - 1; i >= 0; --i)
1824 *argv++ = va_arg(kmp_va_deref(ap), void *);
1825 // AC: revert change made in __kmpc_serialized_parallel()
1826 // because initial code in teams should have level=0
1827 team->t.t_level--;
1828 // AC: call special invoker for outer "parallel" of teams construct
1829 invoker(gtid);
1830#if OMPT_SUPPORT
1831 if (ompt_enabled.enabled) {
1832 ompt_task_info_t *task_info = OMPT_CUR_TASK_INFO(master_th);
1833 if (ompt_enabled.ompt_callback_implicit_task) {
1834 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
1835 ompt_scope_end, NULL, &(task_info->task_data), 0,
1836 OMPT_CUR_TASK_INFO(master_th)->thread_num, ompt_task_initial);
1837 }
1838 if (ompt_enabled.ompt_callback_parallel_end) {
1839 ompt_callbacks.ompt_callback(ompt_callback_parallel_end)(
1840 ompt_parallel_data, *parent_task_data,
1841 OMPT_INVOKER(call_context) | ompt_parallel_league,
1842 *return_address);
1843 }
1844 master_th->th.ompt_thread_info.state = ompt_state_overhead;
1845 }
1846#endif
1847 } else {
1848 argv = args;
1849 for (i = argc - 1; i >= 0; --i)
1850 *argv++ = va_arg(kmp_va_deref(ap), void *);
1851 KMP_MB();
1852
1853#if OMPT_SUPPORT
1854 void *dummy;
1855 void **exit_frame_p;
1856 ompt_task_info_t *task_info;
1857 ompt_lw_taskteam_t lw_taskteam;
1858 ompt_data_t *implicit_task_data;
1859
1860 if (ompt_enabled.enabled) {
1861 __ompt_lw_taskteam_init(&lw_taskteam, master_th, gtid,
1862 ompt_parallel_data, *return_address);
1863 __ompt_lw_taskteam_link(&lw_taskteam, master_th, 0);
1864 // don't use lw_taskteam after linking. content was swaped
1865 task_info = OMPT_CUR_TASK_INFO(master_th);
1866 exit_frame_p = &(task_info->frame.exit_frame.ptr);
1867
1868 /* OMPT implicit task begin */
1869 implicit_task_data = OMPT_CUR_TASK_DATA(master_th);
1870 if (ompt_enabled.ompt_callback_implicit_task) {
1871 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
1872 ompt_scope_begin, OMPT_CUR_TEAM_DATA(master_th),
1873 implicit_task_data, 1, __kmp_tid_from_gtid(gtid),
1874 ompt_task_implicit);
1875 OMPT_CUR_TASK_INFO(master_th)->thread_num = __kmp_tid_from_gtid(gtid);
1876 }
1877
1878 /* OMPT state */
1879 master_th->th.ompt_thread_info.state = ompt_state_work_parallel;
1880 } else {
1881 exit_frame_p = &dummy;
1882 }
1883#endif
1884
1885 {
1886 KMP_TIME_PARTITIONED_BLOCK(OMP_parallel);
1887 KMP_SET_THREAD_STATE_BLOCK(IMPLICIT_TASK);
1888 __kmp_invoke_microtask(microtask, gtid, 0, argc, args
1889#if OMPT_SUPPORT
1890 ,
1891 exit_frame_p
1892#endif
1893 );
1894 }
1895
1896#if OMPT_SUPPORT
1897 if (ompt_enabled.enabled) {
1898 *exit_frame_p = NULL;
1899 if (ompt_enabled.ompt_callback_implicit_task) {
1900 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
1901 ompt_scope_end, NULL, &(task_info->task_data), 1,
1902 OMPT_CUR_TASK_INFO(master_th)->thread_num, ompt_task_implicit);
1903 }
1904
1905 *ompt_parallel_data = *OMPT_CUR_TEAM_DATA(master_th);
1906 __ompt_lw_taskteam_unlink(master_th);
1907 if (ompt_enabled.ompt_callback_parallel_end) {
1908 ompt_callbacks.ompt_callback(ompt_callback_parallel_end)(
1909 ompt_parallel_data, *parent_task_data,
1910 OMPT_INVOKER(call_context) | ompt_parallel_team, *return_address);
1911 }
1912 master_th->th.ompt_thread_info.state = ompt_state_overhead;
1913 }
1914#endif
1915 }
1916 } else if (call_context == fork_context_gnu) {
1917#if OMPT_SUPPORT
1918 if (ompt_enabled.enabled) {
1920 __ompt_lw_taskteam_init(&lwt, master_th, gtid, ompt_parallel_data,
1921 *return_address);
1922
1923 lwt.ompt_task_info.frame.exit_frame = ompt_data_none;
1924 __ompt_lw_taskteam_link(&lwt, master_th, 1);
1925 }
1926// don't use lw_taskteam after linking. content was swaped
1927#endif
1928
1929 // we were called from GNU native code
1930 KA_TRACE(20, ("__kmp_serial_fork_call: T#%d serial exit\n", gtid));
1931 return FALSE;
1932 } else {
1933 KMP_ASSERT2(call_context < fork_context_last,
1934 "__kmp_serial_fork_call: unknown fork_context parameter");
1935 }
1936
1937 KA_TRACE(20, ("__kmp_serial_fork_call: T#%d serial exit\n", gtid));
1938 KMP_MB();
1939 return FALSE;
1940}
1941
1942/* most of the work for a fork */
1943/* return true if we really went parallel, false if serialized */
1945 enum fork_context_e call_context, // Intel, GNU, ...
1946 kmp_int32 argc, microtask_t microtask, launch_t invoker,
1947 kmp_va_list ap) {
1948 void **argv;
1949 int i;
1950 int master_tid;
1951 int master_this_cons;
1952 kmp_team_t *team;
1953 kmp_team_t *parent_team;
1954 kmp_info_t *master_th;
1955 kmp_root_t *root;
1956 int nthreads;
1957 int master_active;
1958 int master_set_numthreads;
1959 int task_thread_limit = 0;
1960 int level;
1961 int active_level;
1962 int teams_level;
1963 kmp_hot_team_ptr_t **p_hot_teams;
1964 { // KMP_TIME_BLOCK
1966 KMP_COUNT_VALUE(OMP_PARALLEL_args, argc);
1967
1968 KA_TRACE(20, ("__kmp_fork_call: enter T#%d\n", gtid));
1969 if (__kmp_stkpadding > 0 && __kmp_root[gtid] != NULL) {
1970 /* Some systems prefer the stack for the root thread(s) to start with */
1971 /* some gap from the parent stack to prevent false sharing. */
1972 void *dummy = KMP_ALLOCA(__kmp_stkpadding);
1973 /* These 2 lines below are so this does not get optimized out */
1975 __kmp_stkpadding += (short)((kmp_int64)dummy);
1976 }
1977
1978 /* initialize if needed */
1980 __kmp_init_serial); // AC: potentially unsafe, not in sync with shutdown
1984
1985 /* setup current data */
1986 // AC: potentially unsafe, not in sync with library shutdown,
1987 // __kmp_threads can be freed
1988 master_th = __kmp_threads[gtid];
1989
1990 parent_team = master_th->th.th_team;
1991 master_tid = master_th->th.th_info.ds.ds_tid;
1992 master_this_cons = master_th->th.th_local.this_construct;
1993 root = master_th->th.th_root;
1994 master_active = root->r.r_active;
1995 master_set_numthreads = master_th->th.th_set_nproc;
1996 task_thread_limit =
1997 master_th->th.th_current_task->td_icvs.task_thread_limit;
1998
1999#if OMPT_SUPPORT
2000 ompt_data_t ompt_parallel_data = ompt_data_none;
2001 ompt_data_t *parent_task_data = NULL;
2002 ompt_frame_t *ompt_frame = NULL;
2003 void *return_address = NULL;
2004
2005 if (ompt_enabled.enabled) {
2006 __ompt_get_task_info_internal(0, NULL, &parent_task_data, &ompt_frame,
2007 NULL, NULL);
2008 return_address = OMPT_LOAD_RETURN_ADDRESS(gtid);
2009 }
2010#endif
2011
2012 // Assign affinity to root thread if it hasn't happened yet
2014
2015 // Nested level will be an index in the nested nthreads array
2016 level = parent_team->t.t_level;
2017 // used to launch non-serial teams even if nested is not allowed
2018 active_level = parent_team->t.t_active_level;
2019 // needed to check nesting inside the teams
2020 teams_level = master_th->th.th_teams_level;
2021 p_hot_teams = &master_th->th.th_hot_teams;
2022 if (*p_hot_teams == NULL && __kmp_hot_teams_max_level > 0) {
2023 *p_hot_teams = (kmp_hot_team_ptr_t *)__kmp_allocate(
2025 (*p_hot_teams)[0].hot_team = root->r.r_hot_team;
2026 // it is either actual or not needed (when active_level > 0)
2027 (*p_hot_teams)[0].hot_team_nth = 1;
2028 }
2029
2030#if OMPT_SUPPORT
2031 if (ompt_enabled.enabled) {
2032 if (ompt_enabled.ompt_callback_parallel_begin) {
2033 int team_size = master_set_numthreads
2034 ? master_set_numthreads
2035 : get__nproc_2(parent_team, master_tid);
2036 int flags = OMPT_INVOKER(call_context) |
2038 ? ompt_parallel_league
2039 : ompt_parallel_team);
2040 ompt_callbacks.ompt_callback(ompt_callback_parallel_begin)(
2041 parent_task_data, ompt_frame, &ompt_parallel_data, team_size, flags,
2042 return_address);
2043 }
2044 master_th->th.ompt_thread_info.state = ompt_state_overhead;
2045 }
2046#endif
2047
2048 master_th->th.th_ident = loc;
2049
2050 // Parallel closely nested in teams construct:
2051 if (__kmp_is_fork_in_teams(master_th, microtask, level, teams_level, ap)) {
2052 return __kmp_fork_in_teams(loc, gtid, parent_team, argc, master_th, root,
2053 call_context, microtask, invoker,
2054 master_set_numthreads, level,
2055#if OMPT_SUPPORT
2056 ompt_parallel_data, return_address,
2057#endif
2058 ap);
2059 } // End parallel closely nested in teams construct
2060
2061 // Need this to happen before we determine the number of threads, not while
2062 // we are allocating the team
2063 //__kmp_push_current_task_to_thread(master_th, parent_team, 0);
2064
2065 KMP_DEBUG_ASSERT_TASKTEAM_INVARIANT(parent_team, master_th);
2066
2067 // Determine the number of threads
2068 int enter_teams =
2069 __kmp_is_entering_teams(active_level, level, teams_level, ap);
2070 if ((!enter_teams &&
2071 (parent_team->t.t_active_level >=
2072 master_th->th.th_current_task->td_icvs.max_active_levels)) ||
2074 KC_TRACE(10, ("__kmp_fork_call: T#%d serializing team\n", gtid));
2075 nthreads = 1;
2076 } else {
2077 nthreads = master_set_numthreads
2078 ? master_set_numthreads
2079 // TODO: get nproc directly from current task
2080 : get__nproc_2(parent_team, master_tid);
2081 // Use the thread_limit set for the current target task if exists, else go
2082 // with the deduced nthreads
2083 nthreads = task_thread_limit > 0 && task_thread_limit < nthreads
2084 ? task_thread_limit
2085 : nthreads;
2086 // Check if we need to take forkjoin lock? (no need for serialized
2087 // parallel out of teams construct).
2088 if (nthreads > 1) {
2089 /* determine how many new threads we can use */
2091 /* AC: If we execute teams from parallel region (on host), then teams
2092 should be created but each can only have 1 thread if nesting is
2093 disabled. If teams called from serial region, then teams and their
2094 threads should be created regardless of the nesting setting. */
2095 nthreads = __kmp_reserve_threads(root, parent_team, master_tid,
2096 nthreads, enter_teams);
2097 if (nthreads == 1) {
2098 // Free lock for single thread execution here; for multi-thread
2099 // execution it will be freed later after team of threads created
2100 // and initialized
2102 }
2103 }
2104 }
2105 KMP_DEBUG_ASSERT(nthreads > 0);
2106
2107 // If we temporarily changed the set number of threads then restore it now
2108 master_th->th.th_set_nproc = 0;
2109
2110 if (nthreads == 1) {
2111 return __kmp_serial_fork_call(loc, gtid, call_context, argc, microtask,
2112 invoker, master_th, parent_team,
2113#if OMPT_SUPPORT
2114 &ompt_parallel_data, &return_address,
2115 &parent_task_data,
2116#endif
2117 ap);
2118 } // if (nthreads == 1)
2119
2120 // GEH: only modify the executing flag in the case when not serialized
2121 // serialized case is handled in kmpc_serialized_parallel
2122 KF_TRACE(10, ("__kmp_fork_call: parent_team_aclevel=%d, master_th=%p, "
2123 "curtask=%p, curtask_max_aclevel=%d\n",
2124 parent_team->t.t_active_level, master_th,
2125 master_th->th.th_current_task,
2126 master_th->th.th_current_task->td_icvs.max_active_levels));
2127 // TODO: GEH - cannot do this assertion because root thread not set up as
2128 // executing
2129 // KMP_ASSERT( master_th->th.th_current_task->td_flags.executing == 1 );
2130 master_th->th.th_current_task->td_flags.executing = 0;
2131
2132 if (!master_th->th.th_teams_microtask || level > teams_level) {
2133 /* Increment our nested depth level */
2134 KMP_ATOMIC_INC(&root->r.r_in_parallel);
2135 }
2136
2137 // See if we need to make a copy of the ICVs.
2138 int nthreads_icv = master_th->th.th_current_task->td_icvs.nproc;
2139 kmp_nested_nthreads_t *nested_nth = NULL;
2140 if (!master_th->th.th_set_nested_nth &&
2141 (level + 1 < parent_team->t.t_nested_nth->used) &&
2142 (parent_team->t.t_nested_nth->nth[level + 1] != nthreads_icv)) {
2143 nthreads_icv = parent_team->t.t_nested_nth->nth[level + 1];
2144 } else if (master_th->th.th_set_nested_nth) {
2145 nested_nth = __kmp_override_nested_nth(master_th, level);
2146 if ((level + 1 < nested_nth->used) &&
2147 (nested_nth->nth[level + 1] != nthreads_icv))
2148 nthreads_icv = nested_nth->nth[level + 1];
2149 else
2150 nthreads_icv = 0; // don't update
2151 } else {
2152 nthreads_icv = 0; // don't update
2153 }
2154
2155 // Figure out the proc_bind_policy for the new team.
2156 kmp_proc_bind_t proc_bind = master_th->th.th_set_proc_bind;
2157 // proc_bind_default means don't update
2158 kmp_proc_bind_t proc_bind_icv = proc_bind_default;
2159 if (master_th->th.th_current_task->td_icvs.proc_bind == proc_bind_false) {
2160 proc_bind = proc_bind_false;
2161 } else {
2162 // No proc_bind clause specified; use current proc-bind-var for this
2163 // parallel region
2164 if (proc_bind == proc_bind_default) {
2165 proc_bind = master_th->th.th_current_task->td_icvs.proc_bind;
2166 }
2167 // Have teams construct take proc_bind value from KMP_TEAMS_PROC_BIND
2168 if (master_th->th.th_teams_microtask &&
2170 proc_bind = __kmp_teams_proc_bind;
2171 }
2172 /* else: The proc_bind policy was specified explicitly on parallel clause.
2173 This overrides proc-bind-var for this parallel region, but does not
2174 change proc-bind-var. */
2175 // Figure the value of proc-bind-var for the child threads.
2176 if ((level + 1 < __kmp_nested_proc_bind.used) &&
2177 (__kmp_nested_proc_bind.bind_types[level + 1] !=
2178 master_th->th.th_current_task->td_icvs.proc_bind)) {
2179 // Do not modify the proc bind icv for the two teams construct forks
2180 // They just let the proc bind icv pass through
2181 if (!master_th->th.th_teams_microtask ||
2182 !(microtask == (microtask_t)__kmp_teams_master || ap == NULL))
2183 proc_bind_icv = __kmp_nested_proc_bind.bind_types[level + 1];
2184 }
2185 }
2186
2187 // Reset for next parallel region
2188 master_th->th.th_set_proc_bind = proc_bind_default;
2189
2190 if ((nthreads_icv > 0) || (proc_bind_icv != proc_bind_default)) {
2191 kmp_internal_control_t new_icvs;
2192 copy_icvs(&new_icvs, &master_th->th.th_current_task->td_icvs);
2193 new_icvs.next = NULL;
2194 if (nthreads_icv > 0) {
2195 new_icvs.nproc = nthreads_icv;
2196 }
2197 if (proc_bind_icv != proc_bind_default) {
2198 new_icvs.proc_bind = proc_bind_icv;
2199 }
2200
2201 /* allocate a new parallel team */
2202 KF_TRACE(10, ("__kmp_fork_call: before __kmp_allocate_team\n"));
2203 team = __kmp_allocate_team(root, nthreads, nthreads,
2204#if OMPT_SUPPORT
2205 ompt_parallel_data,
2206#endif
2207 proc_bind, &new_icvs, argc, master_th);
2209 copy_icvs((kmp_internal_control_t *)team->t.b->team_icvs, &new_icvs);
2210 } else {
2211 /* allocate a new parallel team */
2212 KF_TRACE(10, ("__kmp_fork_call: before __kmp_allocate_team\n"));
2213 team = __kmp_allocate_team(
2214 root, nthreads, nthreads,
2215#if OMPT_SUPPORT
2216 ompt_parallel_data,
2217#endif
2218 proc_bind, &master_th->th.th_current_task->td_icvs, argc, master_th);
2220 copy_icvs((kmp_internal_control_t *)team->t.b->team_icvs,
2221 &master_th->th.th_current_task->td_icvs);
2222 }
2223 KF_TRACE(
2224 10, ("__kmp_fork_call: after __kmp_allocate_team - team = %p\n", team));
2225
2226 /* setup the new team */
2227 KMP_CHECK_UPDATE(team->t.t_master_tid, master_tid);
2228 KMP_CHECK_UPDATE(team->t.t_master_this_cons, master_this_cons);
2229 KMP_CHECK_UPDATE(team->t.t_ident, loc);
2230 KMP_CHECK_UPDATE(team->t.t_parent, parent_team);
2231 KMP_CHECK_UPDATE_SYNC(team->t.t_pkfn, microtask);
2232#if OMPT_SUPPORT
2233 KMP_CHECK_UPDATE_SYNC(team->t.ompt_team_info.master_return_address,
2234 return_address);
2235#endif
2236 KMP_CHECK_UPDATE(team->t.t_invoke, invoker); // TODO move to root, maybe
2237 // TODO: parent_team->t.t_level == INT_MAX ???
2238 if (!master_th->th.th_teams_microtask || level > teams_level) {
2239 int new_level = parent_team->t.t_level + 1;
2240 KMP_CHECK_UPDATE(team->t.t_level, new_level);
2241 new_level = parent_team->t.t_active_level + 1;
2242 KMP_CHECK_UPDATE(team->t.t_active_level, new_level);
2243 } else {
2244 // AC: Do not increase parallel level at start of the teams construct
2245 int new_level = parent_team->t.t_level;
2246 KMP_CHECK_UPDATE(team->t.t_level, new_level);
2247 new_level = parent_team->t.t_active_level;
2248 KMP_CHECK_UPDATE(team->t.t_active_level, new_level);
2249 }
2250 kmp_r_sched_t new_sched = get__sched_2(parent_team, master_tid);
2251 // set primary thread's schedule as new run-time schedule
2252 KMP_CHECK_UPDATE(team->t.t_sched.sched, new_sched.sched);
2253
2254 KMP_CHECK_UPDATE(team->t.t_cancel_request, cancel_noreq);
2255 KMP_CHECK_UPDATE(team->t.t_def_allocator, master_th->th.th_def_allocator);
2256
2257 // Check if hot team has potentially outdated list, and if so, free it
2258 if (team->t.t_nested_nth &&
2259 team->t.t_nested_nth != parent_team->t.t_nested_nth) {
2260 KMP_INTERNAL_FREE(team->t.t_nested_nth->nth);
2261 KMP_INTERNAL_FREE(team->t.t_nested_nth);
2262 team->t.t_nested_nth = NULL;
2263 }
2264 team->t.t_nested_nth = parent_team->t.t_nested_nth;
2265 if (master_th->th.th_set_nested_nth) {
2266 if (!nested_nth)
2267 nested_nth = __kmp_override_nested_nth(master_th, level);
2268 team->t.t_nested_nth = nested_nth;
2269 KMP_INTERNAL_FREE(master_th->th.th_set_nested_nth);
2270 master_th->th.th_set_nested_nth = NULL;
2271 master_th->th.th_set_nested_nth_sz = 0;
2272 master_th->th.th_nt_strict = false;
2273 }
2274
2275 // Update the floating point rounding in the team if required.
2276 propagateFPControl(team);
2277#if OMPD_SUPPORT
2278 if (ompd_state & OMPD_ENABLE_BP)
2279 ompd_bp_parallel_begin();
2280#endif
2281
2282 KA_TRACE(
2283 20,
2284 ("__kmp_fork_call: T#%d(%d:%d)->(%d:0) created a team of %d threads\n",
2285 gtid, parent_team->t.t_id, team->t.t_master_tid, team->t.t_id,
2286 team->t.t_nproc));
2287 KMP_DEBUG_ASSERT(team != root->r.r_hot_team ||
2288 (team->t.t_master_tid == 0 &&
2289 (team->t.t_parent == root->r.r_root_team ||
2290 team->t.t_parent->t.t_serialized)));
2291 KMP_MB();
2292
2293 /* now, setup the arguments */
2294 argv = (void **)team->t.t_argv;
2295 if (ap) {
2296 for (i = argc - 1; i >= 0; --i) {
2297 void *new_argv = va_arg(kmp_va_deref(ap), void *);
2298 KMP_CHECK_UPDATE(*argv, new_argv);
2299 argv++;
2300 }
2301 } else {
2302 for (i = 0; i < argc; ++i) {
2303 // Get args from parent team for teams construct
2304 KMP_CHECK_UPDATE(argv[i], team->t.t_parent->t.t_argv[i]);
2305 }
2306 }
2307
2308 /* now actually fork the threads */
2309 KMP_CHECK_UPDATE(team->t.t_master_active, master_active);
2310 if (!root->r.r_active) // Only do assignment if it prevents cache ping-pong
2311 root->r.r_active = TRUE;
2312
2313 __kmp_fork_team_threads(root, team, master_th, gtid, !ap);
2314 __kmp_setup_icv_copy(team, nthreads,
2315 &master_th->th.th_current_task->td_icvs, loc);
2316
2317#if OMPT_SUPPORT
2318 master_th->th.ompt_thread_info.state = ompt_state_work_parallel;
2319#endif
2320
2322
2323#if USE_ITT_BUILD
2324 if (team->t.t_active_level == 1 // only report frames at level 1
2325 && !master_th->th.th_teams_microtask) { // not in teams construct
2326#if USE_ITT_NOTIFY
2327 if ((__itt_frame_submit_v3_ptr || KMP_ITT_DEBUG) &&
2328 (__kmp_forkjoin_frames_mode == 3 ||
2329 __kmp_forkjoin_frames_mode == 1)) {
2330 kmp_uint64 tmp_time = 0;
2331 if (__itt_get_timestamp_ptr)
2332 tmp_time = __itt_get_timestamp();
2333 // Internal fork - report frame begin
2334 master_th->th.th_frame_time = tmp_time;
2335 if (__kmp_forkjoin_frames_mode == 3)
2336 team->t.t_region_time = tmp_time;
2337 } else
2338// only one notification scheme (either "submit" or "forking/joined", not both)
2339#endif /* USE_ITT_NOTIFY */
2340 if ((__itt_frame_begin_v3_ptr || KMP_ITT_DEBUG) &&
2341 __kmp_forkjoin_frames && !__kmp_forkjoin_frames_mode) {
2342 // Mark start of "parallel" region for Intel(R) VTune(TM) analyzer.
2343 __kmp_itt_region_forking(gtid, team->t.t_nproc, 0);
2344 }
2345 }
2346#endif /* USE_ITT_BUILD */
2347
2348 /* now go on and do the work */
2349 KMP_DEBUG_ASSERT(team == __kmp_threads[gtid]->th.th_team);
2350 KMP_MB();
2351 KF_TRACE(10,
2352 ("__kmp_internal_fork : root=%p, team=%p, master_th=%p, gtid=%d\n",
2353 root, team, master_th, gtid));
2354
2355#if USE_ITT_BUILD
2356 if (__itt_stack_caller_create_ptr) {
2357 // create new stack stitching id before entering fork barrier
2358 if (!enter_teams) {
2359 KMP_DEBUG_ASSERT(team->t.t_stack_id == NULL);
2360 team->t.t_stack_id = __kmp_itt_stack_caller_create();
2361 } else if (parent_team->t.t_serialized) {
2362 // keep stack stitching id in the serialized parent_team;
2363 // current team will be used for parallel inside the teams;
2364 // if parent_team is active, then it already keeps stack stitching id
2365 // for the league of teams
2366 KMP_DEBUG_ASSERT(parent_team->t.t_stack_id == NULL);
2367 parent_team->t.t_stack_id = __kmp_itt_stack_caller_create();
2368 }
2369 }
2370#endif /* USE_ITT_BUILD */
2371
2372 // AC: skip __kmp_internal_fork at teams construct, let only primary
2373 // threads execute
2374 if (ap) {
2375 __kmp_internal_fork(loc, gtid, team);
2376 KF_TRACE(10, ("__kmp_internal_fork : after : root=%p, team=%p, "
2377 "master_th=%p, gtid=%d\n",
2378 root, team, master_th, gtid));
2379 }
2380
2381 if (call_context == fork_context_gnu) {
2382 KA_TRACE(20, ("__kmp_fork_call: parallel exit T#%d\n", gtid));
2383 return TRUE;
2384 }
2385
2386 /* Invoke microtask for PRIMARY thread */
2387 KA_TRACE(20, ("__kmp_fork_call: T#%d(%d:0) invoke microtask = %p\n", gtid,
2388 team->t.t_id, team->t.t_pkfn));
2389 } // END of timer KMP_fork_call block
2390
2391#if KMP_STATS_ENABLED
2392 // If beginning a teams construct, then change thread state
2393 stats_state_e previous_state = KMP_GET_THREAD_STATE();
2394 if (!ap) {
2395 KMP_SET_THREAD_STATE(stats_state_e::TEAMS_REGION);
2396 }
2397#endif
2398
2399 if (!team->t.t_invoke(gtid)) {
2400 KMP_ASSERT2(0, "cannot invoke microtask for PRIMARY thread");
2401 }
2402
2403#if KMP_STATS_ENABLED
2404 // If was beginning of a teams construct, then reset thread state
2405 if (!ap) {
2406 KMP_SET_THREAD_STATE(previous_state);
2407 }
2408#endif
2409
2410 KA_TRACE(20, ("__kmp_fork_call: T#%d(%d:0) done microtask = %p\n", gtid,
2411 team->t.t_id, team->t.t_pkfn));
2412 KMP_MB(); /* Flush all pending memory write invalidates. */
2413
2414 KA_TRACE(20, ("__kmp_fork_call: parallel exit T#%d\n", gtid));
2415#if OMPT_SUPPORT
2416 if (ompt_enabled.enabled) {
2417 master_th->th.ompt_thread_info.state = ompt_state_overhead;
2418 }
2419#endif
2420
2421 return TRUE;
2422}
2423
2424#if OMPT_SUPPORT
2425static inline void __kmp_join_restore_state(kmp_info_t *thread,
2426 kmp_team_t *team) {
2427 // restore state outside the region
2428 thread->th.ompt_thread_info.state =
2429 ((team->t.t_serialized) ? ompt_state_work_serial
2430 : ompt_state_work_parallel);
2431}
2432
2433static inline void __kmp_join_ompt(int gtid, kmp_info_t *thread,
2434 kmp_team_t *team, ompt_data_t *parallel_data,
2435 int flags, void *codeptr) {
2437 if (ompt_enabled.ompt_callback_parallel_end) {
2438 ompt_callbacks.ompt_callback(ompt_callback_parallel_end)(
2439 parallel_data, &(task_info->task_data), flags, codeptr);
2440 }
2441
2442 task_info->frame.enter_frame = ompt_data_none;
2443 __kmp_join_restore_state(thread, team);
2444}
2445#endif
2446
2448#if OMPT_SUPPORT
2449 ,
2450 enum fork_context_e fork_context
2451#endif
2452 ,
2453 int exit_teams) {
2455 kmp_team_t *team;
2456 kmp_team_t *parent_team;
2457 kmp_info_t *master_th;
2458 kmp_root_t *root;
2459 int master_active;
2460
2461 KA_TRACE(20, ("__kmp_join_call: enter T#%d\n", gtid));
2462
2463 /* setup current data */
2464 master_th = __kmp_threads[gtid];
2465 root = master_th->th.th_root;
2466 team = master_th->th.th_team;
2467 parent_team = team->t.t_parent;
2468
2469 master_th->th.th_ident = loc;
2470
2471#if OMPT_SUPPORT
2472 void *team_microtask = (void *)team->t.t_pkfn;
2473 // For GOMP interface with serialized parallel, need the
2474 // __kmpc_end_serialized_parallel to call hooks for OMPT end-implicit-task
2475 // and end-parallel events.
2476 if (ompt_enabled.enabled &&
2477 !(team->t.t_serialized && fork_context == fork_context_gnu)) {
2478 master_th->th.ompt_thread_info.state = ompt_state_overhead;
2479 }
2480#endif
2481
2482#if KMP_DEBUG
2483 if (__kmp_tasking_mode != tskm_immediate_exec && !exit_teams) {
2484 KA_TRACE(20, ("__kmp_join_call: T#%d, old team = %p old task_team = %p, "
2485 "th_task_team = %p\n",
2486 __kmp_gtid_from_thread(master_th), team,
2487 team->t.t_task_team[master_th->th.th_task_state],
2488 master_th->th.th_task_team));
2489 KMP_DEBUG_ASSERT_TASKTEAM_INVARIANT(team, master_th);
2490 }
2491#endif
2492
2493 if (team->t.t_serialized) {
2494 if (master_th->th.th_teams_microtask) {
2495 // We are in teams construct
2496 int level = team->t.t_level;
2497 int tlevel = master_th->th.th_teams_level;
2498 if (level == tlevel) {
2499 // AC: we haven't incremented it earlier at start of teams construct,
2500 // so do it here - at the end of teams construct
2501 team->t.t_level++;
2502 } else if (level == tlevel + 1) {
2503 // AC: we are exiting parallel inside teams, need to increment
2504 // serialization in order to restore it in the next call to
2505 // __kmpc_end_serialized_parallel
2506 team->t.t_serialized++;
2507 }
2508 }
2510
2511#if OMPT_SUPPORT
2512 if (ompt_enabled.enabled) {
2513 if (fork_context == fork_context_gnu) {
2514 __ompt_lw_taskteam_unlink(master_th);
2515 }
2516 __kmp_join_restore_state(master_th, parent_team);
2517 }
2518#endif
2519
2520 return;
2521 }
2522
2523 master_active = team->t.t_master_active;
2524
2525 if (!exit_teams) {
2526 // AC: No barrier for internal teams at exit from teams construct.
2527 // But there is barrier for external team (league).
2528 __kmp_internal_join(loc, gtid, team);
2529#if USE_ITT_BUILD
2530 if (__itt_stack_caller_create_ptr) {
2531 KMP_DEBUG_ASSERT(team->t.t_stack_id != NULL);
2532 // destroy the stack stitching id after join barrier
2533 __kmp_itt_stack_caller_destroy((__itt_caller)team->t.t_stack_id);
2534 team->t.t_stack_id = NULL;
2535 }
2536#endif
2537 } else {
2538 master_th->th.th_task_state =
2539 0; // AC: no tasking in teams (out of any parallel)
2540#if USE_ITT_BUILD
2541 if (__itt_stack_caller_create_ptr && parent_team->t.t_serialized) {
2542 KMP_DEBUG_ASSERT(parent_team->t.t_stack_id != NULL);
2543 // destroy the stack stitching id on exit from the teams construct
2544 // if parent_team is active, then the id will be destroyed later on
2545 // by master of the league of teams
2546 __kmp_itt_stack_caller_destroy((__itt_caller)parent_team->t.t_stack_id);
2547 parent_team->t.t_stack_id = NULL;
2548 }
2549#endif
2550 }
2551
2552 KMP_MB();
2553
2554#if OMPT_SUPPORT
2555 ompt_data_t *parallel_data = &(team->t.ompt_team_info.parallel_data);
2556 void *codeptr = team->t.ompt_team_info.master_return_address;
2557#endif
2558
2559#if USE_ITT_BUILD
2560 // Mark end of "parallel" region for Intel(R) VTune(TM) analyzer.
2561 if (team->t.t_active_level == 1 &&
2562 (!master_th->th.th_teams_microtask || /* not in teams construct */
2563 master_th->th.th_teams_size.nteams == 1)) {
2564 master_th->th.th_ident = loc;
2565 // only one notification scheme (either "submit" or "forking/joined", not
2566 // both)
2567 if ((__itt_frame_submit_v3_ptr || KMP_ITT_DEBUG) &&
2568 __kmp_forkjoin_frames_mode == 3)
2569 __kmp_itt_frame_submit(gtid, team->t.t_region_time,
2570 master_th->th.th_frame_time, 0, loc,
2571 master_th->th.th_team_nproc, 1);
2572 else if ((__itt_frame_end_v3_ptr || KMP_ITT_DEBUG) &&
2573 !__kmp_forkjoin_frames_mode && __kmp_forkjoin_frames)
2574 __kmp_itt_region_joined(gtid);
2575 } // active_level == 1
2576#endif /* USE_ITT_BUILD */
2577
2578#if KMP_AFFINITY_SUPPORTED
2579 if (!exit_teams) {
2580 // Restore master thread's partition.
2581 master_th->th.th_first_place = team->t.t_first_place;
2582 master_th->th.th_last_place = team->t.t_last_place;
2583 }
2584#endif // KMP_AFFINITY_SUPPORTED
2585
2586 if (master_th->th.th_teams_microtask && !exit_teams &&
2587 team->t.t_pkfn != (microtask_t)__kmp_teams_master &&
2588 team->t.t_level == master_th->th.th_teams_level + 1) {
2589// AC: We need to leave the team structure intact at the end of parallel
2590// inside the teams construct, so that at the next parallel same (hot) team
2591// works, only adjust nesting levels
2592#if OMPT_SUPPORT
2593 ompt_data_t ompt_parallel_data = ompt_data_none;
2594 if (ompt_enabled.enabled) {
2596 if (ompt_enabled.ompt_callback_implicit_task) {
2597 int ompt_team_size = team->t.t_nproc;
2598 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
2599 ompt_scope_end, NULL, &(task_info->task_data), ompt_team_size,
2600 OMPT_CUR_TASK_INFO(master_th)->thread_num, ompt_task_implicit);
2601 }
2602 task_info->frame.exit_frame = ompt_data_none;
2603 task_info->task_data = ompt_data_none;
2604 ompt_parallel_data = *OMPT_CUR_TEAM_DATA(master_th);
2605 __ompt_lw_taskteam_unlink(master_th);
2606 }
2607#endif
2608 /* Decrement our nested depth level */
2609 team->t.t_level--;
2610 team->t.t_active_level--;
2611 KMP_ATOMIC_DEC(&root->r.r_in_parallel);
2612
2613 // Restore number of threads in the team if needed. This code relies on
2614 // the proper adjustment of th_teams_size.nth after the fork in
2615 // __kmp_teams_master on each teams primary thread in the case that
2616 // __kmp_reserve_threads reduced it.
2617 if (master_th->th.th_team_nproc < master_th->th.th_teams_size.nth) {
2618 int old_num = master_th->th.th_team_nproc;
2619 int new_num = master_th->th.th_teams_size.nth;
2620 kmp_info_t **other_threads = team->t.t_threads;
2621 team->t.t_nproc = new_num;
2622 for (int i = 0; i < old_num; ++i) {
2623 other_threads[i]->th.th_team_nproc = new_num;
2624 }
2625 // Adjust states of non-used threads of the team
2626 for (int i = old_num; i < new_num; ++i) {
2627 // Re-initialize thread's barrier data.
2628 KMP_DEBUG_ASSERT(other_threads[i]);
2629 kmp_balign_t *balign = other_threads[i]->th.th_bar;
2630 for (int b = 0; b < bs_last_barrier; ++b) {
2631 balign[b].bb.b_arrived = team->t.t_bar[b].b_arrived;
2632 KMP_DEBUG_ASSERT(balign[b].bb.wait_flag != KMP_BARRIER_PARENT_FLAG);
2633#if USE_DEBUGGER
2634 balign[b].bb.b_worker_arrived = team->t.t_bar[b].b_team_arrived;
2635#endif
2636 }
2638 // Synchronize thread's task state
2639 other_threads[i]->th.th_task_state = master_th->th.th_task_state;
2640 }
2641 }
2642 }
2643
2644#if OMPT_SUPPORT
2645 if (ompt_enabled.enabled) {
2646 __kmp_join_ompt(gtid, master_th, parent_team, &ompt_parallel_data,
2647 OMPT_INVOKER(fork_context) | ompt_parallel_team, codeptr);
2648 }
2649#endif
2650
2651 return;
2652 }
2653
2654 /* do cleanup and restore the parent team */
2655 master_th->th.th_info.ds.ds_tid = team->t.t_master_tid;
2656 master_th->th.th_local.this_construct = team->t.t_master_this_cons;
2657
2658 master_th->th.th_dispatch = &parent_team->t.t_dispatch[team->t.t_master_tid];
2659
2660 /* jc: The following lock has instructions with REL and ACQ semantics,
2661 separating the parallel user code called in this parallel region
2662 from the serial user code called after this function returns. */
2664
2665 if (!master_th->th.th_teams_microtask ||
2666 team->t.t_level > master_th->th.th_teams_level) {
2667 /* Decrement our nested depth level */
2668 KMP_ATOMIC_DEC(&root->r.r_in_parallel);
2669 }
2670 KMP_DEBUG_ASSERT(root->r.r_in_parallel >= 0);
2671
2672#if OMPT_SUPPORT
2673 if (ompt_enabled.enabled) {
2675 if (ompt_enabled.ompt_callback_implicit_task) {
2676 int flags = (team_microtask == (void *)__kmp_teams_master)
2677 ? ompt_task_initial
2678 : ompt_task_implicit;
2679 int ompt_team_size = (flags == ompt_task_initial) ? 0 : team->t.t_nproc;
2680 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
2681 ompt_scope_end, NULL, &(task_info->task_data), ompt_team_size,
2682 OMPT_CUR_TASK_INFO(master_th)->thread_num, flags);
2683 }
2684 task_info->frame.exit_frame = ompt_data_none;
2685 task_info->task_data = ompt_data_none;
2686 }
2687#endif
2688
2689 KF_TRACE(10, ("__kmp_join_call1: T#%d, this_thread=%p team=%p\n", 0,
2690 master_th, team));
2692
2693 master_th->th.th_def_allocator = team->t.t_def_allocator;
2694
2695#if OMPD_SUPPORT
2696 if (ompd_state & OMPD_ENABLE_BP)
2697 ompd_bp_parallel_end();
2698#endif
2699 updateHWFPControl(team);
2700
2701 if (root->r.r_active != master_active)
2702 root->r.r_active = master_active;
2703
2704 __kmp_free_team(root, team, master_th); // this will free worker threads
2705
2706 /* this race was fun to find. make sure the following is in the critical
2707 region otherwise assertions may fail occasionally since the old team may be
2708 reallocated and the hierarchy appears inconsistent. it is actually safe to
2709 run and won't cause any bugs, but will cause those assertion failures. it's
2710 only one deref&assign so might as well put this in the critical region */
2711 master_th->th.th_team = parent_team;
2712 master_th->th.th_team_nproc = parent_team->t.t_nproc;
2713 master_th->th.th_team_master = parent_team->t.t_threads[0];
2714 master_th->th.th_team_serialized = parent_team->t.t_serialized;
2715
2716 /* restore serialized team, if need be */
2717 if (parent_team->t.t_serialized &&
2718 parent_team != master_th->th.th_serial_team &&
2719 parent_team != root->r.r_root_team) {
2720 __kmp_free_team(root, master_th->th.th_serial_team, NULL);
2721 master_th->th.th_serial_team = parent_team;
2722 }
2723
2725 // Restore primary thread's task state from team structure
2726 KMP_DEBUG_ASSERT(team->t.t_primary_task_state == 0 ||
2727 team->t.t_primary_task_state == 1);
2728 master_th->th.th_task_state = (kmp_uint8)team->t.t_primary_task_state;
2729
2730 // Copy the task team from the parent team to the primary thread
2731 master_th->th.th_task_team =
2732 parent_team->t.t_task_team[master_th->th.th_task_state];
2733 KA_TRACE(20,
2734 ("__kmp_join_call: Primary T#%d restoring task_team %p, team %p\n",
2735 __kmp_gtid_from_thread(master_th), master_th->th.th_task_team,
2736 parent_team));
2737 }
2738
2739 // TODO: GEH - cannot do this assertion because root thread not set up as
2740 // executing
2741 // KMP_ASSERT( master_th->th.th_current_task->td_flags.executing == 0 );
2742 master_th->th.th_current_task->td_flags.executing = 1;
2743
2745
2746#if KMP_AFFINITY_SUPPORTED
2747 if (master_th->th.th_team->t.t_level == 0 && __kmp_affinity.flags.reset) {
2749 }
2750#endif
2751#if OMPT_SUPPORT
2752 int flags =
2753 OMPT_INVOKER(fork_context) |
2754 ((team_microtask == (void *)__kmp_teams_master) ? ompt_parallel_league
2755 : ompt_parallel_team);
2756 if (ompt_enabled.enabled) {
2757 __kmp_join_ompt(gtid, master_th, parent_team, parallel_data, flags,
2758 codeptr);
2759 }
2760#endif
2761
2762 KMP_MB();
2763 KA_TRACE(20, ("__kmp_join_call: exit T#%d\n", gtid));
2764}
2765
2766/* Check whether we should push an internal control record onto the
2767 serial team stack. If so, do it. */
2769
2770 if (thread->th.th_team != thread->th.th_serial_team) {
2771 return;
2772 }
2773 if (thread->th.th_team->t.t_serialized > 1) {
2774 int push = 0;
2775
2776 if (thread->th.th_team->t.t_control_stack_top == NULL) {
2777 push = 1;
2778 } else {
2779 if (thread->th.th_team->t.t_control_stack_top->serial_nesting_level !=
2780 thread->th.th_team->t.t_serialized) {
2781 push = 1;
2782 }
2783 }
2784 if (push) { /* push a record on the serial team's stack */
2785 kmp_internal_control_t *control =
2787 sizeof(kmp_internal_control_t));
2788
2789 copy_icvs(control, &thread->th.th_current_task->td_icvs);
2790
2791 control->serial_nesting_level = thread->th.th_team->t.t_serialized;
2792
2793 control->next = thread->th.th_team->t.t_control_stack_top;
2794 thread->th.th_team->t.t_control_stack_top = control;
2795 }
2796 }
2797}
2798
2799/* Changes set_nproc */
2800void __kmp_set_num_threads(int new_nth, int gtid) {
2801 kmp_info_t *thread;
2802 kmp_root_t *root;
2803
2804 KF_TRACE(10, ("__kmp_set_num_threads: new __kmp_nth = %d\n", new_nth));
2806
2807 if (new_nth < 1)
2808 new_nth = 1;
2809 else if (new_nth > __kmp_max_nth)
2810 new_nth = __kmp_max_nth;
2811
2812 KMP_COUNT_VALUE(OMP_set_numthreads, new_nth);
2813 thread = __kmp_threads[gtid];
2814 if (thread->th.th_current_task->td_icvs.nproc == new_nth)
2815 return; // nothing to do
2816
2818
2819 set__nproc(thread, new_nth);
2820
2821 // If this omp_set_num_threads() call will cause the hot team size to be
2822 // reduced (in the absence of a num_threads clause), then reduce it now,
2823 // rather than waiting for the next parallel region.
2824 root = thread->th.th_root;
2825 if (__kmp_init_parallel && (!root->r.r_active) &&
2826 (root->r.r_hot_team->t.t_nproc > new_nth) && __kmp_hot_teams_max_level &&
2828 kmp_team_t *hot_team = root->r.r_hot_team;
2829 int f;
2830
2832
2834 __kmp_resize_dist_barrier(hot_team, hot_team->t.t_nproc, new_nth);
2835 }
2836 // Release the extra threads we don't need any more.
2837 for (f = new_nth; f < hot_team->t.t_nproc; f++) {
2838 KMP_DEBUG_ASSERT(hot_team->t.t_threads[f] != NULL);
2840 // When decreasing team size, threads no longer in the team should unref
2841 // task team.
2842 hot_team->t.t_threads[f]->th.th_task_team = NULL;
2843 }
2844 __kmp_free_thread(hot_team->t.t_threads[f]);
2845 hot_team->t.t_threads[f] = NULL;
2846 }
2847 hot_team->t.t_nproc = new_nth;
2848 if (thread->th.th_hot_teams) {
2849 KMP_DEBUG_ASSERT(hot_team == thread->th.th_hot_teams[0].hot_team);
2850 thread->th.th_hot_teams[0].hot_team_nth = new_nth;
2851 }
2852
2854 hot_team->t.b->update_num_threads(new_nth);
2855 __kmp_add_threads_to_team(hot_team, new_nth);
2856 }
2857
2859
2860 // Update the t_nproc field in the threads that are still active.
2861 for (f = 0; f < new_nth; f++) {
2862 KMP_DEBUG_ASSERT(hot_team->t.t_threads[f] != NULL);
2863 hot_team->t.t_threads[f]->th.th_team_nproc = new_nth;
2864 }
2865 // Special flag in case omp_set_num_threads() call
2866 hot_team->t.t_size_changed = -1;
2867 }
2868}
2869
2870/* Changes max_active_levels */
2871void __kmp_set_max_active_levels(int gtid, int max_active_levels) {
2872 kmp_info_t *thread;
2873
2874 KF_TRACE(10, ("__kmp_set_max_active_levels: new max_active_levels for thread "
2875 "%d = (%d)\n",
2876 gtid, max_active_levels));
2878
2879 // validate max_active_levels
2880 if (max_active_levels < 0) {
2881 KMP_WARNING(ActiveLevelsNegative, max_active_levels);
2882 // We ignore this call if the user has specified a negative value.
2883 // The current setting won't be changed. The last valid setting will be
2884 // used. A warning will be issued (if warnings are allowed as controlled by
2885 // the KMP_WARNINGS env var).
2886 KF_TRACE(10, ("__kmp_set_max_active_levels: the call is ignored: new "
2887 "max_active_levels for thread %d = (%d)\n",
2888 gtid, max_active_levels));
2889 return;
2890 }
2891 if (max_active_levels <= KMP_MAX_ACTIVE_LEVELS_LIMIT) {
2892 // it's OK, the max_active_levels is within the valid range: [ 0;
2893 // KMP_MAX_ACTIVE_LEVELS_LIMIT ]
2894 // We allow a zero value. (implementation defined behavior)
2895 } else {
2896 KMP_WARNING(ActiveLevelsExceedLimit, max_active_levels,
2898 max_active_levels = KMP_MAX_ACTIVE_LEVELS_LIMIT;
2899 // Current upper limit is MAX_INT. (implementation defined behavior)
2900 // If the input exceeds the upper limit, we correct the input to be the
2901 // upper limit. (implementation defined behavior)
2902 // Actually, the flow should never get here until we use MAX_INT limit.
2903 }
2904 KF_TRACE(10, ("__kmp_set_max_active_levels: after validation: new "
2905 "max_active_levels for thread %d = (%d)\n",
2906 gtid, max_active_levels));
2907
2908 thread = __kmp_threads[gtid];
2909
2911
2912 set__max_active_levels(thread, max_active_levels);
2913}
2914
2915/* Gets max_active_levels */
2917 kmp_info_t *thread;
2918
2919 KF_TRACE(10, ("__kmp_get_max_active_levels: thread %d\n", gtid));
2921
2922 thread = __kmp_threads[gtid];
2923 KMP_DEBUG_ASSERT(thread->th.th_current_task);
2924 KF_TRACE(10, ("__kmp_get_max_active_levels: thread %d, curtask=%p, "
2925 "curtask_maxaclevel=%d\n",
2926 gtid, thread->th.th_current_task,
2927 thread->th.th_current_task->td_icvs.max_active_levels));
2928 return thread->th.th_current_task->td_icvs.max_active_levels;
2929}
2930
2931// nteams-var per-device ICV
2932void __kmp_set_num_teams(int num_teams) {
2933 if (num_teams > 0)
2934 __kmp_nteams = num_teams;
2935}
2937// teams-thread-limit-var per-device ICV
2939 if (limit > 0)
2941}
2943
2944KMP_BUILD_ASSERT(sizeof(kmp_sched_t) == sizeof(int));
2945KMP_BUILD_ASSERT(sizeof(enum sched_type) == sizeof(int));
2946
2947/* Changes def_sched_var ICV values (run-time schedule kind and chunk) */
2948void __kmp_set_schedule(int gtid, kmp_sched_t kind, int chunk) {
2949 kmp_info_t *thread;
2950 kmp_sched_t orig_kind;
2951 // kmp_team_t *team;
2952
2953 KF_TRACE(10, ("__kmp_set_schedule: new schedule for thread %d = (%d, %d)\n",
2954 gtid, (int)kind, chunk));
2956
2957 // Check if the kind parameter is valid, correct if needed.
2958 // Valid parameters should fit in one of two intervals - standard or extended:
2959 // <lower>, <valid>, <upper_std>, <lower_ext>, <valid>, <upper>
2960 // 2008-01-25: 0, 1 - 4, 5, 100, 101 - 102, 103
2961 orig_kind = kind;
2962 kind = __kmp_sched_without_mods(kind);
2963
2964 if (kind <= kmp_sched_lower || kind >= kmp_sched_upper ||
2965 (kind <= kmp_sched_lower_ext && kind >= kmp_sched_upper_std)) {
2966 // TODO: Hint needs attention in case we change the default schedule.
2967 __kmp_msg(kmp_ms_warning, KMP_MSG(ScheduleKindOutOfRange, kind),
2968 KMP_HNT(DefaultScheduleKindUsed, "static, no chunk"),
2970 kind = kmp_sched_default;
2971 chunk = 0; // ignore chunk value in case of bad kind
2972 }
2973
2974 thread = __kmp_threads[gtid];
2975
2977
2978 if (kind < kmp_sched_upper_std) {
2979 if (kind == kmp_sched_static && chunk < KMP_DEFAULT_CHUNK) {
2980 // differ static chunked vs. unchunked: chunk should be invalid to
2981 // indicate unchunked schedule (which is the default)
2982 thread->th.th_current_task->td_icvs.sched.r_sched_type = kmp_sch_static;
2983 } else {
2984 thread->th.th_current_task->td_icvs.sched.r_sched_type =
2985 __kmp_sch_map[kind - kmp_sched_lower - 1];
2986 }
2987 } else {
2988 // __kmp_sch_map[ kind - kmp_sched_lower_ext + kmp_sched_upper_std -
2989 // kmp_sched_lower - 2 ];
2990 thread->th.th_current_task->td_icvs.sched.r_sched_type =
2992 kmp_sched_lower - 2];
2993 }
2995 orig_kind, &(thread->th.th_current_task->td_icvs.sched.r_sched_type));
2996 if (kind == kmp_sched_auto || chunk < 1) {
2997 // ignore parameter chunk for schedule auto
2998 thread->th.th_current_task->td_icvs.sched.chunk = KMP_DEFAULT_CHUNK;
2999 } else {
3000 thread->th.th_current_task->td_icvs.sched.chunk = chunk;
3001 }
3002}
3003
3004/* Gets def_sched_var ICV values */
3005void __kmp_get_schedule(int gtid, kmp_sched_t *kind, int *chunk) {
3006 kmp_info_t *thread;
3007 enum sched_type th_type;
3008
3009 KF_TRACE(10, ("__kmp_get_schedule: thread %d\n", gtid));
3011
3012 thread = __kmp_threads[gtid];
3013
3014 th_type = thread->th.th_current_task->td_icvs.sched.r_sched_type;
3015 switch (SCHEDULE_WITHOUT_MODIFIERS(th_type)) {
3016 case kmp_sch_static:
3019 *kind = kmp_sched_static;
3020 __kmp_sched_apply_mods_stdkind(kind, th_type);
3021 *chunk = 0; // chunk was not set, try to show this fact via zero value
3022 return;
3024 *kind = kmp_sched_static;
3025 break;
3027 *kind = kmp_sched_dynamic;
3028 break;
3032 *kind = kmp_sched_guided;
3033 break;
3034 case kmp_sch_auto:
3035 *kind = kmp_sched_auto;
3036 break;
3038 *kind = kmp_sched_trapezoidal;
3039 break;
3040#if KMP_STATIC_STEAL_ENABLED
3042 *kind = kmp_sched_static_steal;
3043 break;
3044#endif
3045 default:
3046 KMP_FATAL(UnknownSchedulingType, th_type);
3047 }
3048
3049 __kmp_sched_apply_mods_stdkind(kind, th_type);
3050 *chunk = thread->th.th_current_task->td_icvs.sched.chunk;
3051}
3052
3054
3055 int ii, dd;
3056 kmp_team_t *team;
3057 kmp_info_t *thr;
3058
3059 KF_TRACE(10, ("__kmp_get_ancestor_thread_num: thread %d %d\n", gtid, level));
3061
3062 // validate level
3063 if (level == 0)
3064 return 0;
3065 if (level < 0)
3066 return -1;
3067 thr = __kmp_threads[gtid];
3068 team = thr->th.th_team;
3069 ii = team->t.t_level;
3070 if (level > ii)
3071 return -1;
3072
3073 if (thr->th.th_teams_microtask) {
3074 // AC: we are in teams region where multiple nested teams have same level
3075 int tlevel = thr->th.th_teams_level; // the level of the teams construct
3076 if (level <=
3077 tlevel) { // otherwise usual algorithm works (will not touch the teams)
3078 KMP_DEBUG_ASSERT(ii >= tlevel);
3079 // AC: As we need to pass by the teams league, we need to artificially
3080 // increase ii
3081 if (ii == tlevel) {
3082 ii += 2; // three teams have same level
3083 } else {
3084 ii++; // two teams have same level
3085 }
3086 }
3087 }
3088
3089 if (ii == level)
3090 return __kmp_tid_from_gtid(gtid);
3091
3092 dd = team->t.t_serialized;
3093 level++;
3094 while (ii > level) {
3095 for (dd = team->t.t_serialized; (dd > 0) && (ii > level); dd--, ii--) {
3096 }
3097 if ((team->t.t_serialized) && (!dd)) {
3098 team = team->t.t_parent;
3099 continue;
3100 }
3101 if (ii > level) {
3102 team = team->t.t_parent;
3103 dd = team->t.t_serialized;
3104 ii--;
3105 }
3106 }
3107
3108 return (dd > 1) ? (0) : (team->t.t_master_tid);
3109}
3110
3111int __kmp_get_team_size(int gtid, int level) {
3112
3113 int ii, dd;
3114 kmp_team_t *team;
3115 kmp_info_t *thr;
3116
3117 KF_TRACE(10, ("__kmp_get_team_size: thread %d %d\n", gtid, level));
3119
3120 // validate level
3121 if (level == 0)
3122 return 1;
3123 if (level < 0)
3124 return -1;
3125 thr = __kmp_threads[gtid];
3126 team = thr->th.th_team;
3127 ii = team->t.t_level;
3128 if (level > ii)
3129 return -1;
3130
3131 if (thr->th.th_teams_microtask) {
3132 // AC: we are in teams region where multiple nested teams have same level
3133 int tlevel = thr->th.th_teams_level; // the level of the teams construct
3134 if (level <=
3135 tlevel) { // otherwise usual algorithm works (will not touch the teams)
3136 KMP_DEBUG_ASSERT(ii >= tlevel);
3137 // AC: As we need to pass by the teams league, we need to artificially
3138 // increase ii
3139 if (ii == tlevel) {
3140 ii += 2; // three teams have same level
3141 } else {
3142 ii++; // two teams have same level
3143 }
3144 }
3145 }
3146
3147 while (ii > level) {
3148 for (dd = team->t.t_serialized; (dd > 0) && (ii > level); dd--, ii--) {
3149 }
3150 if (team->t.t_serialized && (!dd)) {
3151 team = team->t.t_parent;
3152 continue;
3153 }
3154 if (ii > level) {
3155 team = team->t.t_parent;
3156 ii--;
3157 }
3158 }
3159
3160 return team->t.t_nproc;
3161}
3162
3164 // This routine created because pairs (__kmp_sched, __kmp_chunk) and
3165 // (__kmp_static, __kmp_guided) may be changed by kmp_set_defaults
3166 // independently. So one can get the updated schedule here.
3167
3168 kmp_r_sched_t r_sched;
3169
3170 // create schedule from 4 globals: __kmp_sched, __kmp_chunk, __kmp_static,
3171 // __kmp_guided. __kmp_sched should keep original value, so that user can set
3172 // KMP_SCHEDULE multiple times, and thus have different run-time schedules in
3173 // different roots (even in OMP 2.5)
3175 enum sched_type sched_modifiers = SCHEDULE_GET_MODIFIERS(__kmp_sched);
3176 if (s == kmp_sch_static) {
3177 // replace STATIC with more detailed schedule (balanced or greedy)
3178 r_sched.r_sched_type = __kmp_static;
3179 } else if (s == kmp_sch_guided_chunked) {
3180 // replace GUIDED with more detailed schedule (iterative or analytical)
3181 r_sched.r_sched_type = __kmp_guided;
3182 } else { // (STATIC_CHUNKED), or (DYNAMIC_CHUNKED), or other
3183 r_sched.r_sched_type = __kmp_sched;
3184 }
3185 SCHEDULE_SET_MODIFIERS(r_sched.r_sched_type, sched_modifiers);
3186
3188 // __kmp_chunk may be wrong here (if it was not ever set)
3189 r_sched.chunk = KMP_DEFAULT_CHUNK;
3190 } else {
3191 r_sched.chunk = __kmp_chunk;
3192 }
3193
3194 return r_sched;
3195}
3196
3197/* Allocate (realloc == FALSE) * or reallocate (realloc == TRUE)
3198 at least argc number of *t_argv entries for the requested team. */
3199static void __kmp_alloc_argv_entries(int argc, kmp_team_t *team, int realloc) {
3200
3201 KMP_DEBUG_ASSERT(team);
3202 if (!realloc || argc > team->t.t_max_argc) {
3203
3204 KA_TRACE(100, ("__kmp_alloc_argv_entries: team %d: needed entries=%d, "
3205 "current entries=%d\n",
3206 team->t.t_id, argc, (realloc) ? team->t.t_max_argc : 0));
3207 /* if previously allocated heap space for args, free them */
3208 if (realloc && team->t.t_argv != &team->t.t_inline_argv[0])
3209 __kmp_free((void *)team->t.t_argv);
3210
3211 if (argc <= KMP_INLINE_ARGV_ENTRIES) {
3212 /* use unused space in the cache line for arguments */
3213 team->t.t_max_argc = KMP_INLINE_ARGV_ENTRIES;
3214 KA_TRACE(100, ("__kmp_alloc_argv_entries: team %d: inline allocate %d "
3215 "argv entries\n",
3216 team->t.t_id, team->t.t_max_argc));
3217 team->t.t_argv = &team->t.t_inline_argv[0];
3218 if (__kmp_storage_map) {
3220 -1, &team->t.t_inline_argv[0],
3221 &team->t.t_inline_argv[KMP_INLINE_ARGV_ENTRIES],
3222 (sizeof(void *) * KMP_INLINE_ARGV_ENTRIES), "team_%d.t_inline_argv",
3223 team->t.t_id);
3224 }
3225 } else {
3226 /* allocate space for arguments in the heap */
3227 team->t.t_max_argc = (argc <= (KMP_MIN_MALLOC_ARGV_ENTRIES >> 1))
3229 : 2 * argc;
3230 KA_TRACE(100, ("__kmp_alloc_argv_entries: team %d: dynamic allocate %d "
3231 "argv entries\n",
3232 team->t.t_id, team->t.t_max_argc));
3233 team->t.t_argv =
3234 (void **)__kmp_page_allocate(sizeof(void *) * team->t.t_max_argc);
3235 if (__kmp_storage_map) {
3236 __kmp_print_storage_map_gtid(-1, &team->t.t_argv[0],
3237 &team->t.t_argv[team->t.t_max_argc],
3238 sizeof(void *) * team->t.t_max_argc,
3239 "team_%d.t_argv", team->t.t_id);
3240 }
3241 }
3242 }
3243}
3244
3245static void __kmp_allocate_team_arrays(kmp_team_t *team, int max_nth) {
3246 int i;
3247 int num_disp_buff = max_nth > 1 ? __kmp_dispatch_num_buffers : 2;
3248 team->t.t_threads =
3249 (kmp_info_t **)__kmp_allocate(sizeof(kmp_info_t *) * max_nth);
3250 team->t.t_disp_buffer = (dispatch_shared_info_t *)__kmp_allocate(
3251 sizeof(dispatch_shared_info_t) * num_disp_buff);
3252 team->t.t_dispatch =
3253 (kmp_disp_t *)__kmp_allocate(sizeof(kmp_disp_t) * max_nth);
3254 team->t.t_implicit_task_taskdata =
3255 (kmp_taskdata_t *)__kmp_allocate(sizeof(kmp_taskdata_t) * max_nth);
3256 team->t.t_max_nproc = max_nth;
3257
3258 /* setup dispatch buffers */
3259 for (i = 0; i < num_disp_buff; ++i) {
3260 team->t.t_disp_buffer[i].buffer_index = i;
3261 team->t.t_disp_buffer[i].doacross_buf_idx = i;
3262 }
3263}
3264
3266 /* Note: this does not free the threads in t_threads (__kmp_free_threads) */
3267 int i;
3268 for (i = 0; i < team->t.t_max_nproc; ++i) {
3269 if (team->t.t_dispatch[i].th_disp_buffer != NULL) {
3270 __kmp_free(team->t.t_dispatch[i].th_disp_buffer);
3271 team->t.t_dispatch[i].th_disp_buffer = NULL;
3272 }
3273 }
3274#if KMP_USE_HIER_SCHED
3276#endif
3277 __kmp_free(team->t.t_threads);
3278 __kmp_free(team->t.t_disp_buffer);
3279 __kmp_free(team->t.t_dispatch);
3280 __kmp_free(team->t.t_implicit_task_taskdata);
3281 team->t.t_threads = NULL;
3282 team->t.t_disp_buffer = NULL;
3283 team->t.t_dispatch = NULL;
3284 team->t.t_implicit_task_taskdata = 0;
3285}
3286
3287static void __kmp_reallocate_team_arrays(kmp_team_t *team, int max_nth) {
3288 kmp_info_t **oldThreads = team->t.t_threads;
3289
3290 __kmp_free(team->t.t_disp_buffer);
3291 __kmp_free(team->t.t_dispatch);
3292 __kmp_free(team->t.t_implicit_task_taskdata);
3293 __kmp_allocate_team_arrays(team, max_nth);
3294
3295 KMP_MEMCPY(team->t.t_threads, oldThreads,
3296 team->t.t_nproc * sizeof(kmp_info_t *));
3297
3298 __kmp_free(oldThreads);
3299}
3300
3302
3303 kmp_r_sched_t r_sched =
3304 __kmp_get_schedule_global(); // get current state of scheduling globals
3305
3307
3308 kmp_internal_control_t g_icvs = {
3309 0, // int serial_nesting_level; //corresponds to value of th_team_serialized
3310 (kmp_int8)__kmp_global.g.g_dynamic, // internal control for dynamic
3311 // adjustment of threads (per thread)
3312 (kmp_int8)__kmp_env_blocktime, // int bt_set; //internal control for
3313 // whether blocktime is explicitly set
3314 __kmp_dflt_blocktime, // int blocktime; //internal control for blocktime
3315#if KMP_USE_MONITOR
3316 __kmp_bt_intervals, // int bt_intervals; //internal control for blocktime
3317// intervals
3318#endif
3319 __kmp_dflt_team_nth, // int nproc; //internal control for # of threads for
3320 // next parallel region (per thread)
3321 // (use a max ub on value if __kmp_parallel_initialize not called yet)
3322 __kmp_cg_max_nth, // int thread_limit;
3323 __kmp_task_max_nth, // int task_thread_limit; // to set the thread_limit
3324 // on task. This is used in the case of target thread_limit
3325 __kmp_dflt_max_active_levels, // int max_active_levels; //internal control
3326 // for max_active_levels
3327 r_sched, // kmp_r_sched_t sched; //internal control for runtime schedule
3328 // {sched,chunk} pair
3329 __kmp_nested_proc_bind.bind_types[0],
3331 NULL // struct kmp_internal_control *next;
3332 };
3333
3334 return g_icvs;
3335}
3336
3338
3339 kmp_internal_control_t gx_icvs;
3340 gx_icvs.serial_nesting_level =
3341 0; // probably =team->t.t_serial like in save_inter_controls
3342 copy_icvs(&gx_icvs, &team->t.t_threads[0]->th.th_current_task->td_icvs);
3343 gx_icvs.next = NULL;
3344
3345 return gx_icvs;
3346}
3347
3349 int f;
3350 kmp_team_t *root_team;
3351 kmp_team_t *hot_team;
3352 int hot_team_max_nth;
3353 kmp_r_sched_t r_sched =
3354 __kmp_get_schedule_global(); // get current state of scheduling globals
3356 KMP_DEBUG_ASSERT(root);
3357 KMP_ASSERT(!root->r.r_begin);
3358
3359 /* setup the root state structure */
3360 __kmp_init_lock(&root->r.r_begin_lock);
3361 root->r.r_begin = FALSE;
3362 root->r.r_active = FALSE;
3363 root->r.r_in_parallel = 0;
3364 root->r.r_blocktime = __kmp_dflt_blocktime;
3365#if KMP_AFFINITY_SUPPORTED
3366 root->r.r_affinity_assigned = FALSE;
3367#endif
3368
3369 /* setup the root team for this task */
3370 /* allocate the root team structure */
3371 KF_TRACE(10, ("__kmp_initialize_root: before root_team\n"));
3372
3373 root_team = __kmp_allocate_team(root,
3374 1, // new_nproc
3375 1, // max_nproc
3376#if OMPT_SUPPORT
3377 ompt_data_none, // root parallel id
3378#endif
3379 __kmp_nested_proc_bind.bind_types[0], &r_icvs,
3380 0, // argc
3381 NULL // primary thread is unknown
3382 );
3383#if USE_DEBUGGER
3384 // Non-NULL value should be assigned to make the debugger display the root
3385 // team.
3386 TCW_SYNC_PTR(root_team->t.t_pkfn, (microtask_t)(~0));
3387#endif
3388
3389 KF_TRACE(10, ("__kmp_initialize_root: after root_team = %p\n", root_team));
3390
3391 root->r.r_root_team = root_team;
3392 root_team->t.t_control_stack_top = NULL;
3393
3394 /* initialize root team */
3395 root_team->t.t_threads[0] = NULL;
3396 root_team->t.t_nproc = 1;
3397 root_team->t.t_serialized = 1;
3398 // TODO???: root_team->t.t_max_active_levels = __kmp_dflt_max_active_levels;
3399 root_team->t.t_sched.sched = r_sched.sched;
3400 root_team->t.t_nested_nth = &__kmp_nested_nth;
3401 KA_TRACE(
3402 20,
3403 ("__kmp_initialize_root: init root team %d arrived: join=%u, plain=%u\n",
3405
3406 /* setup the hot team for this task */
3407 /* allocate the hot team structure */
3408 KF_TRACE(10, ("__kmp_initialize_root: before hot_team\n"));
3409
3410 hot_team = __kmp_allocate_team(root,
3411 1, // new_nproc
3412 __kmp_dflt_team_nth_ub * 2, // max_nproc
3413#if OMPT_SUPPORT
3414 ompt_data_none, // root parallel id
3415#endif
3416 __kmp_nested_proc_bind.bind_types[0], &r_icvs,
3417 0, // argc
3418 NULL // primary thread is unknown
3419 );
3420 KF_TRACE(10, ("__kmp_initialize_root: after hot_team = %p\n", hot_team));
3421
3422 root->r.r_hot_team = hot_team;
3423 root_team->t.t_control_stack_top = NULL;
3424
3425 /* first-time initialization */
3426 hot_team->t.t_parent = root_team;
3427
3428 /* initialize hot team */
3429 hot_team_max_nth = hot_team->t.t_max_nproc;
3430 for (f = 0; f < hot_team_max_nth; ++f) {
3431 hot_team->t.t_threads[f] = NULL;
3432 }
3433 hot_team->t.t_nproc = 1;
3434 // TODO???: hot_team->t.t_max_active_levels = __kmp_dflt_max_active_levels;
3435 hot_team->t.t_sched.sched = r_sched.sched;
3436 hot_team->t.t_size_changed = 0;
3437 hot_team->t.t_nested_nth = &__kmp_nested_nth;
3438}
3439
3440#ifdef KMP_DEBUG
3441
3442typedef struct kmp_team_list_item {
3443 kmp_team_p const *entry;
3444 struct kmp_team_list_item *next;
3445} kmp_team_list_item_t;
3446typedef kmp_team_list_item_t *kmp_team_list_t;
3447
3448static void __kmp_print_structure_team_accum( // Add team to list of teams.
3449 kmp_team_list_t list, // List of teams.
3450 kmp_team_p const *team // Team to add.
3451) {
3452
3453 // List must terminate with item where both entry and next are NULL.
3454 // Team is added to the list only once.
3455 // List is sorted in ascending order by team id.
3456 // Team id is *not* a key.
3457
3458 kmp_team_list_t l;
3459
3460 KMP_DEBUG_ASSERT(list != NULL);
3461 if (team == NULL) {
3462 return;
3463 }
3464
3465 __kmp_print_structure_team_accum(list, team->t.t_parent);
3466 __kmp_print_structure_team_accum(list, team->t.t_next_pool);
3467
3468 // Search list for the team.
3469 l = list;
3470 while (l->next != NULL && l->entry != team) {
3471 l = l->next;
3472 }
3473 if (l->next != NULL) {
3474 return; // Team has been added before, exit.
3475 }
3476
3477 // Team is not found. Search list again for insertion point.
3478 l = list;
3479 while (l->next != NULL && l->entry->t.t_id <= team->t.t_id) {
3480 l = l->next;
3481 }
3482
3483 // Insert team.
3484 {
3485 kmp_team_list_item_t *item = (kmp_team_list_item_t *)KMP_INTERNAL_MALLOC(
3486 sizeof(kmp_team_list_item_t));
3487 *item = *l;
3488 l->entry = team;
3489 l->next = item;
3490 }
3491}
3492
3493static void __kmp_print_structure_team(char const *title, kmp_team_p const *team
3494
3495) {
3496 __kmp_printf("%s", title);
3497 if (team != NULL) {
3498 __kmp_printf("%2x %p\n", team->t.t_id, team);
3499 } else {
3500 __kmp_printf(" - (nil)\n");
3501 }
3502}
3503
3504static void __kmp_print_structure_thread(char const *title,
3505 kmp_info_p const *thread) {
3506 __kmp_printf("%s", title);
3507 if (thread != NULL) {
3508 __kmp_printf("%2d %p\n", thread->th.th_info.ds.ds_gtid, thread);
3509 } else {
3510 __kmp_printf(" - (nil)\n");
3511 }
3512}
3513
3514void __kmp_print_structure(void) {
3515
3516 kmp_team_list_t list;
3517
3518 // Initialize list of teams.
3519 list =
3520 (kmp_team_list_item_t *)KMP_INTERNAL_MALLOC(sizeof(kmp_team_list_item_t));
3521 list->entry = NULL;
3522 list->next = NULL;
3523
3524 __kmp_printf("\n------------------------------\nGlobal Thread "
3525 "Table\n------------------------------\n");
3526 {
3527 int gtid;
3528 for (gtid = 0; gtid < __kmp_threads_capacity; ++gtid) {
3529 __kmp_printf("%2d", gtid);
3530 if (__kmp_threads != NULL) {
3531 __kmp_printf(" %p", __kmp_threads[gtid]);
3532 }
3533 if (__kmp_root != NULL) {
3534 __kmp_printf(" %p", __kmp_root[gtid]);
3535 }
3536 __kmp_printf("\n");
3537 }
3538 }
3539
3540 // Print out __kmp_threads array.
3541 __kmp_printf("\n------------------------------\nThreads\n--------------------"
3542 "----------\n");
3543 if (__kmp_threads != NULL) {
3544 int gtid;
3545 for (gtid = 0; gtid < __kmp_threads_capacity; ++gtid) {
3546 kmp_info_t const *thread = __kmp_threads[gtid];
3547 if (thread != NULL) {
3548 __kmp_printf("GTID %2d %p:\n", gtid, thread);
3549 __kmp_printf(" Our Root: %p\n", thread->th.th_root);
3550 __kmp_print_structure_team(" Our Team: ", thread->th.th_team);
3551 __kmp_print_structure_team(" Serial Team: ",
3552 thread->th.th_serial_team);
3553 __kmp_printf(" Threads: %2d\n", thread->th.th_team_nproc);
3554 __kmp_print_structure_thread(" Primary: ",
3555 thread->th.th_team_master);
3556 __kmp_printf(" Serialized?: %2d\n", thread->th.th_team_serialized);
3557 __kmp_printf(" Set NProc: %2d\n", thread->th.th_set_nproc);
3558 __kmp_printf(" Set Proc Bind: %2d\n", thread->th.th_set_proc_bind);
3559 __kmp_print_structure_thread(" Next in pool: ",
3560 thread->th.th_next_pool);
3561 __kmp_printf("\n");
3562 __kmp_print_structure_team_accum(list, thread->th.th_team);
3563 __kmp_print_structure_team_accum(list, thread->th.th_serial_team);
3564 }
3565 }
3566 } else {
3567 __kmp_printf("Threads array is not allocated.\n");
3568 }
3569
3570 // Print out __kmp_root array.
3571 __kmp_printf("\n------------------------------\nUbers\n----------------------"
3572 "--------\n");
3573 if (__kmp_root != NULL) {
3574 int gtid;
3575 for (gtid = 0; gtid < __kmp_threads_capacity; ++gtid) {
3576 kmp_root_t const *root = __kmp_root[gtid];
3577 if (root != NULL) {
3578 __kmp_printf("GTID %2d %p:\n", gtid, root);
3579 __kmp_print_structure_team(" Root Team: ", root->r.r_root_team);
3580 __kmp_print_structure_team(" Hot Team: ", root->r.r_hot_team);
3581 __kmp_print_structure_thread(" Uber Thread: ",
3582 root->r.r_uber_thread);
3583 __kmp_printf(" Active?: %2d\n", root->r.r_active);
3584 __kmp_printf(" In Parallel: %2d\n",
3585 KMP_ATOMIC_LD_RLX(&root->r.r_in_parallel));
3586 __kmp_printf("\n");
3587 __kmp_print_structure_team_accum(list, root->r.r_root_team);
3588 __kmp_print_structure_team_accum(list, root->r.r_hot_team);
3589 }
3590 }
3591 } else {
3592 __kmp_printf("Ubers array is not allocated.\n");
3593 }
3594
3595 __kmp_printf("\n------------------------------\nTeams\n----------------------"
3596 "--------\n");
3597 while (list->next != NULL) {
3598 kmp_team_p const *team = list->entry;
3599 int i;
3600 __kmp_printf("Team %2x %p:\n", team->t.t_id, team);
3601 __kmp_print_structure_team(" Parent Team: ", team->t.t_parent);
3602 __kmp_printf(" Primary TID: %2d\n", team->t.t_master_tid);
3603 __kmp_printf(" Max threads: %2d\n", team->t.t_max_nproc);
3604 __kmp_printf(" Levels of serial: %2d\n", team->t.t_serialized);
3605 __kmp_printf(" Number threads: %2d\n", team->t.t_nproc);
3606 for (i = 0; i < team->t.t_nproc; ++i) {
3607 __kmp_printf(" Thread %2d: ", i);
3608 __kmp_print_structure_thread("", team->t.t_threads[i]);
3609 }
3610 __kmp_print_structure_team(" Next in pool: ", team->t.t_next_pool);
3611 __kmp_printf("\n");
3612 list = list->next;
3613 }
3614
3615 // Print out __kmp_thread_pool and __kmp_team_pool.
3616 __kmp_printf("\n------------------------------\nPools\n----------------------"
3617 "--------\n");
3618 __kmp_print_structure_thread("Thread pool: ",
3620 __kmp_print_structure_team("Team pool: ",
3622 __kmp_printf("\n");
3623
3624 // Free team list.
3625 while (list != NULL) {
3626 kmp_team_list_item_t *item = list;
3627 list = list->next;
3628 KMP_INTERNAL_FREE(item);
3629 }
3630}
3631
3632#endif
3633
3634//---------------------------------------------------------------------------
3635// Stuff for per-thread fast random number generator
3636// Table of primes
3637static const unsigned __kmp_primes[] = {
3638 0x9e3779b1, 0xffe6cc59, 0x2109f6dd, 0x43977ab5, 0xba5703f5, 0xb495a877,
3639 0xe1626741, 0x79695e6b, 0xbc98c09f, 0xd5bee2b3, 0x287488f9, 0x3af18231,
3640 0x9677cd4d, 0xbe3a6929, 0xadc6a877, 0xdcf0674b, 0xbe4d6fe9, 0x5f15e201,
3641 0x99afc3fd, 0xf3f16801, 0xe222cfff, 0x24ba5fdb, 0x0620452d, 0x79f149e3,
3642 0xc8b93f49, 0x972702cd, 0xb07dd827, 0x6c97d5ed, 0x085a3d61, 0x46eb5ea7,
3643 0x3d9910ed, 0x2e687b5b, 0x29609227, 0x6eb081f1, 0x0954c4e1, 0x9d114db9,
3644 0x542acfa9, 0xb3e6bd7b, 0x0742d917, 0xe9f3ffa7, 0x54581edb, 0xf2480f45,
3645 0x0bb9288f, 0xef1affc7, 0x85fa0ca7, 0x3ccc14db, 0xe6baf34b, 0x343377f7,
3646 0x5ca19031, 0xe6d9293b, 0xf0a9f391, 0x5d2e980b, 0xfc411073, 0xc3749363,
3647 0xb892d829, 0x3549366b, 0x629750ad, 0xb98294e5, 0x892d9483, 0xc235baf3,
3648 0x3d2402a3, 0x6bdef3c9, 0xbec333cd, 0x40c9520f};
3649
3650//---------------------------------------------------------------------------
3651// __kmp_get_random: Get a random number using a linear congruential method.
3652unsigned short __kmp_get_random(kmp_info_t *thread) {
3653 unsigned x = thread->th.th_x;
3654 unsigned short r = (unsigned short)(x >> 16);
3655
3656 thread->th.th_x = x * thread->th.th_a + 1;
3657
3658 KA_TRACE(30, ("__kmp_get_random: THREAD: %d, RETURN: %u\n",
3659 thread->th.th_info.ds.ds_tid, r));
3660
3661 return r;
3662}
3663//--------------------------------------------------------
3664// __kmp_init_random: Initialize a random number generator
3666 unsigned seed = thread->th.th_info.ds.ds_tid;
3667
3668 thread->th.th_a =
3669 __kmp_primes[seed % (sizeof(__kmp_primes) / sizeof(__kmp_primes[0]))];
3670 thread->th.th_x = (seed + 1) * thread->th.th_a + 1;
3671 KA_TRACE(30,
3672 ("__kmp_init_random: THREAD: %u; A: %u\n", seed, thread->th.th_a));
3673}
3674
3675#if KMP_OS_WINDOWS
3676/* reclaim array entries for root threads that are already dead, returns number
3677 * reclaimed */
3678static int __kmp_reclaim_dead_roots(void) {
3679 int i, r = 0;
3680
3681 for (i = 0; i < __kmp_threads_capacity; ++i) {
3682 if (KMP_UBER_GTID(i) &&
3684 !__kmp_root[i]
3685 ->r.r_active) { // AC: reclaim only roots died in non-active state
3686 r += __kmp_unregister_root_other_thread(i);
3687 }
3688 }
3689 return r;
3690}
3691#endif
3692
3693/* This function attempts to create free entries in __kmp_threads and
3694 __kmp_root, and returns the number of free entries generated.
3695
3696 For Windows* OS static library, the first mechanism used is to reclaim array
3697 entries for root threads that are already dead.
3698
3699 On all platforms, expansion is attempted on the arrays __kmp_threads_ and
3700 __kmp_root, with appropriate update to __kmp_threads_capacity. Array
3701 capacity is increased by doubling with clipping to __kmp_tp_capacity, if
3702 threadprivate cache array has been created. Synchronization with
3703 __kmpc_threadprivate_cached is done using __kmp_tp_cached_lock.
3704
3705 After any dead root reclamation, if the clipping value allows array expansion
3706 to result in the generation of a total of nNeed free slots, the function does
3707 that expansion. If not, nothing is done beyond the possible initial root
3708 thread reclamation.
3709
3710 If any argument is negative, the behavior is undefined. */
3711static int __kmp_expand_threads(int nNeed) {
3712 int added = 0;
3713 int minimumRequiredCapacity;
3714 int newCapacity;
3715 kmp_info_t **newThreads;
3716 kmp_root_t **newRoot;
3717
3718 // All calls to __kmp_expand_threads should be under __kmp_forkjoin_lock, so
3719 // resizing __kmp_threads does not need additional protection if foreign
3720 // threads are present
3721
3722#if KMP_OS_WINDOWS && !KMP_DYNAMIC_LIB
3723 /* only for Windows static library */
3724 /* reclaim array entries for root threads that are already dead */
3725 added = __kmp_reclaim_dead_roots();
3726
3727 if (nNeed) {
3728 nNeed -= added;
3729 if (nNeed < 0)
3730 nNeed = 0;
3731 }
3732#endif
3733 if (nNeed <= 0)
3734 return added;
3735
3736 // Note that __kmp_threads_capacity is not bounded by __kmp_max_nth. If
3737 // __kmp_max_nth is set to some value less than __kmp_sys_max_nth by the
3738 // user via KMP_DEVICE_THREAD_LIMIT, then __kmp_threads_capacity may become
3739 // > __kmp_max_nth in one of two ways:
3740 //
3741 // 1) The initialization thread (gtid = 0) exits. __kmp_threads[0]
3742 // may not be reused by another thread, so we may need to increase
3743 // __kmp_threads_capacity to __kmp_max_nth + 1.
3744 //
3745 // 2) New foreign root(s) are encountered. We always register new foreign
3746 // roots. This may cause a smaller # of threads to be allocated at
3747 // subsequent parallel regions, but the worker threads hang around (and
3748 // eventually go to sleep) and need slots in the __kmp_threads[] array.
3749 //
3750 // Anyway, that is the reason for moving the check to see if
3751 // __kmp_max_nth was exceeded into __kmp_reserve_threads()
3752 // instead of having it performed here. -BB
3753
3755
3756 /* compute expansion headroom to check if we can expand */
3758 /* possible expansion too small -- give up */
3759 return added;
3760 }
3761 minimumRequiredCapacity = __kmp_threads_capacity + nNeed;
3762
3763 newCapacity = __kmp_threads_capacity;
3764 do {
3765 newCapacity = newCapacity <= (__kmp_sys_max_nth >> 1) ? (newCapacity << 1)
3767 } while (newCapacity < minimumRequiredCapacity);
3768 newThreads = (kmp_info_t **)__kmp_allocate(
3769 (sizeof(kmp_info_t *) + sizeof(kmp_root_t *)) * newCapacity + CACHE_LINE);
3770 newRoot =
3771 (kmp_root_t **)((char *)newThreads + sizeof(kmp_info_t *) * newCapacity);
3772 KMP_MEMCPY(newThreads, __kmp_threads,
3774 KMP_MEMCPY(newRoot, __kmp_root,
3776 // Put old __kmp_threads array on a list. Any ongoing references to the old
3777 // list will be valid. This list is cleaned up at library shutdown.
3780 node->threads = __kmp_threads;
3783
3784 *(kmp_info_t * *volatile *)&__kmp_threads = newThreads;
3785 *(kmp_root_t * *volatile *)&__kmp_root = newRoot;
3786 added += newCapacity - __kmp_threads_capacity;
3787 *(volatile int *)&__kmp_threads_capacity = newCapacity;
3788
3789 if (newCapacity > __kmp_tp_capacity) {
3791 if (__kmp_tp_cached && newCapacity > __kmp_tp_capacity) {
3793 } else { // increase __kmp_tp_capacity to correspond with kmp_threads size
3794 *(volatile int *)&__kmp_tp_capacity = newCapacity;
3795 }
3797 }
3798
3799 return added;
3800}
3801
3802/* Register the current thread as a root thread and obtain our gtid. We must
3803 have the __kmp_initz_lock held at this point. Argument TRUE only if are the
3804 thread that calls from __kmp_do_serial_initialize() */
3805int __kmp_register_root(int initial_thread) {
3806 kmp_info_t *root_thread;
3807 kmp_root_t *root;
3808 int gtid;
3809 int capacity;
3811 KA_TRACE(20, ("__kmp_register_root: entered\n"));
3812 KMP_MB();
3813
3814 /* 2007-03-02:
3815 If initial thread did not invoke OpenMP RTL yet, and this thread is not an
3816 initial one, "__kmp_all_nth >= __kmp_threads_capacity" condition does not
3817 work as expected -- it may return false (that means there is at least one
3818 empty slot in __kmp_threads array), but it is possible the only free slot
3819 is #0, which is reserved for initial thread and so cannot be used for this
3820 one. Following code workarounds this bug.
3821
3822 However, right solution seems to be not reserving slot #0 for initial
3823 thread because:
3824 (1) there is no magic in slot #0,
3825 (2) we cannot detect initial thread reliably (the first thread which does
3826 serial initialization may be not a real initial thread).
3827 */
3828 capacity = __kmp_threads_capacity;
3829 if (!initial_thread && TCR_PTR(__kmp_threads[0]) == NULL) {
3830 --capacity;
3831 }
3832
3833 // If it is not for initializing the hidden helper team, we need to take
3834 // __kmp_hidden_helper_threads_num out of the capacity because it is included
3835 // in __kmp_threads_capacity.
3838 }
3839
3840 /* see if there are too many threads */
3841 if (__kmp_all_nth >= capacity && !__kmp_expand_threads(1)) {
3842 if (__kmp_tp_cached) {
3843 __kmp_fatal(KMP_MSG(CantRegisterNewThread),
3844 KMP_HNT(Set_ALL_THREADPRIVATE, __kmp_tp_capacity),
3845 KMP_HNT(PossibleSystemLimitOnThreads), __kmp_msg_null);
3846 } else {
3847 __kmp_fatal(KMP_MSG(CantRegisterNewThread), KMP_HNT(SystemLimitOnThreads),
3849 }
3850 }
3851
3852 // When hidden helper task is enabled, __kmp_threads is organized as follows:
3853 // 0: initial thread, also a regular OpenMP thread.
3854 // [1, __kmp_hidden_helper_threads_num]: slots for hidden helper threads.
3855 // [__kmp_hidden_helper_threads_num + 1, __kmp_threads_capacity): slots for
3856 // regular OpenMP threads.
3858 // Find an available thread slot for hidden helper thread. Slots for hidden
3859 // helper threads start from 1 to __kmp_hidden_helper_threads_num.
3860 for (gtid = 1; TCR_PTR(__kmp_threads[gtid]) != NULL &&
3862 gtid++)
3863 ;
3865 KA_TRACE(1, ("__kmp_register_root: found slot in threads array for "
3866 "hidden helper thread: T#%d\n",
3867 gtid));
3868 } else {
3869 /* find an available thread slot */
3870 // Don't reassign the zero slot since we need that to only be used by
3871 // initial thread. Slots for hidden helper threads should also be skipped.
3872 if (initial_thread && TCR_PTR(__kmp_threads[0]) == NULL) {
3873 gtid = 0;
3874 } else {
3875 for (gtid = __kmp_hidden_helper_threads_num + 1;
3876 TCR_PTR(__kmp_threads[gtid]) != NULL; gtid++)
3877 ;
3878 }
3879 KA_TRACE(
3880 1, ("__kmp_register_root: found slot in threads array: T#%d\n", gtid));
3882 }
3883
3884 /* update global accounting */
3885 __kmp_all_nth++;
3887
3888 // if __kmp_adjust_gtid_mode is set, then we use method #1 (sp search) for low
3889 // numbers of procs, and method #2 (keyed API call) for higher numbers.
3892 if (TCR_4(__kmp_gtid_mode) != 2) {
3894 }
3895 } else {
3896 if (TCR_4(__kmp_gtid_mode) != 1) {
3898 }
3899 }
3900 }
3901
3902#ifdef KMP_ADJUST_BLOCKTIME
3903 /* Adjust blocktime to zero if necessary */
3904 /* Middle initialization might not have occurred yet */
3905 if (!__kmp_env_blocktime && (__kmp_avail_proc > 0)) {
3907 __kmp_zero_bt = TRUE;
3908 }
3909 }
3910#endif /* KMP_ADJUST_BLOCKTIME */
3911
3912 /* setup this new hierarchy */
3913 if (!(root = __kmp_root[gtid])) {
3914 root = __kmp_root[gtid] = (kmp_root_t *)__kmp_allocate(sizeof(kmp_root_t));
3915 KMP_DEBUG_ASSERT(!root->r.r_root_team);
3916 }
3917
3918#if KMP_STATS_ENABLED
3919 // Initialize stats as soon as possible (right after gtid assignment).
3920 __kmp_stats_thread_ptr = __kmp_stats_list->push_back(gtid);
3921 __kmp_stats_thread_ptr->startLife();
3922 KMP_SET_THREAD_STATE(SERIAL_REGION);
3923 KMP_INIT_PARTITIONED_TIMERS(OMP_serial);
3924#endif
3926
3927 /* setup new root thread structure */
3928 if (root->r.r_uber_thread) {
3929 root_thread = root->r.r_uber_thread;
3930 } else {
3931 root_thread = (kmp_info_t *)__kmp_allocate(sizeof(kmp_info_t));
3932 if (__kmp_storage_map) {
3933 __kmp_print_thread_storage_map(root_thread, gtid);
3934 }
3935 root_thread->th.th_info.ds.ds_gtid = gtid;
3936#if OMPT_SUPPORT
3937 root_thread->th.ompt_thread_info.thread_data = ompt_data_none;
3938#endif
3939 root_thread->th.th_root = root;
3941 root_thread->th.th_cons = __kmp_allocate_cons_stack(gtid);
3942 }
3943#if USE_FAST_MEMORY
3944 __kmp_initialize_fast_memory(root_thread);
3945#endif /* USE_FAST_MEMORY */
3946
3947#if KMP_USE_BGET
3948 KMP_DEBUG_ASSERT(root_thread->th.th_local.bget_data == NULL);
3949 __kmp_initialize_bget(root_thread);
3950#endif
3951 __kmp_init_random(root_thread); // Initialize random number generator
3952 }
3953
3954 /* setup the serial team held in reserve by the root thread */
3955 if (!root_thread->th.th_serial_team) {
3957 KF_TRACE(10, ("__kmp_register_root: before serial_team\n"));
3958 root_thread->th.th_serial_team =
3959 __kmp_allocate_team(root, 1, 1,
3960#if OMPT_SUPPORT
3961 ompt_data_none, // root parallel id
3962#endif
3963 proc_bind_default, &r_icvs, 0, NULL);
3964 }
3965 KMP_ASSERT(root_thread->th.th_serial_team);
3966 KF_TRACE(10, ("__kmp_register_root: after serial_team = %p\n",
3967 root_thread->th.th_serial_team));
3968
3969 /* drop root_thread into place */
3970 TCW_SYNC_PTR(__kmp_threads[gtid], root_thread);
3971
3972 root->r.r_root_team->t.t_threads[0] = root_thread;
3973 root->r.r_hot_team->t.t_threads[0] = root_thread;
3974 root_thread->th.th_serial_team->t.t_threads[0] = root_thread;
3975 // AC: the team created in reserve, not for execution (it is unused for now).
3976 root_thread->th.th_serial_team->t.t_serialized = 0;
3977 root->r.r_uber_thread = root_thread;
3978
3979 /* initialize the thread, get it ready to go */
3980 __kmp_initialize_info(root_thread, root->r.r_root_team, 0, gtid);
3982
3983 /* prepare the primary thread for get_gtid() */
3985
3986#if USE_ITT_BUILD
3987 __kmp_itt_thread_name(gtid);
3988#endif /* USE_ITT_BUILD */
3989
3990#ifdef KMP_TDATA_GTID
3991 __kmp_gtid = gtid;
3992#endif
3993 __kmp_create_worker(gtid, root_thread, __kmp_stksize);
3995
3996 KA_TRACE(20, ("__kmp_register_root: T#%d init T#%d(%d:%d) arrived: join=%u, "
3997 "plain=%u\n",
3998 gtid, __kmp_gtid_from_tid(0, root->r.r_hot_team),
3999 root->r.r_hot_team->t.t_id, 0, KMP_INIT_BARRIER_STATE,
4001 { // Initialize barrier data.
4002 int b;
4003 for (b = 0; b < bs_last_barrier; ++b) {
4004 root_thread->th.th_bar[b].bb.b_arrived = KMP_INIT_BARRIER_STATE;
4005#if USE_DEBUGGER
4006 root_thread->th.th_bar[b].bb.b_worker_arrived = 0;
4007#endif
4008 }
4009 }
4010 KMP_DEBUG_ASSERT(root->r.r_hot_team->t.t_bar[bs_forkjoin_barrier].b_arrived ==
4012
4013#if KMP_AFFINITY_SUPPORTED
4014 root_thread->th.th_current_place = KMP_PLACE_UNDEFINED;
4015 root_thread->th.th_new_place = KMP_PLACE_UNDEFINED;
4016 root_thread->th.th_first_place = KMP_PLACE_UNDEFINED;
4017 root_thread->th.th_last_place = KMP_PLACE_UNDEFINED;
4018#endif /* KMP_AFFINITY_SUPPORTED */
4019 root_thread->th.th_def_allocator = __kmp_def_allocator;
4020 root_thread->th.th_prev_level = 0;
4021 root_thread->th.th_prev_num_threads = 1;
4022
4024 tmp->cg_root = root_thread;
4026 tmp->cg_nthreads = 1;
4027 KA_TRACE(100, ("__kmp_register_root: Thread %p created node %p with"
4028 " cg_nthreads init to 1\n",
4029 root_thread, tmp));
4030 tmp->up = NULL;
4031 root_thread->th.th_cg_roots = tmp;
4032
4034
4035#if OMPT_SUPPORT
4036 if (ompt_enabled.enabled) {
4037
4038 kmp_info_t *root_thread = ompt_get_thread();
4039
4040 ompt_set_thread_state(root_thread, ompt_state_overhead);
4041
4042 if (ompt_enabled.ompt_callback_thread_begin) {
4043 ompt_callbacks.ompt_callback(ompt_callback_thread_begin)(
4044 ompt_thread_initial, __ompt_get_thread_data_internal());
4045 }
4046 ompt_data_t *task_data;
4047 ompt_data_t *parallel_data;
4048 __ompt_get_task_info_internal(0, NULL, &task_data, NULL, &parallel_data,
4049 NULL);
4050 if (ompt_enabled.ompt_callback_implicit_task) {
4051 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
4052 ompt_scope_begin, parallel_data, task_data, 1, 1, ompt_task_initial);
4053 }
4054
4055 ompt_set_thread_state(root_thread, ompt_state_work_serial);
4056 }
4057#endif
4058#if OMPD_SUPPORT
4059 if (ompd_state & OMPD_ENABLE_BP)
4060 ompd_bp_thread_begin();
4061#endif
4062
4063 KMP_MB();
4065
4066 return gtid;
4067}
4068
4070 const int max_level) {
4071 int i, n, nth;
4072 kmp_hot_team_ptr_t *hot_teams = thr->th.th_hot_teams;
4073 if (!hot_teams || !hot_teams[level].hot_team) {
4074 return 0;
4075 }
4076 KMP_DEBUG_ASSERT(level < max_level);
4077 kmp_team_t *team = hot_teams[level].hot_team;
4078 nth = hot_teams[level].hot_team_nth;
4079 n = nth - 1; // primary thread is not freed
4080 if (level < max_level - 1) {
4081 for (i = 0; i < nth; ++i) {
4082 kmp_info_t *th = team->t.t_threads[i];
4083 n += __kmp_free_hot_teams(root, th, level + 1, max_level);
4084 if (i > 0 && th->th.th_hot_teams) {
4085 __kmp_free(th->th.th_hot_teams);
4086 th->th.th_hot_teams = NULL;
4087 }
4088 }
4089 }
4090 __kmp_free_team(root, team, NULL);
4091 return n;
4092}
4093
4094// Resets a root thread and clear its root and hot teams.
4095// Returns the number of __kmp_threads entries directly and indirectly freed.
4096static int __kmp_reset_root(int gtid, kmp_root_t *root) {
4097 kmp_team_t *root_team = root->r.r_root_team;
4098 kmp_team_t *hot_team = root->r.r_hot_team;
4099 int n = hot_team->t.t_nproc;
4100 int i;
4101
4102 KMP_DEBUG_ASSERT(!root->r.r_active);
4103
4104 root->r.r_root_team = NULL;
4105 root->r.r_hot_team = NULL;
4106 // __kmp_free_team() does not free hot teams, so we have to clear r_hot_team
4107 // before call to __kmp_free_team().
4108 __kmp_free_team(root, root_team, NULL);
4110 0) { // need to free nested hot teams and their threads if any
4111 for (i = 0; i < hot_team->t.t_nproc; ++i) {
4112 kmp_info_t *th = hot_team->t.t_threads[i];
4113 if (__kmp_hot_teams_max_level > 1) {
4115 }
4116 if (th->th.th_hot_teams) {
4117 __kmp_free(th->th.th_hot_teams);
4118 th->th.th_hot_teams = NULL;
4119 }
4120 }
4121 }
4122 __kmp_free_team(root, hot_team, NULL);
4123
4124 // Before we can reap the thread, we need to make certain that all other
4125 // threads in the teams that had this root as ancestor have stopped trying to
4126 // steal tasks.
4129 }
4130
4131#if KMP_OS_WINDOWS
4132 /* Close Handle of root duplicated in __kmp_create_worker (tr #62919) */
4133 KA_TRACE(
4134 10, ("__kmp_reset_root: free handle, th = %p, handle = %" KMP_UINTPTR_SPEC
4135 "\n",
4136 (LPVOID) & (root->r.r_uber_thread->th),
4137 root->r.r_uber_thread->th.th_info.ds.ds_thread));
4138 __kmp_free_handle(root->r.r_uber_thread->th.th_info.ds.ds_thread);
4139#endif /* KMP_OS_WINDOWS */
4140
4141#if OMPD_SUPPORT
4142 if (ompd_state & OMPD_ENABLE_BP)
4143 ompd_bp_thread_end();
4144#endif
4145
4146#if OMPT_SUPPORT
4147 ompt_data_t *task_data;
4148 ompt_data_t *parallel_data;
4149 __ompt_get_task_info_internal(0, NULL, &task_data, NULL, &parallel_data,
4150 NULL);
4151 if (ompt_enabled.ompt_callback_implicit_task) {
4152 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
4153 ompt_scope_end, parallel_data, task_data, 0, 1, ompt_task_initial);
4154 }
4155 if (ompt_enabled.ompt_callback_thread_end) {
4156 ompt_callbacks.ompt_callback(ompt_callback_thread_end)(
4157 &(root->r.r_uber_thread->th.ompt_thread_info.thread_data));
4158 }
4159#endif
4160
4162 __kmp_nth - 1); // __kmp_reap_thread will decrement __kmp_all_nth.
4163 i = root->r.r_uber_thread->th.th_cg_roots->cg_nthreads--;
4164 KA_TRACE(100, ("__kmp_reset_root: Thread %p decrement cg_nthreads on node %p"
4165 " to %d\n",
4166 root->r.r_uber_thread, root->r.r_uber_thread->th.th_cg_roots,
4167 root->r.r_uber_thread->th.th_cg_roots->cg_nthreads));
4168 if (i == 1) {
4169 // need to free contention group structure
4170 KMP_DEBUG_ASSERT(root->r.r_uber_thread ==
4171 root->r.r_uber_thread->th.th_cg_roots->cg_root);
4172 KMP_DEBUG_ASSERT(root->r.r_uber_thread->th.th_cg_roots->up == NULL);
4173 __kmp_free(root->r.r_uber_thread->th.th_cg_roots);
4174 root->r.r_uber_thread->th.th_cg_roots = NULL;
4175 }
4176 __kmp_reap_thread(root->r.r_uber_thread, 1);
4177
4178 // We canot put root thread to __kmp_thread_pool, so we have to reap it
4179 // instead of freeing.
4180 root->r.r_uber_thread = NULL;
4181 /* mark root as no longer in use */
4182 root->r.r_begin = FALSE;
4183
4184 return n;
4185}
4186
4188 KA_TRACE(1, ("__kmp_unregister_root_current_thread: enter T#%d\n", gtid));
4189 /* this lock should be ok, since unregister_root_current_thread is never
4190 called during an abort, only during a normal close. furthermore, if you
4191 have the forkjoin lock, you should never try to get the initz lock */
4193 if (TCR_4(__kmp_global.g.g_done) || !__kmp_init_serial) {
4194 KC_TRACE(10, ("__kmp_unregister_root_current_thread: already finished, "
4195 "exiting T#%d\n",
4196 gtid));
4198 return;
4199 }
4200 kmp_root_t *root = __kmp_root[gtid];
4201
4204 KMP_ASSERT(root == __kmp_threads[gtid]->th.th_root);
4205 KMP_ASSERT(root->r.r_active == FALSE);
4206
4207 KMP_MB();
4208
4209 kmp_info_t *thread = __kmp_threads[gtid];
4210 kmp_team_t *team = thread->th.th_team;
4211 kmp_task_team_t *task_team = thread->th.th_task_team;
4212
4213 // we need to wait for the proxy tasks before finishing the thread
4214 if (task_team != NULL && (task_team->tt.tt_found_proxy_tasks ||
4216#if OMPT_SUPPORT
4217 // the runtime is shutting down so we won't report any events
4218 thread->th.ompt_thread_info.state = ompt_state_undefined;
4219#endif
4220 __kmp_task_team_wait(thread, team USE_ITT_BUILD_ARG(NULL));
4221 }
4222
4223 __kmp_reset_root(gtid, root);
4224
4225 KMP_MB();
4226 KC_TRACE(10,
4227 ("__kmp_unregister_root_current_thread: T#%d unregistered\n", gtid));
4228
4230}
4231
4232#if KMP_OS_WINDOWS
4233/* __kmp_forkjoin_lock must be already held
4234 Unregisters a root thread that is not the current thread. Returns the number
4235 of __kmp_threads entries freed as a result. */
4236static int __kmp_unregister_root_other_thread(int gtid) {
4237 kmp_root_t *root = __kmp_root[gtid];
4238 int r;
4239
4240 KA_TRACE(1, ("__kmp_unregister_root_other_thread: enter T#%d\n", gtid));
4243 KMP_ASSERT(root == __kmp_threads[gtid]->th.th_root);
4244 KMP_ASSERT(root->r.r_active == FALSE);
4245
4246 r = __kmp_reset_root(gtid, root);
4247 KC_TRACE(10,
4248 ("__kmp_unregister_root_other_thread: T#%d unregistered\n", gtid));
4249 return r;
4250}
4251#endif
4252
4253#if KMP_DEBUG
4254void __kmp_task_info() {
4255
4256 kmp_int32 gtid = __kmp_entry_gtid();
4257 kmp_int32 tid = __kmp_tid_from_gtid(gtid);
4258 kmp_info_t *this_thr = __kmp_threads[gtid];
4259 kmp_team_t *steam = this_thr->th.th_serial_team;
4260 kmp_team_t *team = this_thr->th.th_team;
4261
4263 "__kmp_task_info: gtid=%d tid=%d t_thread=%p team=%p steam=%p curtask=%p "
4264 "ptask=%p\n",
4265 gtid, tid, this_thr, team, steam, this_thr->th.th_current_task,
4266 team->t.t_implicit_task_taskdata[tid].td_parent);
4267}
4268#endif // KMP_DEBUG
4269
4270/* TODO optimize with one big memclr, take out what isn't needed, split
4271 responsibility to workers as much as possible, and delay initialization of
4272 features as much as possible */
4273static void __kmp_initialize_info(kmp_info_t *this_thr, kmp_team_t *team,
4274 int tid, int gtid) {
4275 /* this_thr->th.th_info.ds.ds_gtid is setup in
4276 kmp_allocate_thread/create_worker.
4277 this_thr->th.th_serial_team is setup in __kmp_allocate_thread */
4278 KMP_DEBUG_ASSERT(this_thr != NULL);
4279 KMP_DEBUG_ASSERT(this_thr->th.th_serial_team);
4280 KMP_DEBUG_ASSERT(team);
4281 KMP_DEBUG_ASSERT(team->t.t_threads);
4282 KMP_DEBUG_ASSERT(team->t.t_dispatch);
4283 kmp_info_t *master = team->t.t_threads[0];
4284 KMP_DEBUG_ASSERT(master);
4285 KMP_DEBUG_ASSERT(master->th.th_root);
4286
4287 KMP_MB();
4288
4289 TCW_SYNC_PTR(this_thr->th.th_team, team);
4290
4291 this_thr->th.th_info.ds.ds_tid = tid;
4292 this_thr->th.th_set_nproc = 0;
4294 // When tasking is possible, threads are not safe to reap until they are
4295 // done tasking; this will be set when tasking code is exited in wait
4296 this_thr->th.th_reap_state = KMP_NOT_SAFE_TO_REAP;
4297 else // no tasking --> always safe to reap
4298 this_thr->th.th_reap_state = KMP_SAFE_TO_REAP;
4299 this_thr->th.th_set_proc_bind = proc_bind_default;
4300
4301#if KMP_AFFINITY_SUPPORTED
4302 this_thr->th.th_new_place = this_thr->th.th_current_place;
4303#endif
4304 this_thr->th.th_root = master->th.th_root;
4305
4306 /* setup the thread's cache of the team structure */
4307 this_thr->th.th_team_nproc = team->t.t_nproc;
4308 this_thr->th.th_team_master = master;
4309 this_thr->th.th_team_serialized = team->t.t_serialized;
4310
4311 KMP_DEBUG_ASSERT(team->t.t_implicit_task_taskdata);
4312
4313 KF_TRACE(10, ("__kmp_initialize_info1: T#%d:%d this_thread=%p curtask=%p\n",
4314 tid, gtid, this_thr, this_thr->th.th_current_task));
4315
4316 __kmp_init_implicit_task(this_thr->th.th_team_master->th.th_ident, this_thr,
4317 team, tid, TRUE);
4318
4319 KF_TRACE(10, ("__kmp_initialize_info2: T#%d:%d this_thread=%p curtask=%p\n",
4320 tid, gtid, this_thr, this_thr->th.th_current_task));
4321 // TODO: Initialize ICVs from parent; GEH - isn't that already done in
4322 // __kmp_initialize_team()?
4323
4324 /* TODO no worksharing in speculative threads */
4325 this_thr->th.th_dispatch = &team->t.t_dispatch[tid];
4326
4327 this_thr->th.th_local.this_construct = 0;
4328
4329 if (!this_thr->th.th_pri_common) {
4330 this_thr->th.th_pri_common =
4331 (struct common_table *)__kmp_allocate(sizeof(struct common_table));
4332 if (__kmp_storage_map) {
4334 gtid, this_thr->th.th_pri_common, this_thr->th.th_pri_common + 1,
4335 sizeof(struct common_table), "th_%d.th_pri_common\n", gtid);
4336 }
4337 this_thr->th.th_pri_head = NULL;
4338 }
4339
4340 if (this_thr != master && // Primary thread's CG root is initialized elsewhere
4341 this_thr->th.th_cg_roots != master->th.th_cg_roots) { // CG root not set
4342 // Make new thread's CG root same as primary thread's
4343 KMP_DEBUG_ASSERT(master->th.th_cg_roots);
4344 kmp_cg_root_t *tmp = this_thr->th.th_cg_roots;
4345 if (tmp) {
4346 // worker changes CG, need to check if old CG should be freed
4347 int i = tmp->cg_nthreads--;
4348 KA_TRACE(100, ("__kmp_initialize_info: Thread %p decrement cg_nthreads"
4349 " on node %p of thread %p to %d\n",
4350 this_thr, tmp, tmp->cg_root, tmp->cg_nthreads));
4351 if (i == 1) {
4352 __kmp_free(tmp); // last thread left CG --> free it
4353 }
4354 }
4355 this_thr->th.th_cg_roots = master->th.th_cg_roots;
4356 // Increment new thread's CG root's counter to add the new thread
4357 this_thr->th.th_cg_roots->cg_nthreads++;
4358 KA_TRACE(100, ("__kmp_initialize_info: Thread %p increment cg_nthreads on"
4359 " node %p of thread %p to %d\n",
4360 this_thr, this_thr->th.th_cg_roots,
4361 this_thr->th.th_cg_roots->cg_root,
4362 this_thr->th.th_cg_roots->cg_nthreads));
4363 this_thr->th.th_current_task->td_icvs.thread_limit =
4364 this_thr->th.th_cg_roots->cg_thread_limit;
4365 }
4366
4367 /* Initialize dynamic dispatch */
4368 {
4369 volatile kmp_disp_t *dispatch = this_thr->th.th_dispatch;
4370 // Use team max_nproc since this will never change for the team.
4371 size_t disp_size =
4372 sizeof(dispatch_private_info_t) *
4373 (team->t.t_max_nproc == 1 ? 1 : __kmp_dispatch_num_buffers);
4374 KD_TRACE(10, ("__kmp_initialize_info: T#%d max_nproc: %d\n", gtid,
4375 team->t.t_max_nproc));
4376 KMP_ASSERT(dispatch);
4377 KMP_DEBUG_ASSERT(team->t.t_dispatch);
4378 KMP_DEBUG_ASSERT(dispatch == &team->t.t_dispatch[tid]);
4379
4380 dispatch->th_disp_index = 0;
4381 dispatch->th_doacross_buf_idx = 0;
4382 if (!dispatch->th_disp_buffer) {
4383 dispatch->th_disp_buffer =
4385
4386 if (__kmp_storage_map) {
4388 gtid, &dispatch->th_disp_buffer[0],
4389 &dispatch->th_disp_buffer[team->t.t_max_nproc == 1
4390 ? 1
4392 disp_size,
4393 "th_%d.th_dispatch.th_disp_buffer "
4394 "(team_%d.t_dispatch[%d].th_disp_buffer)",
4395 gtid, team->t.t_id, gtid);
4396 }
4397 } else {
4398 memset(&dispatch->th_disp_buffer[0], '\0', disp_size);
4399 }
4400
4401 dispatch->th_dispatch_pr_current = 0;
4402 dispatch->th_dispatch_sh_current = 0;
4403
4404 dispatch->th_deo_fcn = 0; /* ORDERED */
4405 dispatch->th_dxo_fcn = 0; /* END ORDERED */
4406 }
4407
4408 this_thr->th.th_next_pool = NULL;
4409
4410 KMP_DEBUG_ASSERT(!this_thr->th.th_spin_here);
4411 KMP_DEBUG_ASSERT(this_thr->th.th_next_waiting == 0);
4412
4413 KMP_MB();
4414}
4415
4416/* allocate a new thread for the requesting team. this is only called from
4417 within a forkjoin critical section. we will first try to get an available
4418 thread from the thread pool. if none is available, we will fork a new one
4419 assuming we are able to create a new one. this should be assured, as the
4420 caller should check on this first. */
4422 int new_tid) {
4423 kmp_team_t *serial_team;
4424 kmp_info_t *new_thr;
4425 int new_gtid;
4426
4427 KA_TRACE(20, ("__kmp_allocate_thread: T#%d\n", __kmp_get_gtid()));
4428 KMP_DEBUG_ASSERT(root && team);
4429 KMP_MB();
4430
4431 /* first, try to get one from the thread pool unless allocating thread is
4432 * the main hidden helper thread. The hidden helper team should always
4433 * allocate new OS threads. */
4435 new_thr = CCAST(kmp_info_t *, __kmp_thread_pool);
4436 __kmp_thread_pool = (volatile kmp_info_t *)new_thr->th.th_next_pool;
4437 if (new_thr == __kmp_thread_pool_insert_pt) {
4439 }
4440 TCW_4(new_thr->th.th_in_pool, FALSE);
4442 __kmp_lock_suspend_mx(new_thr);
4443 if (new_thr->th.th_active_in_pool == TRUE) {
4444 KMP_DEBUG_ASSERT(new_thr->th.th_active == TRUE);
4446 new_thr->th.th_active_in_pool = FALSE;
4447 }
4448 __kmp_unlock_suspend_mx(new_thr);
4449
4450 KA_TRACE(20, ("__kmp_allocate_thread: T#%d using thread T#%d\n",
4451 __kmp_get_gtid(), new_thr->th.th_info.ds.ds_gtid));
4452 KMP_ASSERT(!new_thr->th.th_team);
4454
4455 /* setup the thread structure */
4456 __kmp_initialize_info(new_thr, team, new_tid,
4457 new_thr->th.th_info.ds.ds_gtid);
4458 KMP_DEBUG_ASSERT(new_thr->th.th_serial_team);
4459
4461
4462 new_thr->th.th_task_state = 0;
4463
4465 // Make sure pool thread has transitioned to waiting on own thread struct
4466 KMP_DEBUG_ASSERT(new_thr->th.th_used_in_team.load() == 0);
4467 // Thread activated in __kmp_allocate_team when increasing team size
4468 }
4469
4470#ifdef KMP_ADJUST_BLOCKTIME
4471 /* Adjust blocktime back to zero if necessary */
4472 /* Middle initialization might not have occurred yet */
4473 if (!__kmp_env_blocktime && (__kmp_avail_proc > 0)) {
4475 __kmp_zero_bt = TRUE;
4476 }
4477 }
4478#endif /* KMP_ADJUST_BLOCKTIME */
4479
4480#if KMP_DEBUG
4481 // If thread entered pool via __kmp_free_thread, wait_flag should !=
4482 // KMP_BARRIER_PARENT_FLAG.
4483 int b;
4484 kmp_balign_t *balign = new_thr->th.th_bar;
4485 for (b = 0; b < bs_last_barrier; ++b)
4486 KMP_DEBUG_ASSERT(balign[b].bb.wait_flag != KMP_BARRIER_PARENT_FLAG);
4487#endif
4488
4489 KF_TRACE(10, ("__kmp_allocate_thread: T#%d using thread %p T#%d\n",
4490 __kmp_get_gtid(), new_thr, new_thr->th.th_info.ds.ds_gtid));
4491
4492 KMP_MB();
4493 return new_thr;
4494 }
4495
4496 /* no, well fork a new one */
4499
4500#if KMP_USE_MONITOR
4501 // If this is the first worker thread the RTL is creating, then also
4502 // launch the monitor thread. We try to do this as early as possible.
4503 if (!TCR_4(__kmp_init_monitor)) {
4504 __kmp_acquire_bootstrap_lock(&__kmp_monitor_lock);
4505 if (!TCR_4(__kmp_init_monitor)) {
4506 KF_TRACE(10, ("before __kmp_create_monitor\n"));
4507 TCW_4(__kmp_init_monitor, 1);
4508 __kmp_create_monitor(&__kmp_monitor);
4509 KF_TRACE(10, ("after __kmp_create_monitor\n"));
4510#if KMP_OS_WINDOWS
4511 // AC: wait until monitor has started. This is a fix for CQ232808.
4512 // The reason is that if the library is loaded/unloaded in a loop with
4513 // small (parallel) work in between, then there is high probability that
4514 // monitor thread started after the library shutdown. At shutdown it is
4515 // too late to cope with the problem, because when the primary thread is
4516 // in DllMain (process detach) the monitor has no chances to start (it is
4517 // blocked), and primary thread has no means to inform the monitor that
4518 // the library has gone, because all the memory which the monitor can
4519 // access is going to be released/reset.
4520 while (TCR_4(__kmp_init_monitor) < 2) {
4521 KMP_YIELD(TRUE);
4522 }
4523 KF_TRACE(10, ("after monitor thread has started\n"));
4524#endif
4525 }
4526 __kmp_release_bootstrap_lock(&__kmp_monitor_lock);
4527 }
4528#endif
4529
4530 KMP_MB();
4531
4532 {
4533 int new_start_gtid = TCR_4(__kmp_init_hidden_helper_threads)
4534 ? 1
4536
4537 for (new_gtid = new_start_gtid; TCR_PTR(__kmp_threads[new_gtid]) != NULL;
4538 ++new_gtid) {
4540 }
4541
4544 }
4545 }
4546
4547 /* allocate space for it. */
4548 new_thr = (kmp_info_t *)__kmp_allocate(sizeof(kmp_info_t));
4549
4550 new_thr->th.th_nt_strict = false;
4551 new_thr->th.th_nt_loc = NULL;
4552 new_thr->th.th_nt_sev = severity_fatal;
4553 new_thr->th.th_nt_msg = NULL;
4554
4555 TCW_SYNC_PTR(__kmp_threads[new_gtid], new_thr);
4556
4557#if USE_ITT_BUILD && USE_ITT_NOTIFY && KMP_DEBUG
4558 // suppress race conditions detection on synchronization flags in debug mode
4559 // this helps to analyze library internals eliminating false positives
4560 __itt_suppress_mark_range(
4561 __itt_suppress_range, __itt_suppress_threading_errors,
4562 &new_thr->th.th_sleep_loc, sizeof(new_thr->th.th_sleep_loc));
4563 __itt_suppress_mark_range(
4564 __itt_suppress_range, __itt_suppress_threading_errors,
4565 &new_thr->th.th_reap_state, sizeof(new_thr->th.th_reap_state));
4566#if KMP_OS_WINDOWS
4567 __itt_suppress_mark_range(
4568 __itt_suppress_range, __itt_suppress_threading_errors,
4569 &new_thr->th.th_suspend_init, sizeof(new_thr->th.th_suspend_init));
4570#else
4571 __itt_suppress_mark_range(__itt_suppress_range,
4572 __itt_suppress_threading_errors,
4573 &new_thr->th.th_suspend_init_count,
4574 sizeof(new_thr->th.th_suspend_init_count));
4575#endif
4576 // TODO: check if we need to also suppress b_arrived flags
4577 __itt_suppress_mark_range(__itt_suppress_range,
4578 __itt_suppress_threading_errors,
4579 CCAST(kmp_uint64 *, &new_thr->th.th_bar[0].bb.b_go),
4580 sizeof(new_thr->th.th_bar[0].bb.b_go));
4581 __itt_suppress_mark_range(__itt_suppress_range,
4582 __itt_suppress_threading_errors,
4583 CCAST(kmp_uint64 *, &new_thr->th.th_bar[1].bb.b_go),
4584 sizeof(new_thr->th.th_bar[1].bb.b_go));
4585 __itt_suppress_mark_range(__itt_suppress_range,
4586 __itt_suppress_threading_errors,
4587 CCAST(kmp_uint64 *, &new_thr->th.th_bar[2].bb.b_go),
4588 sizeof(new_thr->th.th_bar[2].bb.b_go));
4589#endif /* USE_ITT_BUILD && USE_ITT_NOTIFY && KMP_DEBUG */
4590 if (__kmp_storage_map) {
4591 __kmp_print_thread_storage_map(new_thr, new_gtid);
4592 }
4593
4594 // add the reserve serialized team, initialized from the team's primary thread
4595 {
4597 KF_TRACE(10, ("__kmp_allocate_thread: before th_serial/serial_team\n"));
4598 new_thr->th.th_serial_team = serial_team =
4599 (kmp_team_t *)__kmp_allocate_team(root, 1, 1,
4600#if OMPT_SUPPORT
4601 ompt_data_none, // root parallel id
4602#endif
4603 proc_bind_default, &r_icvs, 0, NULL);
4604 }
4605 KMP_ASSERT(serial_team);
4606 serial_team->t.t_serialized = 0; // AC: the team created in reserve, not for
4607 // execution (it is unused for now).
4608 serial_team->t.t_threads[0] = new_thr;
4609 KF_TRACE(10,
4610 ("__kmp_allocate_thread: after th_serial/serial_team : new_thr=%p\n",
4611 new_thr));
4612
4613 /* setup the thread structures */
4614 __kmp_initialize_info(new_thr, team, new_tid, new_gtid);
4615
4616#if USE_FAST_MEMORY
4617 __kmp_initialize_fast_memory(new_thr);
4618#endif /* USE_FAST_MEMORY */
4619
4620#if KMP_USE_BGET
4621 KMP_DEBUG_ASSERT(new_thr->th.th_local.bget_data == NULL);
4622 __kmp_initialize_bget(new_thr);
4623#endif
4624
4625 __kmp_init_random(new_thr); // Initialize random number generator
4626
4627 /* Initialize these only once when thread is grabbed for a team allocation */
4628 KA_TRACE(20,
4629 ("__kmp_allocate_thread: T#%d init go fork=%u, plain=%u\n",
4631
4632 int b;
4633 kmp_balign_t *balign = new_thr->th.th_bar;
4634 for (b = 0; b < bs_last_barrier; ++b) {
4635 balign[b].bb.b_go = KMP_INIT_BARRIER_STATE;
4636 balign[b].bb.team = NULL;
4637 balign[b].bb.wait_flag = KMP_BARRIER_NOT_WAITING;
4638 balign[b].bb.use_oncore_barrier = 0;
4639 }
4640
4641 TCW_PTR(new_thr->th.th_sleep_loc, NULL);
4642 new_thr->th.th_sleep_loc_type = flag_unset;
4643
4644 new_thr->th.th_spin_here = FALSE;
4645 new_thr->th.th_next_waiting = 0;
4646#if KMP_OS_UNIX
4647 new_thr->th.th_blocking = false;
4648#endif
4649
4650#if KMP_AFFINITY_SUPPORTED
4651 new_thr->th.th_current_place = KMP_PLACE_UNDEFINED;
4652 new_thr->th.th_new_place = KMP_PLACE_UNDEFINED;
4653 new_thr->th.th_first_place = KMP_PLACE_UNDEFINED;
4654 new_thr->th.th_last_place = KMP_PLACE_UNDEFINED;
4655#endif
4656 new_thr->th.th_def_allocator = __kmp_def_allocator;
4657 new_thr->th.th_prev_level = 0;
4658 new_thr->th.th_prev_num_threads = 1;
4659
4660 TCW_4(new_thr->th.th_in_pool, FALSE);
4661 new_thr->th.th_active_in_pool = FALSE;
4662 TCW_4(new_thr->th.th_active, TRUE);
4663
4664 new_thr->th.th_set_nested_nth = NULL;
4665 new_thr->th.th_set_nested_nth_sz = 0;
4666
4667 /* adjust the global counters */
4668 __kmp_all_nth++;
4669 __kmp_nth++;
4670
4671 // if __kmp_adjust_gtid_mode is set, then we use method #1 (sp search) for low
4672 // numbers of procs, and method #2 (keyed API call) for higher numbers.
4675 if (TCR_4(__kmp_gtid_mode) != 2) {
4677 }
4678 } else {
4679 if (TCR_4(__kmp_gtid_mode) != 1) {
4681 }
4682 }
4683 }
4684
4685#ifdef KMP_ADJUST_BLOCKTIME
4686 /* Adjust blocktime back to zero if necessary */
4687 /* Middle initialization might not have occurred yet */
4688 if (!__kmp_env_blocktime && (__kmp_avail_proc > 0)) {
4690 __kmp_zero_bt = TRUE;
4691 }
4692 }
4693#endif /* KMP_ADJUST_BLOCKTIME */
4694
4695#if KMP_AFFINITY_SUPPORTED
4696 // Set the affinity and topology information for new thread
4697 __kmp_affinity_set_init_mask(new_gtid, /*isa_root=*/FALSE);
4698#endif
4699
4700 /* actually fork it and create the new worker thread */
4701 KF_TRACE(
4702 10, ("__kmp_allocate_thread: before __kmp_create_worker: %p\n", new_thr));
4703 __kmp_create_worker(new_gtid, new_thr, __kmp_stksize);
4704 KF_TRACE(10,
4705 ("__kmp_allocate_thread: after __kmp_create_worker: %p\n", new_thr));
4706
4707 KA_TRACE(20, ("__kmp_allocate_thread: T#%d forked T#%d\n", __kmp_get_gtid(),
4708 new_gtid));
4709 KMP_MB();
4710 return new_thr;
4711}
4712
4713/* Reinitialize team for reuse.
4714 The hot team code calls this case at every fork barrier, so EPCC barrier
4715 test are extremely sensitive to changes in it, esp. writes to the team
4716 struct, which cause a cache invalidation in all threads.
4717 IF YOU TOUCH THIS ROUTINE, RUN EPCC C SYNCBENCH ON A BIG-IRON MACHINE!!! */
4719 kmp_internal_control_t *new_icvs,
4720 ident_t *loc) {
4721 KF_TRACE(10, ("__kmp_reinitialize_team: enter this_thread=%p team=%p\n",
4722 team->t.t_threads[0], team));
4723 KMP_DEBUG_ASSERT(team && new_icvs);
4725 KMP_CHECK_UPDATE(team->t.t_ident, loc);
4726
4727 KMP_CHECK_UPDATE(team->t.t_id, KMP_GEN_TEAM_ID());
4728 // Copy ICVs to the primary thread's implicit taskdata
4729 __kmp_init_implicit_task(loc, team->t.t_threads[0], team, 0, FALSE);
4730 copy_icvs(&team->t.t_implicit_task_taskdata[0].td_icvs, new_icvs);
4731
4732 KF_TRACE(10, ("__kmp_reinitialize_team: exit this_thread=%p team=%p\n",
4733 team->t.t_threads[0], team));
4734}
4735
4736/* Initialize the team data structure.
4737 This assumes the t_threads and t_max_nproc are already set.
4738 Also, we don't touch the arguments */
4739static void __kmp_initialize_team(kmp_team_t *team, int new_nproc,
4740 kmp_internal_control_t *new_icvs,
4741 ident_t *loc) {
4742 KF_TRACE(10, ("__kmp_initialize_team: enter: team=%p\n", team));
4743
4744 /* verify */
4745 KMP_DEBUG_ASSERT(team);
4746 KMP_DEBUG_ASSERT(new_nproc <= team->t.t_max_nproc);
4747 KMP_DEBUG_ASSERT(team->t.t_threads);
4748 KMP_MB();
4749
4750 team->t.t_master_tid = 0; /* not needed */
4751 /* team->t.t_master_bar; not needed */
4752 team->t.t_serialized = new_nproc > 1 ? 0 : 1;
4753 team->t.t_nproc = new_nproc;
4754
4755 /* team->t.t_parent = NULL; TODO not needed & would mess up hot team */
4756 team->t.t_next_pool = NULL;
4757 /* memset( team->t.t_threads, 0, sizeof(kmp_info_t*)*new_nproc ); would mess
4758 * up hot team */
4759
4760 TCW_SYNC_PTR(team->t.t_pkfn, NULL); /* not needed */
4761 team->t.t_invoke = NULL; /* not needed */
4762
4763 // TODO???: team->t.t_max_active_levels = new_max_active_levels;
4764 team->t.t_sched.sched = new_icvs->sched.sched;
4765
4766#if KMP_ARCH_X86 || KMP_ARCH_X86_64
4767 team->t.t_fp_control_saved = FALSE; /* not needed */
4768 team->t.t_x87_fpu_control_word = 0; /* not needed */
4769 team->t.t_mxcsr = 0; /* not needed */
4770#endif /* KMP_ARCH_X86 || KMP_ARCH_X86_64 */
4771
4772 team->t.t_construct = 0;
4773
4774 team->t.t_ordered.dt.t_value = 0;
4775 team->t.t_master_active = FALSE;
4776
4777#ifdef KMP_DEBUG
4778 team->t.t_copypriv_data = NULL; /* not necessary, but nice for debugging */
4779#endif
4780#if KMP_OS_WINDOWS
4781 team->t.t_copyin_counter = 0; /* for barrier-free copyin implementation */
4782#endif
4783
4784 team->t.t_control_stack_top = NULL;
4785
4786 __kmp_reinitialize_team(team, new_icvs, loc);
4787
4788 KMP_MB();
4789 KF_TRACE(10, ("__kmp_initialize_team: exit: team=%p\n", team));
4790}
4791
4792#if KMP_AFFINITY_SUPPORTED
4793static inline void __kmp_set_thread_place(kmp_team_t *team, kmp_info_t *th,
4794 int first, int last, int newp) {
4795 th->th.th_first_place = first;
4796 th->th.th_last_place = last;
4797 th->th.th_new_place = newp;
4798 if (newp != th->th.th_current_place) {
4799 if (__kmp_display_affinity && team->t.t_display_affinity != 1)
4800 team->t.t_display_affinity = 1;
4801 // Copy topology information associated with the new place
4802 th->th.th_topology_ids = __kmp_affinity.ids[th->th.th_new_place];
4803 th->th.th_topology_attrs = __kmp_affinity.attrs[th->th.th_new_place];
4804 }
4805}
4806
4807// __kmp_partition_places() is the heart of the OpenMP 4.0 affinity mechanism.
4808// It calculates the worker + primary thread's partition based upon the parent
4809// thread's partition, and binds each worker to a thread in their partition.
4810// The primary thread's partition should already include its current binding.
4811static void __kmp_partition_places(kmp_team_t *team, int update_master_only) {
4812 // Do not partition places for the hidden helper team
4813 if (KMP_HIDDEN_HELPER_TEAM(team))
4814 return;
4815 // Copy the primary thread's place partition to the team struct
4816 kmp_info_t *master_th = team->t.t_threads[0];
4817 KMP_DEBUG_ASSERT(master_th != NULL);
4818 kmp_proc_bind_t proc_bind = team->t.t_proc_bind;
4819 int first_place = master_th->th.th_first_place;
4820 int last_place = master_th->th.th_last_place;
4821 int masters_place = master_th->th.th_current_place;
4822 int num_masks = __kmp_affinity.num_masks;
4823 team->t.t_first_place = first_place;
4824 team->t.t_last_place = last_place;
4825
4826 KA_TRACE(20, ("__kmp_partition_places: enter: proc_bind = %d T#%d(%d:0) "
4827 "bound to place %d partition = [%d,%d]\n",
4828 proc_bind, __kmp_gtid_from_thread(team->t.t_threads[0]),
4829 team->t.t_id, masters_place, first_place, last_place));
4830
4831 switch (proc_bind) {
4832
4833 case proc_bind_default:
4834 // Serial teams might have the proc_bind policy set to proc_bind_default.
4835 // Not an issue -- we don't rebind primary thread for any proc_bind policy.
4836 KMP_DEBUG_ASSERT(team->t.t_nproc == 1);
4837 break;
4838
4839 case proc_bind_primary: {
4840 int f;
4841 int n_th = team->t.t_nproc;
4842 for (f = 1; f < n_th; f++) {
4843 kmp_info_t *th = team->t.t_threads[f];
4844 KMP_DEBUG_ASSERT(th != NULL);
4845 __kmp_set_thread_place(team, th, first_place, last_place, masters_place);
4846
4847 KA_TRACE(100, ("__kmp_partition_places: primary: T#%d(%d:%d) place %d "
4848 "partition = [%d,%d]\n",
4849 __kmp_gtid_from_thread(team->t.t_threads[f]), team->t.t_id,
4850 f, masters_place, first_place, last_place));
4851 }
4852 } break;
4853
4854 case proc_bind_close: {
4855 int f;
4856 int n_th = team->t.t_nproc;
4857 int n_places;
4858 if (first_place <= last_place) {
4859 n_places = last_place - first_place + 1;
4860 } else {
4861 n_places = num_masks - first_place + last_place + 1;
4862 }
4863 if (n_th <= n_places) {
4864 int place = masters_place;
4865 for (f = 1; f < n_th; f++) {
4866 kmp_info_t *th = team->t.t_threads[f];
4867 KMP_DEBUG_ASSERT(th != NULL);
4868
4869 if (place == last_place) {
4870 place = first_place;
4871 } else if (place == (num_masks - 1)) {
4872 place = 0;
4873 } else {
4874 place++;
4875 }
4876 __kmp_set_thread_place(team, th, first_place, last_place, place);
4877
4878 KA_TRACE(100, ("__kmp_partition_places: close: T#%d(%d:%d) place %d "
4879 "partition = [%d,%d]\n",
4880 __kmp_gtid_from_thread(team->t.t_threads[f]),
4881 team->t.t_id, f, place, first_place, last_place));
4882 }
4883 } else {
4884 int S, rem, gap, s_count;
4885 S = n_th / n_places;
4886 s_count = 0;
4887 rem = n_th - (S * n_places);
4888 gap = rem > 0 ? n_places / rem : n_places;
4889 int place = masters_place;
4890 int gap_ct = gap;
4891 for (f = 0; f < n_th; f++) {
4892 kmp_info_t *th = team->t.t_threads[f];
4893 KMP_DEBUG_ASSERT(th != NULL);
4894
4895 __kmp_set_thread_place(team, th, first_place, last_place, place);
4896 s_count++;
4897
4898 if ((s_count == S) && rem && (gap_ct == gap)) {
4899 // do nothing, add an extra thread to place on next iteration
4900 } else if ((s_count == S + 1) && rem && (gap_ct == gap)) {
4901 // we added an extra thread to this place; move to next place
4902 if (place == last_place) {
4903 place = first_place;
4904 } else if (place == (num_masks - 1)) {
4905 place = 0;
4906 } else {
4907 place++;
4908 }
4909 s_count = 0;
4910 gap_ct = 1;
4911 rem--;
4912 } else if (s_count == S) { // place full; don't add extra
4913 if (place == last_place) {
4914 place = first_place;
4915 } else if (place == (num_masks - 1)) {
4916 place = 0;
4917 } else {
4918 place++;
4919 }
4920 gap_ct++;
4921 s_count = 0;
4922 }
4923
4924 KA_TRACE(100,
4925 ("__kmp_partition_places: close: T#%d(%d:%d) place %d "
4926 "partition = [%d,%d]\n",
4927 __kmp_gtid_from_thread(team->t.t_threads[f]), team->t.t_id, f,
4928 th->th.th_new_place, first_place, last_place));
4929 }
4930 KMP_DEBUG_ASSERT(place == masters_place);
4931 }
4932 } break;
4933
4934 case proc_bind_spread: {
4935 int f;
4936 int n_th = team->t.t_nproc;
4937 int n_places;
4938 int thidx;
4939 if (first_place <= last_place) {
4940 n_places = last_place - first_place + 1;
4941 } else {
4942 n_places = num_masks - first_place + last_place + 1;
4943 }
4944 if (n_th <= n_places) {
4945 int place = -1;
4946
4947 if (n_places != num_masks) {
4948 int S = n_places / n_th;
4949 int s_count, rem, gap, gap_ct;
4950
4951 place = masters_place;
4952 rem = n_places - n_th * S;
4953 gap = rem ? n_th / rem : 1;
4954 gap_ct = gap;
4955 thidx = n_th;
4956 if (update_master_only == 1)
4957 thidx = 1;
4958 for (f = 0; f < thidx; f++) {
4959 kmp_info_t *th = team->t.t_threads[f];
4960 KMP_DEBUG_ASSERT(th != NULL);
4961
4962 int fplace = place, nplace = place;
4963 s_count = 1;
4964 while (s_count < S) {
4965 if (place == last_place) {
4966 place = first_place;
4967 } else if (place == (num_masks - 1)) {
4968 place = 0;
4969 } else {
4970 place++;
4971 }
4972 s_count++;
4973 }
4974 if (rem && (gap_ct == gap)) {
4975 if (place == last_place) {
4976 place = first_place;
4977 } else if (place == (num_masks - 1)) {
4978 place = 0;
4979 } else {
4980 place++;
4981 }
4982 rem--;
4983 gap_ct = 0;
4984 }
4985 __kmp_set_thread_place(team, th, fplace, place, nplace);
4986 gap_ct++;
4987
4988 if (place == last_place) {
4989 place = first_place;
4990 } else if (place == (num_masks - 1)) {
4991 place = 0;
4992 } else {
4993 place++;
4994 }
4995
4996 KA_TRACE(100,
4997 ("__kmp_partition_places: spread: T#%d(%d:%d) place %d "
4998 "partition = [%d,%d], num_masks: %u\n",
4999 __kmp_gtid_from_thread(team->t.t_threads[f]), team->t.t_id,
5000 f, th->th.th_new_place, th->th.th_first_place,
5001 th->th.th_last_place, num_masks));
5002 }
5003 } else {
5004 /* Having uniform space of available computation places I can create
5005 T partitions of round(P/T) size and put threads into the first
5006 place of each partition. */
5007 double current = static_cast<double>(masters_place);
5008 double spacing =
5009 (static_cast<double>(n_places + 1) / static_cast<double>(n_th));
5010 int first, last;
5011 kmp_info_t *th;
5012
5013 thidx = n_th + 1;
5014 if (update_master_only == 1)
5015 thidx = 1;
5016 for (f = 0; f < thidx; f++) {
5017 first = static_cast<int>(current);
5018 last = static_cast<int>(current + spacing) - 1;
5019 KMP_DEBUG_ASSERT(last >= first);
5020 if (first >= n_places) {
5021 if (masters_place) {
5022 first -= n_places;
5023 last -= n_places;
5024 if (first == (masters_place + 1)) {
5025 KMP_DEBUG_ASSERT(f == n_th);
5026 first--;
5027 }
5028 if (last == masters_place) {
5029 KMP_DEBUG_ASSERT(f == (n_th - 1));
5030 last--;
5031 }
5032 } else {
5033 KMP_DEBUG_ASSERT(f == n_th);
5034 first = 0;
5035 last = 0;
5036 }
5037 }
5038 if (last >= n_places) {
5039 last = (n_places - 1);
5040 }
5041 place = first;
5042 current += spacing;
5043 if (f < n_th) {
5044 KMP_DEBUG_ASSERT(0 <= first);
5045 KMP_DEBUG_ASSERT(n_places > first);
5046 KMP_DEBUG_ASSERT(0 <= last);
5047 KMP_DEBUG_ASSERT(n_places > last);
5048 KMP_DEBUG_ASSERT(last_place >= first_place);
5049 th = team->t.t_threads[f];
5050 KMP_DEBUG_ASSERT(th);
5051 __kmp_set_thread_place(team, th, first, last, place);
5052 KA_TRACE(100,
5053 ("__kmp_partition_places: spread: T#%d(%d:%d) place %d "
5054 "partition = [%d,%d], spacing = %.4f\n",
5055 __kmp_gtid_from_thread(team->t.t_threads[f]),
5056 team->t.t_id, f, th->th.th_new_place,
5057 th->th.th_first_place, th->th.th_last_place, spacing));
5058 }
5059 }
5060 }
5061 KMP_DEBUG_ASSERT(update_master_only || place == masters_place);
5062 } else {
5063 int S, rem, gap, s_count;
5064 S = n_th / n_places;
5065 s_count = 0;
5066 rem = n_th - (S * n_places);
5067 gap = rem > 0 ? n_places / rem : n_places;
5068 int place = masters_place;
5069 int gap_ct = gap;
5070 thidx = n_th;
5071 if (update_master_only == 1)
5072 thidx = 1;
5073 for (f = 0; f < thidx; f++) {
5074 kmp_info_t *th = team->t.t_threads[f];
5075 KMP_DEBUG_ASSERT(th != NULL);
5076
5077 __kmp_set_thread_place(team, th, place, place, place);
5078 s_count++;
5079
5080 if ((s_count == S) && rem && (gap_ct == gap)) {
5081 // do nothing, add an extra thread to place on next iteration
5082 } else if ((s_count == S + 1) && rem && (gap_ct == gap)) {
5083 // we added an extra thread to this place; move on to next place
5084 if (place == last_place) {
5085 place = first_place;
5086 } else if (place == (num_masks - 1)) {
5087 place = 0;
5088 } else {
5089 place++;
5090 }
5091 s_count = 0;
5092 gap_ct = 1;
5093 rem--;
5094 } else if (s_count == S) { // place is full; don't add extra thread
5095 if (place == last_place) {
5096 place = first_place;
5097 } else if (place == (num_masks - 1)) {
5098 place = 0;
5099 } else {
5100 place++;
5101 }
5102 gap_ct++;
5103 s_count = 0;
5104 }
5105
5106 KA_TRACE(100, ("__kmp_partition_places: spread: T#%d(%d:%d) place %d "
5107 "partition = [%d,%d]\n",
5108 __kmp_gtid_from_thread(team->t.t_threads[f]),
5109 team->t.t_id, f, th->th.th_new_place,
5110 th->th.th_first_place, th->th.th_last_place));
5111 }
5112 KMP_DEBUG_ASSERT(update_master_only || place == masters_place);
5113 }
5114 } break;
5115
5116 default:
5117 break;
5118 }
5119
5120 KA_TRACE(20, ("__kmp_partition_places: exit T#%d\n", team->t.t_id));
5121}
5122
5123#endif // KMP_AFFINITY_SUPPORTED
5124
5125/* allocate a new team data structure to use. take one off of the free pool if
5126 available */
5127kmp_team_t *__kmp_allocate_team(kmp_root_t *root, int new_nproc, int max_nproc,
5128#if OMPT_SUPPORT
5129 ompt_data_t ompt_parallel_data,
5130#endif
5131 kmp_proc_bind_t new_proc_bind,
5132 kmp_internal_control_t *new_icvs, int argc,
5133 kmp_info_t *master) {
5134 KMP_TIME_DEVELOPER_PARTITIONED_BLOCK(KMP_allocate_team);
5135 int f;
5136 kmp_team_t *team;
5137 int use_hot_team = !root->r.r_active;
5138 int level = 0;
5139 int do_place_partition = 1;
5140
5141 KA_TRACE(20, ("__kmp_allocate_team: called\n"));
5142 KMP_DEBUG_ASSERT(new_nproc >= 1 && argc >= 0);
5143 KMP_DEBUG_ASSERT(max_nproc >= new_nproc);
5144 KMP_MB();
5145
5146 kmp_hot_team_ptr_t *hot_teams;
5147 if (master) {
5148 team = master->th.th_team;
5149 level = team->t.t_active_level;
5150 if (master->th.th_teams_microtask) { // in teams construct?
5151 if (master->th.th_teams_size.nteams > 1 &&
5152 ( // #teams > 1
5153 team->t.t_pkfn ==
5154 (microtask_t)__kmp_teams_master || // inner fork of the teams
5155 master->th.th_teams_level <
5156 team->t.t_level)) { // or nested parallel inside the teams
5157 ++level; // not increment if #teams==1, or for outer fork of the teams;
5158 // increment otherwise
5159 }
5160 // Do not perform the place partition if inner fork of the teams
5161 // Wait until nested parallel region encountered inside teams construct
5162 if ((master->th.th_teams_size.nteams == 1 &&
5163 master->th.th_teams_level >= team->t.t_level) ||
5164 (team->t.t_pkfn == (microtask_t)__kmp_teams_master))
5165 do_place_partition = 0;
5166 }
5167 hot_teams = master->th.th_hot_teams;
5168 if (level < __kmp_hot_teams_max_level && hot_teams &&
5169 hot_teams[level].hot_team) {
5170 // hot team has already been allocated for given level
5171 use_hot_team = 1;
5172 } else {
5173 use_hot_team = 0;
5174 }
5175 } else {
5176 // check we won't access uninitialized hot_teams, just in case
5177 KMP_DEBUG_ASSERT(new_nproc == 1);
5178 }
5179 // Optimization to use a "hot" team
5180 if (use_hot_team && new_nproc > 1) {
5181 KMP_DEBUG_ASSERT(new_nproc <= max_nproc);
5182 team = hot_teams[level].hot_team;
5183#if KMP_DEBUG
5185 KA_TRACE(20, ("__kmp_allocate_team: hot team task_team[0] = %p "
5186 "task_team[1] = %p before reinit\n",
5187 team->t.t_task_team[0], team->t.t_task_team[1]));
5188 }
5189#endif
5190
5191 if (team->t.t_nproc != new_nproc &&
5193 // Distributed barrier may need a resize
5194 int old_nthr = team->t.t_nproc;
5195 __kmp_resize_dist_barrier(team, old_nthr, new_nproc);
5196 }
5197
5198 // If not doing the place partition, then reset the team's proc bind
5199 // to indicate that partitioning of all threads still needs to take place
5200 if (do_place_partition == 0)
5201 team->t.t_proc_bind = proc_bind_default;
5202 // Has the number of threads changed?
5203 /* Let's assume the most common case is that the number of threads is
5204 unchanged, and put that case first. */
5205 if (team->t.t_nproc == new_nproc) { // Check changes in number of threads
5206 KA_TRACE(20, ("__kmp_allocate_team: reusing hot team\n"));
5207 // This case can mean that omp_set_num_threads() was called and the hot
5208 // team size was already reduced, so we check the special flag
5209 if (team->t.t_size_changed == -1) {
5210 team->t.t_size_changed = 1;
5211 } else {
5212 KMP_CHECK_UPDATE(team->t.t_size_changed, 0);
5213 }
5214
5215 // TODO???: team->t.t_max_active_levels = new_max_active_levels;
5216 kmp_r_sched_t new_sched = new_icvs->sched;
5217 // set primary thread's schedule as new run-time schedule
5218 KMP_CHECK_UPDATE(team->t.t_sched.sched, new_sched.sched);
5219
5220 __kmp_reinitialize_team(team, new_icvs,
5221 root->r.r_uber_thread->th.th_ident);
5222
5223 KF_TRACE(10, ("__kmp_allocate_team2: T#%d, this_thread=%p team=%p\n", 0,
5224 team->t.t_threads[0], team));
5225 __kmp_push_current_task_to_thread(team->t.t_threads[0], team, 0);
5226
5227#if KMP_AFFINITY_SUPPORTED
5228 if ((team->t.t_size_changed == 0) &&
5229 (team->t.t_proc_bind == new_proc_bind)) {
5230 if (new_proc_bind == proc_bind_spread) {
5231 if (do_place_partition) {
5232 // add flag to update only master for spread
5233 __kmp_partition_places(team, 1);
5234 }
5235 }
5236 KA_TRACE(200, ("__kmp_allocate_team: reusing hot team #%d bindings: "
5237 "proc_bind = %d, partition = [%d,%d]\n",
5238 team->t.t_id, new_proc_bind, team->t.t_first_place,
5239 team->t.t_last_place));
5240 } else {
5241 if (do_place_partition) {
5242 KMP_CHECK_UPDATE(team->t.t_proc_bind, new_proc_bind);
5243 __kmp_partition_places(team);
5244 }
5245 }
5246#else
5247 KMP_CHECK_UPDATE(team->t.t_proc_bind, new_proc_bind);
5248#endif /* KMP_AFFINITY_SUPPORTED */
5249 } else if (team->t.t_nproc > new_nproc) {
5250 KA_TRACE(20,
5251 ("__kmp_allocate_team: decreasing hot team thread count to %d\n",
5252 new_nproc));
5253
5254 team->t.t_size_changed = 1;
5256 // Barrier size already reduced earlier in this function
5257 // Activate team threads via th_used_in_team
5258 __kmp_add_threads_to_team(team, new_nproc);
5259 }
5260 // When decreasing team size, threads no longer in the team should
5261 // unref task team.
5263 for (f = new_nproc; f < team->t.t_nproc; f++) {
5264 kmp_info_t *th = team->t.t_threads[f];
5265 KMP_DEBUG_ASSERT(th);
5266 th->th.th_task_team = NULL;
5267 }
5268 }
5269 if (__kmp_hot_teams_mode == 0) {
5270 // AC: saved number of threads should correspond to team's value in this
5271 // mode, can be bigger in mode 1, when hot team has threads in reserve
5272 KMP_DEBUG_ASSERT(hot_teams[level].hot_team_nth == team->t.t_nproc);
5273 hot_teams[level].hot_team_nth = new_nproc;
5274 /* release the extra threads we don't need any more */
5275 for (f = new_nproc; f < team->t.t_nproc; f++) {
5276 KMP_DEBUG_ASSERT(team->t.t_threads[f]);
5277 __kmp_free_thread(team->t.t_threads[f]);
5278 team->t.t_threads[f] = NULL;
5279 }
5280 } // (__kmp_hot_teams_mode == 0)
5281 else {
5282 // When keeping extra threads in team, switch threads to wait on own
5283 // b_go flag
5284 for (f = new_nproc; f < team->t.t_nproc; ++f) {
5285 KMP_DEBUG_ASSERT(team->t.t_threads[f]);
5286 kmp_balign_t *balign = team->t.t_threads[f]->th.th_bar;
5287 for (int b = 0; b < bs_last_barrier; ++b) {
5288 if (balign[b].bb.wait_flag == KMP_BARRIER_PARENT_FLAG) {
5289 balign[b].bb.wait_flag = KMP_BARRIER_SWITCH_TO_OWN_FLAG;
5290 }
5291 KMP_CHECK_UPDATE(balign[b].bb.leaf_kids, 0);
5292 }
5293 }
5294 }
5295 team->t.t_nproc = new_nproc;
5296 // TODO???: team->t.t_max_active_levels = new_max_active_levels;
5297 KMP_CHECK_UPDATE(team->t.t_sched.sched, new_icvs->sched.sched);
5298 __kmp_reinitialize_team(team, new_icvs,
5299 root->r.r_uber_thread->th.th_ident);
5300
5301 // Update remaining threads
5302 for (f = 0; f < new_nproc; ++f) {
5303 team->t.t_threads[f]->th.th_team_nproc = new_nproc;
5304 }
5305
5306 // restore the current task state of the primary thread: should be the
5307 // implicit task
5308 KF_TRACE(10, ("__kmp_allocate_team: T#%d, this_thread=%p team=%p\n", 0,
5309 team->t.t_threads[0], team));
5310
5311 __kmp_push_current_task_to_thread(team->t.t_threads[0], team, 0);
5312
5313#ifdef KMP_DEBUG
5314 for (f = 0; f < team->t.t_nproc; f++) {
5315 KMP_DEBUG_ASSERT(team->t.t_threads[f] &&
5316 team->t.t_threads[f]->th.th_team_nproc ==
5317 team->t.t_nproc);
5318 }
5319#endif
5320
5321 if (do_place_partition) {
5322 KMP_CHECK_UPDATE(team->t.t_proc_bind, new_proc_bind);
5323#if KMP_AFFINITY_SUPPORTED
5324 __kmp_partition_places(team);
5325#endif
5326 }
5327 } else { // team->t.t_nproc < new_nproc
5328
5329 KA_TRACE(20,
5330 ("__kmp_allocate_team: increasing hot team thread count to %d\n",
5331 new_nproc));
5332 int old_nproc = team->t.t_nproc; // save old value and use to update only
5333 team->t.t_size_changed = 1;
5334
5335 int avail_threads = hot_teams[level].hot_team_nth;
5336 if (new_nproc < avail_threads)
5337 avail_threads = new_nproc;
5338 kmp_info_t **other_threads = team->t.t_threads;
5339 for (f = team->t.t_nproc; f < avail_threads; ++f) {
5340 // Adjust barrier data of reserved threads (if any) of the team
5341 // Other data will be set in __kmp_initialize_info() below.
5342 int b;
5343 kmp_balign_t *balign = other_threads[f]->th.th_bar;
5344 for (b = 0; b < bs_last_barrier; ++b) {
5345 balign[b].bb.b_arrived = team->t.t_bar[b].b_arrived;
5346 KMP_DEBUG_ASSERT(balign[b].bb.wait_flag != KMP_BARRIER_PARENT_FLAG);
5347#if USE_DEBUGGER
5348 balign[b].bb.b_worker_arrived = team->t.t_bar[b].b_team_arrived;
5349#endif
5350 }
5351 }
5352 if (hot_teams[level].hot_team_nth >= new_nproc) {
5353 // we have all needed threads in reserve, no need to allocate any
5354 // this only possible in mode 1, cannot have reserved threads in mode 0
5356 team->t.t_nproc = new_nproc; // just get reserved threads involved
5357 } else {
5358 // We may have some threads in reserve, but not enough;
5359 // get reserved threads involved if any.
5360 team->t.t_nproc = hot_teams[level].hot_team_nth;
5361 hot_teams[level].hot_team_nth = new_nproc; // adjust hot team max size
5362 if (team->t.t_max_nproc < new_nproc) {
5363 /* reallocate larger arrays */
5364 __kmp_reallocate_team_arrays(team, new_nproc);
5365 __kmp_reinitialize_team(team, new_icvs, NULL);
5366 }
5367
5368#if (KMP_OS_LINUX || KMP_OS_FREEBSD || KMP_OS_NETBSD || KMP_OS_DRAGONFLY) && \
5369 KMP_AFFINITY_SUPPORTED
5370 /* Temporarily set full mask for primary thread before creation of
5371 workers. The reason is that workers inherit the affinity from the
5372 primary thread, so if a lot of workers are created on the single
5373 core quickly, they don't get a chance to set their own affinity for
5374 a long time. */
5375 kmp_affinity_raii_t new_temp_affinity{__kmp_affin_fullMask};
5376#endif
5377
5378 /* allocate new threads for the hot team */
5379 for (f = team->t.t_nproc; f < new_nproc; f++) {
5380 kmp_info_t *new_worker = __kmp_allocate_thread(root, team, f);
5381 KMP_DEBUG_ASSERT(new_worker);
5382 team->t.t_threads[f] = new_worker;
5383
5384 KA_TRACE(20,
5385 ("__kmp_allocate_team: team %d init T#%d arrived: "
5386 "join=%llu, plain=%llu\n",
5387 team->t.t_id, __kmp_gtid_from_tid(f, team), team->t.t_id, f,
5388 team->t.t_bar[bs_forkjoin_barrier].b_arrived,
5389 team->t.t_bar[bs_plain_barrier].b_arrived));
5390
5391 { // Initialize barrier data for new threads.
5392 int b;
5393 kmp_balign_t *balign = new_worker->th.th_bar;
5394 for (b = 0; b < bs_last_barrier; ++b) {
5395 balign[b].bb.b_arrived = team->t.t_bar[b].b_arrived;
5396 KMP_DEBUG_ASSERT(balign[b].bb.wait_flag !=
5398#if USE_DEBUGGER
5399 balign[b].bb.b_worker_arrived = team->t.t_bar[b].b_team_arrived;
5400#endif
5401 }
5402 }
5403 }
5404
5405#if (KMP_OS_LINUX || KMP_OS_FREEBSD || KMP_OS_NETBSD || KMP_OS_DRAGONFLY) && \
5406 KMP_AFFINITY_SUPPORTED
5407 /* Restore initial primary thread's affinity mask */
5408 new_temp_affinity.restore();
5409#endif
5410 } // end of check of t_nproc vs. new_nproc vs. hot_team_nth
5412 // Barrier size already increased earlier in this function
5413 // Activate team threads via th_used_in_team
5414 __kmp_add_threads_to_team(team, new_nproc);
5415 }
5416 /* make sure everyone is syncronized */
5417 // new threads below
5418 __kmp_initialize_team(team, new_nproc, new_icvs,
5419 root->r.r_uber_thread->th.th_ident);
5420
5421 /* reinitialize the threads */
5422 KMP_DEBUG_ASSERT(team->t.t_nproc == new_nproc);
5423 for (f = 0; f < team->t.t_nproc; ++f)
5424 __kmp_initialize_info(team->t.t_threads[f], team, f,
5425 __kmp_gtid_from_tid(f, team));
5426
5427 // set th_task_state for new threads in hot team with older thread's state
5428 kmp_uint8 old_state = team->t.t_threads[old_nproc - 1]->th.th_task_state;
5429 for (f = old_nproc; f < team->t.t_nproc; ++f)
5430 team->t.t_threads[f]->th.th_task_state = old_state;
5431
5432#ifdef KMP_DEBUG
5433 for (f = 0; f < team->t.t_nproc; ++f) {
5434 KMP_DEBUG_ASSERT(team->t.t_threads[f] &&
5435 team->t.t_threads[f]->th.th_team_nproc ==
5436 team->t.t_nproc);
5437 }
5438#endif
5439
5440 if (do_place_partition) {
5441 KMP_CHECK_UPDATE(team->t.t_proc_bind, new_proc_bind);
5442#if KMP_AFFINITY_SUPPORTED
5443 __kmp_partition_places(team);
5444#endif
5445 }
5446 } // Check changes in number of threads
5447
5448 if (master->th.th_teams_microtask) {
5449 for (f = 1; f < new_nproc; ++f) {
5450 // propagate teams construct specific info to workers
5451 kmp_info_t *thr = team->t.t_threads[f];
5452 thr->th.th_teams_microtask = master->th.th_teams_microtask;
5453 thr->th.th_teams_level = master->th.th_teams_level;
5454 thr->th.th_teams_size = master->th.th_teams_size;
5455 }
5456 }
5457 if (level) {
5458 // Sync barrier state for nested hot teams, not needed for outermost hot
5459 // team.
5460 for (f = 1; f < new_nproc; ++f) {
5461 kmp_info_t *thr = team->t.t_threads[f];
5462 int b;
5463 kmp_balign_t *balign = thr->th.th_bar;
5464 for (b = 0; b < bs_last_barrier; ++b) {
5465 balign[b].bb.b_arrived = team->t.t_bar[b].b_arrived;
5466 KMP_DEBUG_ASSERT(balign[b].bb.wait_flag != KMP_BARRIER_PARENT_FLAG);
5467#if USE_DEBUGGER
5468 balign[b].bb.b_worker_arrived = team->t.t_bar[b].b_team_arrived;
5469#endif
5470 }
5471 }
5472 }
5473
5474 /* reallocate space for arguments if necessary */
5475 __kmp_alloc_argv_entries(argc, team, TRUE);
5476 KMP_CHECK_UPDATE(team->t.t_argc, argc);
5477 // The hot team re-uses the previous task team,
5478 // if untouched during the previous release->gather phase.
5479
5480 KF_TRACE(10, (" hot_team = %p\n", team));
5481
5482#if KMP_DEBUG
5484 KA_TRACE(20, ("__kmp_allocate_team: hot team task_team[0] = %p "
5485 "task_team[1] = %p after reinit\n",
5486 team->t.t_task_team[0], team->t.t_task_team[1]));
5487 }
5488#endif
5489
5490#if OMPT_SUPPORT
5491 __ompt_team_assign_id(team, ompt_parallel_data);
5492#endif
5493
5494 KMP_MB();
5495
5496 return team;
5497 }
5498
5499 /* next, let's try to take one from the team pool */
5500 KMP_MB();
5501 for (team = CCAST(kmp_team_t *, __kmp_team_pool); (team);) {
5502 /* TODO: consider resizing undersized teams instead of reaping them, now
5503 that we have a resizing mechanism */
5504 if (team->t.t_max_nproc >= max_nproc) {
5505 /* take this team from the team pool */
5506 __kmp_team_pool = team->t.t_next_pool;
5507
5508 if (max_nproc > 1 &&
5510 if (!team->t.b) { // Allocate barrier structure
5512 }
5513 }
5514
5515 /* setup the team for fresh use */
5516 __kmp_initialize_team(team, new_nproc, new_icvs, NULL);
5517
5518 KA_TRACE(20, ("__kmp_allocate_team: setting task_team[0] %p and "
5519 "task_team[1] %p to NULL\n",
5520 &team->t.t_task_team[0], &team->t.t_task_team[1]));
5521 team->t.t_task_team[0] = NULL;
5522 team->t.t_task_team[1] = NULL;
5523
5524 /* reallocate space for arguments if necessary */
5525 __kmp_alloc_argv_entries(argc, team, TRUE);
5526 KMP_CHECK_UPDATE(team->t.t_argc, argc);
5527
5528 KA_TRACE(
5529 20, ("__kmp_allocate_team: team %d init arrived: join=%u, plain=%u\n",
5531 { // Initialize barrier data.
5532 int b;
5533 for (b = 0; b < bs_last_barrier; ++b) {
5534 team->t.t_bar[b].b_arrived = KMP_INIT_BARRIER_STATE;
5535#if USE_DEBUGGER
5536 team->t.t_bar[b].b_master_arrived = 0;
5537 team->t.t_bar[b].b_team_arrived = 0;
5538#endif
5539 }
5540 }
5541
5542 team->t.t_proc_bind = new_proc_bind;
5543
5544 KA_TRACE(20, ("__kmp_allocate_team: using team from pool %d.\n",
5545 team->t.t_id));
5546
5547#if OMPT_SUPPORT
5548 __ompt_team_assign_id(team, ompt_parallel_data);
5549#endif
5550
5551 team->t.t_nested_nth = NULL;
5552
5553 KMP_MB();
5554
5555 return team;
5556 }
5557
5558 /* reap team if it is too small, then loop back and check the next one */
5559 // not sure if this is wise, but, will be redone during the hot-teams
5560 // rewrite.
5561 /* TODO: Use technique to find the right size hot-team, don't reap them */
5562 team = __kmp_reap_team(team);
5563 __kmp_team_pool = team;
5564 }
5565
5566 /* nothing available in the pool, no matter, make a new team! */
5567 KMP_MB();
5568 team = (kmp_team_t *)__kmp_allocate(sizeof(kmp_team_t));
5569
5570 /* and set it up */
5571 team->t.t_max_nproc = max_nproc;
5572 if (max_nproc > 1 &&
5574 // Allocate barrier structure
5576 }
5577
5578 /* NOTE well, for some reason allocating one big buffer and dividing it up
5579 seems to really hurt performance a lot on the P4, so, let's not use this */
5580 __kmp_allocate_team_arrays(team, max_nproc);
5581
5582 KA_TRACE(20, ("__kmp_allocate_team: making a new team\n"));
5583 __kmp_initialize_team(team, new_nproc, new_icvs, NULL);
5584
5585 KA_TRACE(20, ("__kmp_allocate_team: setting task_team[0] %p and task_team[1] "
5586 "%p to NULL\n",
5587 &team->t.t_task_team[0], &team->t.t_task_team[1]));
5588 team->t.t_task_team[0] = NULL; // to be removed, as __kmp_allocate zeroes
5589 // memory, no need to duplicate
5590 team->t.t_task_team[1] = NULL; // to be removed, as __kmp_allocate zeroes
5591 // memory, no need to duplicate
5592
5593 if (__kmp_storage_map) {
5594 __kmp_print_team_storage_map("team", team, team->t.t_id, new_nproc);
5595 }
5596
5597 /* allocate space for arguments */
5598 __kmp_alloc_argv_entries(argc, team, FALSE);
5599 team->t.t_argc = argc;
5600
5601 KA_TRACE(20,
5602 ("__kmp_allocate_team: team %d init arrived: join=%u, plain=%u\n",
5604 { // Initialize barrier data.
5605 int b;
5606 for (b = 0; b < bs_last_barrier; ++b) {
5607 team->t.t_bar[b].b_arrived = KMP_INIT_BARRIER_STATE;
5608#if USE_DEBUGGER
5609 team->t.t_bar[b].b_master_arrived = 0;
5610 team->t.t_bar[b].b_team_arrived = 0;
5611#endif
5612 }
5613 }
5614
5615 team->t.t_proc_bind = new_proc_bind;
5616
5617#if OMPT_SUPPORT
5618 __ompt_team_assign_id(team, ompt_parallel_data);
5619 team->t.ompt_serialized_team_info = NULL;
5620#endif
5621
5622 KMP_MB();
5623
5624 team->t.t_nested_nth = NULL;
5625
5626 KA_TRACE(20, ("__kmp_allocate_team: done creating a new team %d.\n",
5627 team->t.t_id));
5628
5629 return team;
5630}
5631
5632/* TODO implement hot-teams at all levels */
5633/* TODO implement lazy thread release on demand (disband request) */
5634
5635/* free the team. return it to the team pool. release all the threads
5636 * associated with it */
5638 int f;
5639 KA_TRACE(20, ("__kmp_free_team: T#%d freeing team %d\n", __kmp_get_gtid(),
5640 team->t.t_id));
5641
5642 /* verify state */
5643 KMP_DEBUG_ASSERT(root);
5644 KMP_DEBUG_ASSERT(team);
5645 KMP_DEBUG_ASSERT(team->t.t_nproc <= team->t.t_max_nproc);
5646 KMP_DEBUG_ASSERT(team->t.t_threads);
5647
5648 int use_hot_team = team == root->r.r_hot_team;
5649 int level;
5650 if (master) {
5651 level = team->t.t_active_level - 1;
5652 if (master->th.th_teams_microtask) { // in teams construct?
5653 if (master->th.th_teams_size.nteams > 1) {
5654 ++level; // level was not increased in teams construct for
5655 // team_of_masters
5656 }
5657 if (team->t.t_pkfn != (microtask_t)__kmp_teams_master &&
5658 master->th.th_teams_level == team->t.t_level) {
5659 ++level; // level was not increased in teams construct for
5660 // team_of_workers before the parallel
5661 } // team->t.t_level will be increased inside parallel
5662 }
5663#if KMP_DEBUG
5664 kmp_hot_team_ptr_t *hot_teams = master->th.th_hot_teams;
5665#endif
5667 KMP_DEBUG_ASSERT(team == hot_teams[level].hot_team);
5668 use_hot_team = 1;
5669 }
5670 }
5671
5672 /* team is done working */
5673 TCW_SYNC_PTR(team->t.t_pkfn,
5674 NULL); // Important for Debugging Support Library.
5675#if KMP_OS_WINDOWS
5676 team->t.t_copyin_counter = 0; // init counter for possible reuse
5677#endif
5678 // Do not reset pointer to parent team to NULL for hot teams.
5679
5680 /* if we are non-hot team, release our threads */
5681 if (!use_hot_team) {
5683 // Wait for threads to reach reapable state
5684 for (f = 1; f < team->t.t_nproc; ++f) {
5685 KMP_DEBUG_ASSERT(team->t.t_threads[f]);
5686 kmp_info_t *th = team->t.t_threads[f];
5687 volatile kmp_uint32 *state = &th->th.th_reap_state;
5688 while (*state != KMP_SAFE_TO_REAP) {
5689#if KMP_OS_WINDOWS
5690 // On Windows a thread can be killed at any time, check this
5691 DWORD ecode;
5692 if (!__kmp_is_thread_alive(th, &ecode)) {
5693 *state = KMP_SAFE_TO_REAP; // reset the flag for dead thread
5694 break;
5695 }
5696#endif
5697 // first check if thread is sleeping
5698 if (th->th.th_sleep_loc)
5700 KMP_CPU_PAUSE();
5701 }
5702 }
5703
5704 // Delete task teams
5705 int tt_idx;
5706 for (tt_idx = 0; tt_idx < 2; ++tt_idx) {
5707 kmp_task_team_t *task_team = team->t.t_task_team[tt_idx];
5708 if (task_team != NULL) {
5709 for (f = 0; f < team->t.t_nproc; ++f) { // threads unref task teams
5710 KMP_DEBUG_ASSERT(team->t.t_threads[f]);
5711 team->t.t_threads[f]->th.th_task_team = NULL;
5712 }
5713 KA_TRACE(
5714 20,
5715 ("__kmp_free_team: T#%d deactivating task_team %p on team %d\n",
5716 __kmp_get_gtid(), task_team, team->t.t_id));
5717 __kmp_free_task_team(master, task_team);
5718 team->t.t_task_team[tt_idx] = NULL;
5719 }
5720 }
5721 }
5722
5723 // Before clearing parent pointer, check if nested_nth list should be freed
5724 if (team->t.t_nested_nth && team->t.t_nested_nth != &__kmp_nested_nth &&
5725 team->t.t_nested_nth != team->t.t_parent->t.t_nested_nth) {
5726 KMP_INTERNAL_FREE(team->t.t_nested_nth->nth);
5727 KMP_INTERNAL_FREE(team->t.t_nested_nth);
5728 }
5729 team->t.t_nested_nth = NULL;
5730
5731 // Reset pointer to parent team only for non-hot teams.
5732 team->t.t_parent = NULL;
5733 team->t.t_level = 0;
5734 team->t.t_active_level = 0;
5735
5736 /* free the worker threads */
5737 for (f = 1; f < team->t.t_nproc; ++f) {
5738 KMP_DEBUG_ASSERT(team->t.t_threads[f]);
5741 &(team->t.t_threads[f]->th.th_used_in_team), 1, 2);
5742 }
5743 __kmp_free_thread(team->t.t_threads[f]);
5744 }
5745
5747 if (team->t.b) {
5748 // wake up thread at old location
5749 team->t.b->go_release();
5751 for (f = 1; f < team->t.t_nproc; ++f) {
5752 if (team->t.b->sleep[f].sleep) {
5754 team->t.t_threads[f]->th.th_info.ds.ds_gtid,
5755 (kmp_atomic_flag_64<> *)NULL);
5756 }
5757 }
5758 }
5759 // Wait for threads to be removed from team
5760 for (int f = 1; f < team->t.t_nproc; ++f) {
5761 while (team->t.t_threads[f]->th.th_used_in_team.load() != 0)
5762 KMP_CPU_PAUSE();
5763 }
5764 }
5765 }
5766
5767 for (f = 1; f < team->t.t_nproc; ++f) {
5768 team->t.t_threads[f] = NULL;
5769 }
5770
5771 if (team->t.t_max_nproc > 1 &&
5774 team->t.b = NULL;
5775 }
5776 /* put the team back in the team pool */
5777 /* TODO limit size of team pool, call reap_team if pool too large */
5778 team->t.t_next_pool = CCAST(kmp_team_t *, __kmp_team_pool);
5779 __kmp_team_pool = (volatile kmp_team_t *)team;
5780 } else { // Check if team was created for primary threads in teams construct
5781 // See if first worker is a CG root
5782 KMP_DEBUG_ASSERT(team->t.t_threads[1] &&
5783 team->t.t_threads[1]->th.th_cg_roots);
5784 if (team->t.t_threads[1]->th.th_cg_roots->cg_root == team->t.t_threads[1]) {
5785 // Clean up the CG root nodes on workers so that this team can be re-used
5786 for (f = 1; f < team->t.t_nproc; ++f) {
5787 kmp_info_t *thr = team->t.t_threads[f];
5788 KMP_DEBUG_ASSERT(thr && thr->th.th_cg_roots &&
5789 thr->th.th_cg_roots->cg_root == thr);
5790 // Pop current CG root off list
5791 kmp_cg_root_t *tmp = thr->th.th_cg_roots;
5792 thr->th.th_cg_roots = tmp->up;
5793 KA_TRACE(100, ("__kmp_free_team: Thread %p popping node %p and moving"
5794 " up to node %p. cg_nthreads was %d\n",
5795 thr, tmp, thr->th.th_cg_roots, tmp->cg_nthreads));
5796 int i = tmp->cg_nthreads--;
5797 if (i == 1) {
5798 __kmp_free(tmp); // free CG if we are the last thread in it
5799 }
5800 // Restore current task's thread_limit from CG root
5801 if (thr->th.th_cg_roots)
5802 thr->th.th_current_task->td_icvs.thread_limit =
5803 thr->th.th_cg_roots->cg_thread_limit;
5804 }
5805 }
5806 }
5807
5808 KMP_MB();
5809}
5810
5811/* reap the team. destroy it, reclaim all its resources and free its memory */
5813 kmp_team_t *next_pool = team->t.t_next_pool;
5814
5815 KMP_DEBUG_ASSERT(team);
5816 KMP_DEBUG_ASSERT(team->t.t_dispatch);
5817 KMP_DEBUG_ASSERT(team->t.t_disp_buffer);
5818 KMP_DEBUG_ASSERT(team->t.t_threads);
5819 KMP_DEBUG_ASSERT(team->t.t_argv);
5820
5821 /* TODO clean the threads that are a part of this? */
5822
5823 /* free stuff */
5825 if (team->t.t_argv != &team->t.t_inline_argv[0])
5826 __kmp_free((void *)team->t.t_argv);
5827 __kmp_free(team);
5828
5829 KMP_MB();
5830 return next_pool;
5831}
5832
5833// Free the thread. Don't reap it, just place it on the pool of available
5834// threads.
5835//
5836// Changes for Quad issue 527845: We need a predictable OMP tid <-> gtid
5837// binding for the affinity mechanism to be useful.
5838//
5839// Now, we always keep the free list (__kmp_thread_pool) sorted by gtid.
5840// However, we want to avoid a potential performance problem by always
5841// scanning through the list to find the correct point at which to insert
5842// the thread (potential N**2 behavior). To do this we keep track of the
5843// last place a thread struct was inserted (__kmp_thread_pool_insert_pt).
5844// With single-level parallelism, threads will always be added to the tail
5845// of the list, kept track of by __kmp_thread_pool_insert_pt. With nested
5846// parallelism, all bets are off and we may need to scan through the entire
5847// free list.
5848//
5849// This change also has a potentially large performance benefit, for some
5850// applications. Previously, as threads were freed from the hot team, they
5851// would be placed back on the free list in inverse order. If the hot team
5852// grew back to it's original size, then the freed thread would be placed
5853// back on the hot team in reverse order. This could cause bad cache
5854// locality problems on programs where the size of the hot team regularly
5855// grew and shrunk.
5856//
5857// Now, for single-level parallelism, the OMP tid is always == gtid.
5859 int gtid;
5860 kmp_info_t **scan;
5861
5862 KA_TRACE(20, ("__kmp_free_thread: T#%d putting T#%d back on free pool.\n",
5863 __kmp_get_gtid(), this_th->th.th_info.ds.ds_gtid));
5864
5865 KMP_DEBUG_ASSERT(this_th);
5866
5867 // When moving thread to pool, switch thread to wait on own b_go flag, and
5868 // uninitialized (NULL team).
5869 int b;
5870 kmp_balign_t *balign = this_th->th.th_bar;
5871 for (b = 0; b < bs_last_barrier; ++b) {
5872 if (balign[b].bb.wait_flag == KMP_BARRIER_PARENT_FLAG)
5873 balign[b].bb.wait_flag = KMP_BARRIER_SWITCH_TO_OWN_FLAG;
5874 balign[b].bb.team = NULL;
5875 balign[b].bb.leaf_kids = 0;
5876 }
5877 this_th->th.th_task_state = 0;
5878 this_th->th.th_reap_state = KMP_SAFE_TO_REAP;
5879
5880 /* put thread back on the free pool */
5881 TCW_PTR(this_th->th.th_team, NULL);
5882 TCW_PTR(this_th->th.th_root, NULL);
5883 TCW_PTR(this_th->th.th_dispatch, NULL); /* NOT NEEDED */
5884
5885 while (this_th->th.th_cg_roots) {
5886 this_th->th.th_cg_roots->cg_nthreads--;
5887 KA_TRACE(100, ("__kmp_free_thread: Thread %p decrement cg_nthreads on node"
5888 " %p of thread %p to %d\n",
5889 this_th, this_th->th.th_cg_roots,
5890 this_th->th.th_cg_roots->cg_root,
5891 this_th->th.th_cg_roots->cg_nthreads));
5892 kmp_cg_root_t *tmp = this_th->th.th_cg_roots;
5893 if (tmp->cg_root == this_th) { // Thread is a cg_root
5894 KMP_DEBUG_ASSERT(tmp->cg_nthreads == 0);
5895 KA_TRACE(
5896 5, ("__kmp_free_thread: Thread %p freeing node %p\n", this_th, tmp));
5897 this_th->th.th_cg_roots = tmp->up;
5898 __kmp_free(tmp);
5899 } else { // Worker thread
5900 if (tmp->cg_nthreads == 0) { // last thread leaves contention group
5901 __kmp_free(tmp);
5902 }
5903 this_th->th.th_cg_roots = NULL;
5904 break;
5905 }
5906 }
5907
5908 /* If the implicit task assigned to this thread can be used by other threads
5909 * -> multiple threads can share the data and try to free the task at
5910 * __kmp_reap_thread at exit. This duplicate use of the task data can happen
5911 * with higher probability when hot team is disabled but can occurs even when
5912 * the hot team is enabled */
5913 __kmp_free_implicit_task(this_th);
5914 this_th->th.th_current_task = NULL;
5915
5916 // If the __kmp_thread_pool_insert_pt is already past the new insert
5917 // point, then we need to re-scan the entire list.
5918 gtid = this_th->th.th_info.ds.ds_gtid;
5919 if (__kmp_thread_pool_insert_pt != NULL) {
5921 if (__kmp_thread_pool_insert_pt->th.th_info.ds.ds_gtid > gtid) {
5923 }
5924 }
5925
5926 // Scan down the list to find the place to insert the thread.
5927 // scan is the address of a link in the list, possibly the address of
5928 // __kmp_thread_pool itself.
5929 //
5930 // In the absence of nested parallelism, the for loop will have 0 iterations.
5931 if (__kmp_thread_pool_insert_pt != NULL) {
5932 scan = &(__kmp_thread_pool_insert_pt->th.th_next_pool);
5933 } else {
5934 scan = CCAST(kmp_info_t **, &__kmp_thread_pool);
5935 }
5936 for (; (*scan != NULL) && ((*scan)->th.th_info.ds.ds_gtid < gtid);
5937 scan = &((*scan)->th.th_next_pool))
5938 ;
5939
5940 // Insert the new element on the list, and set __kmp_thread_pool_insert_pt
5941 // to its address.
5942 TCW_PTR(this_th->th.th_next_pool, *scan);
5943 __kmp_thread_pool_insert_pt = *scan = this_th;
5944 KMP_DEBUG_ASSERT((this_th->th.th_next_pool == NULL) ||
5945 (this_th->th.th_info.ds.ds_gtid <
5946 this_th->th.th_next_pool->th.th_info.ds.ds_gtid));
5947 TCW_4(this_th->th.th_in_pool, TRUE);
5949 __kmp_lock_suspend_mx(this_th);
5950 if (this_th->th.th_active == TRUE) {
5952 this_th->th.th_active_in_pool = TRUE;
5953 }
5954#if KMP_DEBUG
5955 else {
5956 KMP_DEBUG_ASSERT(this_th->th.th_active_in_pool == FALSE);
5957 }
5958#endif
5959 __kmp_unlock_suspend_mx(this_th);
5960
5962
5963#ifdef KMP_ADJUST_BLOCKTIME
5964 /* Adjust blocktime back to user setting or default if necessary */
5965 /* Middle initialization might never have occurred */
5966 if (!__kmp_env_blocktime && (__kmp_avail_proc > 0)) {
5968 if (__kmp_nth <= __kmp_avail_proc) {
5969 __kmp_zero_bt = FALSE;
5970 }
5971 }
5972#endif /* KMP_ADJUST_BLOCKTIME */
5973
5974 KMP_MB();
5975}
5976
5977/* ------------------------------------------------------------------------ */
5978
5980#if OMP_PROFILING_SUPPORT
5981 ProfileTraceFile = getenv("LIBOMPTARGET_PROFILE");
5982 // TODO: add a configuration option for time granularity
5983 if (ProfileTraceFile)
5984 llvm::timeTraceProfilerInitialize(500 /* us */, "libomptarget");
5985#endif
5986
5987 int gtid = this_thr->th.th_info.ds.ds_gtid;
5988 /* void *stack_data;*/
5989 kmp_team_t **volatile pteam;
5990
5991 KMP_MB();
5992 KA_TRACE(10, ("__kmp_launch_thread: T#%d start\n", gtid));
5993
5995 this_thr->th.th_cons = __kmp_allocate_cons_stack(gtid); // ATT: Memory leak?
5996 }
5997
5998#if OMPD_SUPPORT
5999 if (ompd_state & OMPD_ENABLE_BP)
6000 ompd_bp_thread_begin();
6001#endif
6002
6003#if OMPT_SUPPORT
6004 ompt_data_t *thread_data = nullptr;
6005 if (ompt_enabled.enabled) {
6006 thread_data = &(this_thr->th.ompt_thread_info.thread_data);
6007 *thread_data = ompt_data_none;
6008
6009 this_thr->th.ompt_thread_info.state = ompt_state_overhead;
6010 this_thr->th.ompt_thread_info.wait_id = 0;
6011 this_thr->th.ompt_thread_info.idle_frame = OMPT_GET_FRAME_ADDRESS(0);
6012 this_thr->th.ompt_thread_info.parallel_flags = 0;
6013 if (ompt_enabled.ompt_callback_thread_begin) {
6014 ompt_callbacks.ompt_callback(ompt_callback_thread_begin)(
6015 ompt_thread_worker, thread_data);
6016 }
6017 this_thr->th.ompt_thread_info.state = ompt_state_idle;
6018 }
6019#endif
6020
6021 /* This is the place where threads wait for work */
6022 while (!TCR_4(__kmp_global.g.g_done)) {
6023 KMP_DEBUG_ASSERT(this_thr == __kmp_threads[gtid]);
6024 KMP_MB();
6025
6026 /* wait for work to do */
6027 KA_TRACE(20, ("__kmp_launch_thread: T#%d waiting for work\n", gtid));
6028
6029 /* No tid yet since not part of a team */
6031
6032#if OMPT_SUPPORT
6033 if (ompt_enabled.enabled) {
6034 this_thr->th.ompt_thread_info.state = ompt_state_overhead;
6035 }
6036#endif
6037
6038 pteam = &this_thr->th.th_team;
6039
6040 /* have we been allocated? */
6041 if (TCR_SYNC_PTR(*pteam) && !TCR_4(__kmp_global.g.g_done)) {
6042 /* we were just woken up, so run our new task */
6043 if (TCR_SYNC_PTR((*pteam)->t.t_pkfn) != NULL) {
6044 int rc;
6045 KA_TRACE(20,
6046 ("__kmp_launch_thread: T#%d(%d:%d) invoke microtask = %p\n",
6047 gtid, (*pteam)->t.t_id, __kmp_tid_from_gtid(gtid),
6048 (*pteam)->t.t_pkfn));
6049
6050 updateHWFPControl(*pteam);
6051
6052#if OMPT_SUPPORT
6053 if (ompt_enabled.enabled) {
6054 this_thr->th.ompt_thread_info.state = ompt_state_work_parallel;
6055 }
6056#endif
6057
6058 rc = (*pteam)->t.t_invoke(gtid);
6059 KMP_ASSERT(rc);
6060
6061 KMP_MB();
6062 KA_TRACE(20, ("__kmp_launch_thread: T#%d(%d:%d) done microtask = %p\n",
6063 gtid, (*pteam)->t.t_id, __kmp_tid_from_gtid(gtid),
6064 (*pteam)->t.t_pkfn));
6065 }
6066#if OMPT_SUPPORT
6067 if (ompt_enabled.enabled) {
6068 /* no frame set while outside task */
6069 __ompt_get_task_info_object(0)->frame.exit_frame = ompt_data_none;
6070
6071 this_thr->th.ompt_thread_info.state = ompt_state_overhead;
6072 }
6073#endif
6074 /* join barrier after parallel region */
6075 __kmp_join_barrier(gtid);
6076 }
6077 }
6078
6079#if OMPD_SUPPORT
6080 if (ompd_state & OMPD_ENABLE_BP)
6081 ompd_bp_thread_end();
6082#endif
6083
6084#if OMPT_SUPPORT
6085 if (ompt_enabled.ompt_callback_thread_end) {
6086 ompt_callbacks.ompt_callback(ompt_callback_thread_end)(thread_data);
6087 }
6088#endif
6089
6090 this_thr->th.th_task_team = NULL;
6091 /* run the destructors for the threadprivate data for this thread */
6093
6094 KA_TRACE(10, ("__kmp_launch_thread: T#%d done\n", gtid));
6095 KMP_MB();
6096
6097#if OMP_PROFILING_SUPPORT
6098 llvm::timeTraceProfilerFinishThread();
6099#endif
6100 return this_thr;
6101}
6102
6103/* ------------------------------------------------------------------------ */
6104
6105void __kmp_internal_end_dest(void *specific_gtid) {
6106 // Make sure no significant bits are lost
6107 int gtid;
6108 __kmp_type_convert((kmp_intptr_t)specific_gtid - 1, &gtid);
6109
6110 KA_TRACE(30, ("__kmp_internal_end_dest: T#%d\n", gtid));
6111 /* NOTE: the gtid is stored as gitd+1 in the thread-local-storage
6112 * this is because 0 is reserved for the nothing-stored case */
6113
6115}
6116
6117#if KMP_OS_UNIX && KMP_DYNAMIC_LIB
6118
6119__attribute__((destructor)) void __kmp_internal_end_dtor(void) {
6121}
6122
6123#endif
6124
6125/* [Windows] josh: when the atexit handler is called, there may still be more
6126 than one thread alive */
6128 KA_TRACE(30, ("__kmp_internal_end_atexit\n"));
6129 /* [Windows]
6130 josh: ideally, we want to completely shutdown the library in this atexit
6131 handler, but stat code that depends on thread specific data for gtid fails
6132 because that data becomes unavailable at some point during the shutdown, so
6133 we call __kmp_internal_end_thread instead. We should eventually remove the
6134 dependency on __kmp_get_specific_gtid in the stat code and use
6135 __kmp_internal_end_library to cleanly shutdown the library.
6136
6137 // TODO: Can some of this comment about GVS be removed?
6138 I suspect that the offending stat code is executed when the calling thread
6139 tries to clean up a dead root thread's data structures, resulting in GVS
6140 code trying to close the GVS structures for that thread, but since the stat
6141 code uses __kmp_get_specific_gtid to get the gtid with the assumption that
6142 the calling thread is cleaning up itself instead of another thread, it get
6143 confused. This happens because allowing a thread to unregister and cleanup
6144 another thread is a recent modification for addressing an issue.
6145 Based on the current design (20050722), a thread may end up
6146 trying to unregister another thread only if thread death does not trigger
6147 the calling of __kmp_internal_end_thread. For Linux* OS, there is the
6148 thread specific data destructor function to detect thread death. For
6149 Windows dynamic, there is DllMain(THREAD_DETACH). For Windows static, there
6150 is nothing. Thus, the workaround is applicable only for Windows static
6151 stat library. */
6153#if KMP_OS_WINDOWS
6155#endif
6156}
6157
6158static void __kmp_reap_thread(kmp_info_t *thread, int is_root) {
6159 // It is assumed __kmp_forkjoin_lock is acquired.
6160
6161 int gtid;
6162
6163 KMP_DEBUG_ASSERT(thread != NULL);
6164
6165 gtid = thread->th.th_info.ds.ds_gtid;
6166
6167 if (!is_root) {
6169 /* Assume the threads are at the fork barrier here */
6170 KA_TRACE(
6171 20, ("__kmp_reap_thread: releasing T#%d from fork barrier for reap\n",
6172 gtid));
6174 while (
6175 !KMP_COMPARE_AND_STORE_ACQ32(&(thread->th.th_used_in_team), 0, 3))
6176 KMP_CPU_PAUSE();
6178 } else {
6179 /* Need release fence here to prevent seg faults for tree forkjoin
6180 barrier (GEH) */
6181 kmp_flag_64<> flag(&thread->th.th_bar[bs_forkjoin_barrier].bb.b_go,
6182 thread);
6184 }
6185 }
6186
6187 // Terminate OS thread.
6188 __kmp_reap_worker(thread);
6189
6190 // The thread was killed asynchronously. If it was actively
6191 // spinning in the thread pool, decrement the global count.
6192 //
6193 // There is a small timing hole here - if the worker thread was just waking
6194 // up after sleeping in the pool, had reset it's th_active_in_pool flag but
6195 // not decremented the global counter __kmp_thread_pool_active_nth yet, then
6196 // the global counter might not get updated.
6197 //
6198 // Currently, this can only happen as the library is unloaded,
6199 // so there are no harmful side effects.
6200 if (thread->th.th_active_in_pool) {
6201 thread->th.th_active_in_pool = FALSE;
6204 }
6205 }
6206
6208
6209// Free the fast memory for tasking
6210#if USE_FAST_MEMORY
6211 __kmp_free_fast_memory(thread);
6212#endif /* USE_FAST_MEMORY */
6213
6215
6216 KMP_DEBUG_ASSERT(__kmp_threads[gtid] == thread);
6217 TCW_SYNC_PTR(__kmp_threads[gtid], NULL);
6218
6219 --__kmp_all_nth;
6220 // __kmp_nth was decremented when thread is added to the pool.
6221
6222#ifdef KMP_ADJUST_BLOCKTIME
6223 /* Adjust blocktime back to user setting or default if necessary */
6224 /* Middle initialization might never have occurred */
6225 if (!__kmp_env_blocktime && (__kmp_avail_proc > 0)) {
6227 if (__kmp_nth <= __kmp_avail_proc) {
6228 __kmp_zero_bt = FALSE;
6229 }
6230 }
6231#endif /* KMP_ADJUST_BLOCKTIME */
6232
6233 /* free the memory being used */
6235 if (thread->th.th_cons) {
6236 __kmp_free_cons_stack(thread->th.th_cons);
6237 thread->th.th_cons = NULL;
6238 }
6239 }
6240
6241 if (thread->th.th_pri_common != NULL) {
6242 __kmp_free(thread->th.th_pri_common);
6243 thread->th.th_pri_common = NULL;
6244 }
6245
6246#if KMP_USE_BGET
6247 if (thread->th.th_local.bget_data != NULL) {
6248 __kmp_finalize_bget(thread);
6249 }
6250#endif
6251
6252#if KMP_AFFINITY_SUPPORTED
6253 if (thread->th.th_affin_mask != NULL) {
6254 KMP_CPU_FREE(thread->th.th_affin_mask);
6255 thread->th.th_affin_mask = NULL;
6256 }
6257#endif /* KMP_AFFINITY_SUPPORTED */
6258
6259#if KMP_USE_HIER_SCHED
6260 if (thread->th.th_hier_bar_data != NULL) {
6261 __kmp_free(thread->th.th_hier_bar_data);
6262 thread->th.th_hier_bar_data = NULL;
6263 }
6264#endif
6265
6266 __kmp_reap_team(thread->th.th_serial_team);
6267 thread->th.th_serial_team = NULL;
6268 __kmp_free(thread);
6269
6270 KMP_MB();
6271
6272} // __kmp_reap_thread
6273
6275#if USE_ITT_NOTIFY
6276 if (__kmp_itt_region_domains.count > 0) {
6277 for (int i = 0; i < KMP_MAX_FRAME_DOMAINS; ++i) {
6278 kmp_itthash_entry_t *bucket = __kmp_itt_region_domains.buckets[i];
6279 while (bucket) {
6280 kmp_itthash_entry_t *next = bucket->next_in_bucket;
6281 __kmp_thread_free(th, bucket);
6282 bucket = next;
6283 }
6284 }
6285 }
6286 if (__kmp_itt_barrier_domains.count > 0) {
6287 for (int i = 0; i < KMP_MAX_FRAME_DOMAINS; ++i) {
6288 kmp_itthash_entry_t *bucket = __kmp_itt_barrier_domains.buckets[i];
6289 while (bucket) {
6290 kmp_itthash_entry_t *next = bucket->next_in_bucket;
6291 __kmp_thread_free(th, bucket);
6292 bucket = next;
6293 }
6294 }
6295 }
6296#endif
6297}
6298
6299static void __kmp_internal_end(void) {
6300 int i;
6301
6302 /* First, unregister the library */
6304
6305#if KMP_OS_WINDOWS
6306 /* In Win static library, we can't tell when a root actually dies, so we
6307 reclaim the data structures for any root threads that have died but not
6308 unregistered themselves, in order to shut down cleanly.
6309 In Win dynamic library we also can't tell when a thread dies. */
6310 __kmp_reclaim_dead_roots(); // AC: moved here to always clean resources of
6311// dead roots
6312#endif
6313
6314 for (i = 0; i < __kmp_threads_capacity; i++)
6315 if (__kmp_root[i])
6316 if (__kmp_root[i]->r.r_active)
6317 break;
6318 KMP_MB(); /* Flush all pending memory write invalidates. */
6319 TCW_SYNC_4(__kmp_global.g.g_done, TRUE);
6320
6321 if (i < __kmp_threads_capacity) {
6322#if KMP_USE_MONITOR
6323 // 2009-09-08 (lev): Other alive roots found. Why do we kill the monitor??
6324 KMP_MB(); /* Flush all pending memory write invalidates. */
6325
6326 // Need to check that monitor was initialized before reaping it. If we are
6327 // called form __kmp_atfork_child (which sets __kmp_init_parallel = 0), then
6328 // __kmp_monitor will appear to contain valid data, but it is only valid in
6329 // the parent process, not the child.
6330 // New behavior (201008): instead of keying off of the flag
6331 // __kmp_init_parallel, the monitor thread creation is keyed off
6332 // of the new flag __kmp_init_monitor.
6333 __kmp_acquire_bootstrap_lock(&__kmp_monitor_lock);
6334 if (TCR_4(__kmp_init_monitor)) {
6336 TCW_4(__kmp_init_monitor, 0);
6337 }
6338 __kmp_release_bootstrap_lock(&__kmp_monitor_lock);
6339 KA_TRACE(10, ("__kmp_internal_end: monitor reaped\n"));
6340#endif // KMP_USE_MONITOR
6341 } else {
6342/* TODO move this to cleanup code */
6343#ifdef KMP_DEBUG
6344 /* make sure that everything has properly ended */
6345 for (i = 0; i < __kmp_threads_capacity; i++) {
6346 if (__kmp_root[i]) {
6347 // KMP_ASSERT( ! KMP_UBER_GTID( i ) ); // AC:
6348 // there can be uber threads alive here
6349 KMP_ASSERT(!__kmp_root[i]->r.r_active); // TODO: can they be active?
6350 }
6351 }
6352#endif
6353
6354 KMP_MB();
6355
6356 // Reap the worker threads.
6357 // This is valid for now, but be careful if threads are reaped sooner.
6358 while (__kmp_thread_pool != NULL) { // Loop thru all the thread in the pool.
6359 // Get the next thread from the pool.
6361 __kmp_thread_pool = thread->th.th_next_pool;
6362 // Reap it.
6363 KMP_DEBUG_ASSERT(thread->th.th_reap_state == KMP_SAFE_TO_REAP);
6364 thread->th.th_next_pool = NULL;
6365 thread->th.th_in_pool = FALSE;
6366 __kmp_reap_thread(thread, 0);
6367 }
6369
6370 // Reap teams.
6371 while (__kmp_team_pool != NULL) { // Loop thru all the teams in the pool.
6372 // Get the next team from the pool.
6374 __kmp_team_pool = team->t.t_next_pool;
6375 // Reap it.
6376 team->t.t_next_pool = NULL;
6377 __kmp_reap_team(team);
6378 }
6379
6381
6382#if KMP_OS_UNIX
6383 // Threads that are not reaped should not access any resources since they
6384 // are going to be deallocated soon, so the shutdown sequence should wait
6385 // until all threads either exit the final spin-waiting loop or begin
6386 // sleeping after the given blocktime.
6387 for (i = 0; i < __kmp_threads_capacity; i++) {
6388 kmp_info_t *thr = __kmp_threads[i];
6389 while (thr && KMP_ATOMIC_LD_ACQ(&thr->th.th_blocking))
6390 KMP_CPU_PAUSE();
6391 }
6392#endif
6393
6394 for (i = 0; i < __kmp_threads_capacity; ++i) {
6395 // TBD: Add some checking...
6396 // Something like KMP_DEBUG_ASSERT( __kmp_thread[ i ] == NULL );
6397 }
6398
6399 /* Make sure all threadprivate destructors get run by joining with all
6400 worker threads before resetting this flag */
6402
6403 KA_TRACE(10, ("__kmp_internal_end: all workers reaped\n"));
6404 KMP_MB();
6405
6406#if KMP_USE_MONITOR
6407 // See note above: One of the possible fixes for CQ138434 / CQ140126
6408 //
6409 // FIXME: push both code fragments down and CSE them?
6410 // push them into __kmp_cleanup() ?
6411 __kmp_acquire_bootstrap_lock(&__kmp_monitor_lock);
6412 if (TCR_4(__kmp_init_monitor)) {
6414 TCW_4(__kmp_init_monitor, 0);
6415 }
6416 __kmp_release_bootstrap_lock(&__kmp_monitor_lock);
6417 KA_TRACE(10, ("__kmp_internal_end: monitor reaped\n"));
6418#endif
6419 } /* else !__kmp_global.t_active */
6421 KMP_MB(); /* Flush all pending memory write invalidates. */
6422
6423 __kmp_cleanup();
6424#if OMPT_SUPPORT
6425 ompt_fini();
6426#endif
6427}
6428
6429void __kmp_internal_end_library(int gtid_req) {
6430 /* if we have already cleaned up, don't try again, it wouldn't be pretty */
6431 /* this shouldn't be a race condition because __kmp_internal_end() is the
6432 only place to clear __kmp_serial_init */
6433 /* we'll check this later too, after we get the lock */
6434 // 2009-09-06: We do not set g_abort without setting g_done. This check looks
6435 // redundant, because the next check will work in any case.
6436 if (__kmp_global.g.g_abort) {
6437 KA_TRACE(11, ("__kmp_internal_end_library: abort, exiting\n"));
6438 /* TODO abort? */
6439 return;
6440 }
6441 if (TCR_4(__kmp_global.g.g_done) || !__kmp_init_serial) {
6442 KA_TRACE(10, ("__kmp_internal_end_library: already finished\n"));
6443 return;
6444 }
6445
6446 // If hidden helper team has been initialized, we need to deinit it
6450 // First release the main thread to let it continue its work
6452 // Wait until the hidden helper team has been destroyed
6454 }
6455
6456 KMP_MB(); /* Flush all pending memory write invalidates. */
6457 /* find out who we are and what we should do */
6458 {
6459 int gtid = (gtid_req >= 0) ? gtid_req : __kmp_gtid_get_specific();
6460 KA_TRACE(
6461 10, ("__kmp_internal_end_library: enter T#%d (%d)\n", gtid, gtid_req));
6462 if (gtid == KMP_GTID_SHUTDOWN) {
6463 KA_TRACE(10, ("__kmp_internal_end_library: !__kmp_init_runtime, system "
6464 "already shutdown\n"));
6465 return;
6466 } else if (gtid == KMP_GTID_MONITOR) {
6467 KA_TRACE(10, ("__kmp_internal_end_library: monitor thread, gtid not "
6468 "registered, or system shutdown\n"));
6469 return;
6470 } else if (gtid == KMP_GTID_DNE) {
6471 KA_TRACE(10, ("__kmp_internal_end_library: gtid not registered or system "
6472 "shutdown\n"));
6473 /* we don't know who we are, but we may still shutdown the library */
6474 } else if (KMP_UBER_GTID(gtid)) {
6475 /* unregister ourselves as an uber thread. gtid is no longer valid */
6476 if (__kmp_root[gtid]->r.r_active) {
6477 __kmp_global.g.g_abort = -1;
6478 TCW_SYNC_4(__kmp_global.g.g_done, TRUE);
6480 KA_TRACE(10,
6481 ("__kmp_internal_end_library: root still active, abort T#%d\n",
6482 gtid));
6483 return;
6484 } else {
6486 KA_TRACE(
6487 10,
6488 ("__kmp_internal_end_library: unregistering sibling T#%d\n", gtid));
6490 }
6491 } else {
6492/* worker threads may call this function through the atexit handler, if they
6493 * call exit() */
6494/* For now, skip the usual subsequent processing and just dump the debug buffer.
6495 TODO: do a thorough shutdown instead */
6496#ifdef DUMP_DEBUG_ON_EXIT
6497 if (__kmp_debug_buf)
6499#endif
6500 // added unregister library call here when we switch to shm linux
6501 // if we don't, it will leave lots of files in /dev/shm
6502 // cleanup shared memory file before exiting.
6504 return;
6505 }
6506 }
6507 /* synchronize the termination process */
6509
6510 /* have we already finished */
6511 if (__kmp_global.g.g_abort) {
6512 KA_TRACE(10, ("__kmp_internal_end_library: abort, exiting\n"));
6513 /* TODO abort? */
6515 return;
6516 }
6517 if (TCR_4(__kmp_global.g.g_done) || !__kmp_init_serial) {
6519 return;
6520 }
6521
6522 /* We need this lock to enforce mutex between this reading of
6523 __kmp_threads_capacity and the writing by __kmp_register_root.
6524 Alternatively, we can use a counter of roots that is atomically updated by
6525 __kmp_get_global_thread_id_reg, __kmp_do_serial_initialize and
6526 __kmp_internal_end_*. */
6528
6529 /* now we can safely conduct the actual termination */
6531
6534
6535 KA_TRACE(10, ("__kmp_internal_end_library: exit\n"));
6536
6537#ifdef DUMP_DEBUG_ON_EXIT
6538 if (__kmp_debug_buf)
6540#endif
6541
6542#if KMP_OS_WINDOWS
6544#endif
6545
6547
6548} // __kmp_internal_end_library
6549
6550void __kmp_internal_end_thread(int gtid_req) {
6551 int i;
6552
6553 /* if we have already cleaned up, don't try again, it wouldn't be pretty */
6554 /* this shouldn't be a race condition because __kmp_internal_end() is the
6555 * only place to clear __kmp_serial_init */
6556 /* we'll check this later too, after we get the lock */
6557 // 2009-09-06: We do not set g_abort without setting g_done. This check looks
6558 // redundant, because the next check will work in any case.
6559 if (__kmp_global.g.g_abort) {
6560 KA_TRACE(11, ("__kmp_internal_end_thread: abort, exiting\n"));
6561 /* TODO abort? */
6562 return;
6563 }
6564 if (TCR_4(__kmp_global.g.g_done) || !__kmp_init_serial) {
6565 KA_TRACE(10, ("__kmp_internal_end_thread: already finished\n"));
6566 return;
6567 }
6568
6569 // If hidden helper team has been initialized, we need to deinit it
6573 // First release the main thread to let it continue its work
6575 // Wait until the hidden helper team has been destroyed
6577 }
6578
6579 KMP_MB(); /* Flush all pending memory write invalidates. */
6580
6581 /* find out who we are and what we should do */
6582 {
6583 int gtid = (gtid_req >= 0) ? gtid_req : __kmp_gtid_get_specific();
6584 KA_TRACE(10,
6585 ("__kmp_internal_end_thread: enter T#%d (%d)\n", gtid, gtid_req));
6586 if (gtid == KMP_GTID_SHUTDOWN) {
6587 KA_TRACE(10, ("__kmp_internal_end_thread: !__kmp_init_runtime, system "
6588 "already shutdown\n"));
6589 return;
6590 } else if (gtid == KMP_GTID_MONITOR) {
6591 KA_TRACE(10, ("__kmp_internal_end_thread: monitor thread, gtid not "
6592 "registered, or system shutdown\n"));
6593 return;
6594 } else if (gtid == KMP_GTID_DNE) {
6595 KA_TRACE(10, ("__kmp_internal_end_thread: gtid not registered or system "
6596 "shutdown\n"));
6597 return;
6598 /* we don't know who we are */
6599 } else if (KMP_UBER_GTID(gtid)) {
6600 /* unregister ourselves as an uber thread. gtid is no longer valid */
6601 if (__kmp_root[gtid]->r.r_active) {
6602 __kmp_global.g.g_abort = -1;
6603 TCW_SYNC_4(__kmp_global.g.g_done, TRUE);
6604 KA_TRACE(10,
6605 ("__kmp_internal_end_thread: root still active, abort T#%d\n",
6606 gtid));
6607 return;
6608 } else {
6609 KA_TRACE(10, ("__kmp_internal_end_thread: unregistering sibling T#%d\n",
6610 gtid));
6612 }
6613 } else {
6614 /* just a worker thread, let's leave */
6615 KA_TRACE(10, ("__kmp_internal_end_thread: worker thread T#%d\n", gtid));
6616
6617 if (gtid >= 0) {
6618 __kmp_threads[gtid]->th.th_task_team = NULL;
6619 }
6620
6621 KA_TRACE(10,
6622 ("__kmp_internal_end_thread: worker thread done, exiting T#%d\n",
6623 gtid));
6624 return;
6625 }
6626 }
6627#if KMP_DYNAMIC_LIB
6629 // AC: lets not shutdown the dynamic library at the exit of uber thread,
6630 // because we will better shutdown later in the library destructor.
6631 {
6632 KA_TRACE(10, ("__kmp_internal_end_thread: exiting T#%d\n", gtid_req));
6633 return;
6634 }
6635#endif
6636 /* synchronize the termination process */
6638
6639 /* have we already finished */
6640 if (__kmp_global.g.g_abort) {
6641 KA_TRACE(10, ("__kmp_internal_end_thread: abort, exiting\n"));
6642 /* TODO abort? */
6644 return;
6645 }
6646 if (TCR_4(__kmp_global.g.g_done) || !__kmp_init_serial) {
6648 return;
6649 }
6650
6651 /* We need this lock to enforce mutex between this reading of
6652 __kmp_threads_capacity and the writing by __kmp_register_root.
6653 Alternatively, we can use a counter of roots that is atomically updated by
6654 __kmp_get_global_thread_id_reg, __kmp_do_serial_initialize and
6655 __kmp_internal_end_*. */
6656
6657 /* should we finish the run-time? are all siblings done? */
6659
6660 for (i = 0; i < __kmp_threads_capacity; ++i) {
6661 if (KMP_UBER_GTID(i)) {
6662 KA_TRACE(
6663 10,
6664 ("__kmp_internal_end_thread: remaining sibling task: gtid==%d\n", i));
6667 return;
6668 }
6669 }
6670
6671 /* now we can safely conduct the actual termination */
6672
6674
6677
6678 KA_TRACE(10, ("__kmp_internal_end_thread: exit T#%d\n", gtid_req));
6679
6680#ifdef DUMP_DEBUG_ON_EXIT
6681 if (__kmp_debug_buf)
6683#endif
6684} // __kmp_internal_end_thread
6685
6686// -----------------------------------------------------------------------------
6687// Library registration stuff.
6688
6690// Random value used to indicate library initialization.
6691static char *__kmp_registration_str = NULL;
6692// Value to be saved in env var __KMP_REGISTERED_LIB_<pid>.
6693
6694static inline char *__kmp_reg_status_name() {
6695/* On RHEL 3u5 if linked statically, getpid() returns different values in
6696 each thread. If registration and unregistration go in different threads
6697 (omp_misc_other_root_exit.cpp test case), the name of registered_lib_env
6698 env var can not be found, because the name will contain different pid. */
6699// macOS* complains about name being too long with additional getuid()
6700#if KMP_OS_UNIX && !KMP_OS_DARWIN && KMP_DYNAMIC_LIB
6701 return __kmp_str_format("__KMP_REGISTERED_LIB_%d_%d", (int)getpid(),
6702 (int)getuid());
6703#else
6704 return __kmp_str_format("__KMP_REGISTERED_LIB_%d", (int)getpid());
6705#endif
6706} // __kmp_reg_status_get
6707
6708#if defined(KMP_USE_SHM)
6709bool __kmp_shm_available = false;
6710bool __kmp_tmp_available = false;
6711// If /dev/shm is not accessible, we will create a temporary file under /tmp.
6712char *temp_reg_status_file_name = nullptr;
6713#endif
6714
6716
6717 char *name = __kmp_reg_status_name(); // Name of the environment variable.
6718 int done = 0;
6719 union {
6720 double dtime;
6721 long ltime;
6722 } time;
6723#if KMP_ARCH_X86 || KMP_ARCH_X86_64
6725#endif
6726 __kmp_read_system_time(&time.dtime);
6727 __kmp_registration_flag = 0xCAFE0000L | (time.ltime & 0x0000FFFFL);
6730 __kmp_registration_flag, KMP_LIBRARY_FILE);
6731
6732 KA_TRACE(50, ("__kmp_register_library_startup: %s=\"%s\"\n", name,
6734
6735 while (!done) {
6736
6737 char *value = NULL; // Actual value of the environment variable.
6738
6739#if defined(KMP_USE_SHM)
6740 char *shm_name = nullptr;
6741 char *data1 = nullptr;
6742 __kmp_shm_available = __kmp_detect_shm();
6743 if (__kmp_shm_available) {
6744 int fd1 = -1;
6745 shm_name = __kmp_str_format("/%s", name);
6746 int shm_preexist = 0;
6747 fd1 = shm_open(shm_name, O_CREAT | O_EXCL | O_RDWR, 0600);
6748 if ((fd1 == -1) && (errno == EEXIST)) {
6749 // file didn't open because it already exists.
6750 // try opening existing file
6751 fd1 = shm_open(shm_name, O_RDWR, 0600);
6752 if (fd1 == -1) { // file didn't open
6753 KMP_WARNING(FunctionError, "Can't open SHM");
6754 __kmp_shm_available = false;
6755 } else { // able to open existing file
6756 shm_preexist = 1;
6757 }
6758 }
6759 if (__kmp_shm_available && shm_preexist == 0) { // SHM created, set size
6760 if (ftruncate(fd1, SHM_SIZE) == -1) { // error occurred setting size;
6761 KMP_WARNING(FunctionError, "Can't set size of SHM");
6762 __kmp_shm_available = false;
6763 }
6764 }
6765 if (__kmp_shm_available) { // SHM exists, now map it
6766 data1 = (char *)mmap(0, SHM_SIZE, PROT_READ | PROT_WRITE, MAP_SHARED,
6767 fd1, 0);
6768 if (data1 == MAP_FAILED) { // failed to map shared memory
6769 KMP_WARNING(FunctionError, "Can't map SHM");
6770 __kmp_shm_available = false;
6771 }
6772 }
6773 if (__kmp_shm_available) { // SHM mapped
6774 if (shm_preexist == 0) { // set data to SHM, set value
6775 KMP_STRCPY_S(data1, SHM_SIZE, __kmp_registration_str);
6776 }
6777 // Read value from either what we just wrote or existing file.
6778 value = __kmp_str_format("%s", data1); // read value from SHM
6779 munmap(data1, SHM_SIZE);
6780 }
6781 if (fd1 != -1)
6782 close(fd1);
6783 }
6784 if (!__kmp_shm_available)
6785 __kmp_tmp_available = __kmp_detect_tmp();
6786 if (!__kmp_shm_available && __kmp_tmp_available) {
6787 // SHM failed to work due to an error other than that the file already
6788 // exists. Try to create a temp file under /tmp.
6789 // If /tmp isn't accessible, fall back to using environment variable.
6790 // TODO: /tmp might not always be the temporary directory. For now we will
6791 // not consider TMPDIR.
6792 int fd1 = -1;
6793 temp_reg_status_file_name = __kmp_str_format("/tmp/%s", name);
6794 int tmp_preexist = 0;
6795 fd1 = open(temp_reg_status_file_name, O_CREAT | O_EXCL | O_RDWR, 0600);
6796 if ((fd1 == -1) && (errno == EEXIST)) {
6797 // file didn't open because it already exists.
6798 // try opening existing file
6799 fd1 = open(temp_reg_status_file_name, O_RDWR, 0600);
6800 if (fd1 == -1) { // file didn't open if (fd1 == -1) {
6801 KMP_WARNING(FunctionError, "Can't open TEMP");
6802 __kmp_tmp_available = false;
6803 } else {
6804 tmp_preexist = 1;
6805 }
6806 }
6807 if (__kmp_tmp_available && tmp_preexist == 0) {
6808 // we created /tmp file now set size
6809 if (ftruncate(fd1, SHM_SIZE) == -1) { // error occurred setting size;
6810 KMP_WARNING(FunctionError, "Can't set size of /tmp file");
6811 __kmp_tmp_available = false;
6812 }
6813 }
6814 if (__kmp_tmp_available) {
6815 data1 = (char *)mmap(0, SHM_SIZE, PROT_READ | PROT_WRITE, MAP_SHARED,
6816 fd1, 0);
6817 if (data1 == MAP_FAILED) { // failed to map /tmp
6818 KMP_WARNING(FunctionError, "Can't map /tmp");
6819 __kmp_tmp_available = false;
6820 }
6821 }
6822 if (__kmp_tmp_available) {
6823 if (tmp_preexist == 0) { // set data to TMP, set value
6824 KMP_STRCPY_S(data1, SHM_SIZE, __kmp_registration_str);
6825 }
6826 // Read value from either what we just wrote or existing file.
6827 value = __kmp_str_format("%s", data1); // read value from SHM
6828 munmap(data1, SHM_SIZE);
6829 }
6830 if (fd1 != -1)
6831 close(fd1);
6832 }
6833 if (!__kmp_shm_available && !__kmp_tmp_available) {
6834 // no /dev/shm and no /tmp -- fall back to environment variable
6835 // Set environment variable, but do not overwrite if it exists.
6837 // read value to see if it got set
6839 }
6840#else // Windows and unix with static library
6841 // Set environment variable, but do not overwrite if it exists.
6843 // read value to see if it got set
6845#endif
6846
6847 if (value != NULL && strcmp(value, __kmp_registration_str) == 0) {
6848 done = 1; // Ok, environment variable set successfully, exit the loop.
6849 } else {
6850 // Oops. Write failed. Another copy of OpenMP RTL is in memory.
6851 // Check whether it alive or dead.
6852 int neighbor = 0; // 0 -- unknown status, 1 -- alive, 2 -- dead.
6853 char *tail = value;
6854 char *flag_addr_str = NULL;
6855 char *flag_val_str = NULL;
6856 char const *file_name = NULL;
6857 __kmp_str_split(tail, '-', &flag_addr_str, &tail);
6858 __kmp_str_split(tail, '-', &flag_val_str, &tail);
6859 file_name = tail;
6860 if (tail != NULL) {
6861 unsigned long *flag_addr = 0;
6862 unsigned long flag_val = 0;
6863 KMP_SSCANF(flag_addr_str, "%p", RCAST(void **, &flag_addr));
6864 KMP_SSCANF(flag_val_str, "%lx", &flag_val);
6865 if (flag_addr != 0 && flag_val != 0 && strcmp(file_name, "") != 0) {
6866 // First, check whether environment-encoded address is mapped into
6867 // addr space.
6868 // If so, dereference it to see if it still has the right value.
6869 if (__kmp_is_address_mapped(flag_addr) && *flag_addr == flag_val) {
6870 neighbor = 1;
6871 } else {
6872 // If not, then we know the other copy of the library is no longer
6873 // running.
6874 neighbor = 2;
6875 }
6876 }
6877 }
6878 switch (neighbor) {
6879 case 0: // Cannot parse environment variable -- neighbor status unknown.
6880 // Assume it is the incompatible format of future version of the
6881 // library. Assume the other library is alive.
6882 // WARN( ... ); // TODO: Issue a warning.
6883 file_name = "unknown library";
6885 // Attention! Falling to the next case. That's intentional.
6886 case 1: { // Neighbor is alive.
6887 // Check it is allowed.
6888 char *duplicate_ok = __kmp_env_get("KMP_DUPLICATE_LIB_OK");
6889 if (!__kmp_str_match_true(duplicate_ok)) {
6890 // That's not allowed. Issue fatal error.
6891 __kmp_fatal(KMP_MSG(DuplicateLibrary, KMP_LIBRARY_FILE, file_name),
6892 KMP_HNT(DuplicateLibrary), __kmp_msg_null);
6893 }
6894 KMP_INTERNAL_FREE(duplicate_ok);
6896 done = 1; // Exit the loop.
6897 } break;
6898 case 2: { // Neighbor is dead.
6899
6900#if defined(KMP_USE_SHM)
6901 if (__kmp_shm_available) { // close shared memory.
6902 shm_unlink(shm_name); // this removes file in /dev/shm
6903 } else if (__kmp_tmp_available) {
6904 unlink(temp_reg_status_file_name); // this removes the temp file
6905 } else {
6906 // Clear the variable and try to register library again.
6908 }
6909#else
6910 // Clear the variable and try to register library again.
6912#endif
6913 } break;
6914 default: {
6916 } break;
6917 }
6918 }
6919 KMP_INTERNAL_FREE((void *)value);
6920#if defined(KMP_USE_SHM)
6921 if (shm_name)
6922 KMP_INTERNAL_FREE((void *)shm_name);
6923#endif
6924 } // while
6925 KMP_INTERNAL_FREE((void *)name);
6926
6927} // func __kmp_register_library_startup
6928
6930
6931 // Claim the unregistration. Teardown can be entered concurrently from
6932 // library shutdown, __kmp_abort_process() and the signal handler, none of
6933 // which share a lock, and the library may never have registered itself at
6934 // all (e.g. a fatal error raised before registration). Atomically take
6935 // ownership of __kmp_registration_str so exactly one caller runs the
6936 // teardown below; the others return without touching the freed string.
6937 char *reg_str = __kmp_registration_str;
6938 if (reg_str == NULL ||
6940 return;
6942
6943 char *name = __kmp_reg_status_name();
6944 char *value = NULL;
6945
6946#if defined(KMP_USE_SHM)
6947 char *shm_name = nullptr;
6948 int fd1;
6949 if (__kmp_shm_available) {
6950 shm_name = __kmp_str_format("/%s", name);
6951 fd1 = shm_open(shm_name, O_RDONLY, 0600);
6952 if (fd1 != -1) { // File opened successfully
6953 char *data1 = (char *)mmap(0, SHM_SIZE, PROT_READ, MAP_SHARED, fd1, 0);
6954 if (data1 != MAP_FAILED) {
6955 value = __kmp_str_format("%s", data1); // read value from SHM
6956 munmap(data1, SHM_SIZE);
6957 }
6958 close(fd1);
6959 }
6960 } else if (__kmp_tmp_available) { // try /tmp
6961 fd1 = open(temp_reg_status_file_name, O_RDONLY);
6962 if (fd1 != -1) { // File opened successfully
6963 char *data1 = (char *)mmap(0, SHM_SIZE, PROT_READ, MAP_SHARED, fd1, 0);
6964 if (data1 != MAP_FAILED) {
6965 value = __kmp_str_format("%s", data1); // read value from /tmp
6966 munmap(data1, SHM_SIZE);
6967 }
6968 close(fd1);
6969 }
6970 } else { // fall back to envirable
6972 }
6973#else
6975#endif
6976
6977 if (value != NULL && strcmp(value, reg_str) == 0) {
6978// Ok, this is our variable. Delete it.
6979#if defined(KMP_USE_SHM)
6980 if (__kmp_shm_available) {
6981 shm_unlink(shm_name); // this removes file in /dev/shm
6982 } else if (__kmp_tmp_available) {
6983 unlink(temp_reg_status_file_name); // this removes the temp file
6984 } else {
6986 }
6987#else
6989#endif
6990 }
6991
6992#if defined(KMP_USE_SHM)
6993 if (shm_name)
6994 KMP_INTERNAL_FREE(shm_name);
6995 if (temp_reg_status_file_name)
6996 KMP_INTERNAL_FREE(temp_reg_status_file_name);
6997#endif
6998
6999 KMP_INTERNAL_FREE(reg_str);
7002
7003} // __kmp_unregister_library
7004
7005// End of Library registration stuff.
7006// -----------------------------------------------------------------------------
7007
7008#if KMP_MIC_SUPPORTED
7009
7010static void __kmp_check_mic_type() {
7011 kmp_cpuid_t cpuid_state = {0};
7012 kmp_cpuid_t *cs_p = &cpuid_state;
7013 __kmp_x86_cpuid(1, 0, cs_p);
7014 // We don't support mic1 at the moment
7015 if ((cs_p->eax & 0xff0) == 0xB10) {
7016 __kmp_mic_type = mic2;
7017 } else if ((cs_p->eax & 0xf0ff0) == 0x50670) {
7018 __kmp_mic_type = mic3;
7019 } else {
7020 __kmp_mic_type = non_mic;
7021 }
7022}
7023
7024#endif /* KMP_MIC_SUPPORTED */
7025
7026#if KMP_HAVE_UMWAIT
7027static void __kmp_user_level_mwait_init() {
7028 struct kmp_cpuid buf;
7029 __kmp_x86_cpuid(7, 0, &buf);
7030 __kmp_waitpkg_enabled = ((buf.ecx >> 5) & 1);
7031 __kmp_umwait_enabled = __kmp_waitpkg_enabled && __kmp_user_level_mwait;
7032 __kmp_tpause_enabled = __kmp_waitpkg_enabled && (__kmp_tpause_state > 0);
7033 KF_TRACE(30, ("__kmp_user_level_mwait_init: __kmp_umwait_enabled = %d\n",
7034 __kmp_umwait_enabled));
7035}
7036#elif KMP_HAVE_MWAIT
7037#ifndef AT_INTELPHIUSERMWAIT
7038// Spurious, non-existent value that should always fail to return anything.
7039// Will be replaced with the correct value when we know that.
7040#define AT_INTELPHIUSERMWAIT 10000
7041#endif
7042// getauxval() function is available in RHEL7 and SLES12. If a system with an
7043// earlier OS is used to build the RTL, we'll use the following internal
7044// function when the entry is not found.
7045unsigned long getauxval(unsigned long) KMP_WEAK_ATTRIBUTE_EXTERNAL;
7046unsigned long getauxval(unsigned long) { return 0; }
7047
7048static void __kmp_user_level_mwait_init() {
7049 // When getauxval() and correct value of AT_INTELPHIUSERMWAIT are available
7050 // use them to find if the user-level mwait is enabled. Otherwise, forcibly
7051 // set __kmp_mwait_enabled=TRUE on Intel MIC if the environment variable
7052 // KMP_USER_LEVEL_MWAIT was set to TRUE.
7053 if (__kmp_mic_type == mic3) {
7054 unsigned long res = getauxval(AT_INTELPHIUSERMWAIT);
7055 if ((res & 0x1) || __kmp_user_level_mwait) {
7056 __kmp_mwait_enabled = TRUE;
7057 if (__kmp_user_level_mwait) {
7058 KMP_INFORM(EnvMwaitWarn);
7059 }
7060 } else {
7061 __kmp_mwait_enabled = FALSE;
7062 }
7063 }
7064 KF_TRACE(30, ("__kmp_user_level_mwait_init: __kmp_mic_type = %d, "
7065 "__kmp_mwait_enabled = %d\n",
7066 __kmp_mic_type, __kmp_mwait_enabled));
7067}
7068#endif /* KMP_HAVE_UMWAIT */
7069
7071 int i, gtid;
7072 size_t size;
7073
7074 KA_TRACE(10, ("__kmp_do_serial_initialize: enter\n"));
7075
7076 KMP_DEBUG_ASSERT(sizeof(kmp_int32) == 4);
7077 KMP_DEBUG_ASSERT(sizeof(kmp_uint32) == 4);
7078 KMP_DEBUG_ASSERT(sizeof(kmp_int64) == 8);
7079 KMP_DEBUG_ASSERT(sizeof(kmp_uint64) == 8);
7080 KMP_DEBUG_ASSERT(sizeof(kmp_intptr_t) == sizeof(void *));
7081
7082#if OMPT_SUPPORT
7083 ompt_pre_init();
7084#endif
7085#if OMPD_SUPPORT
7086 __kmp_env_dump();
7087 ompd_init();
7088#endif
7089
7091
7092#if ENABLE_LIBOMPTARGET
7093 /* Initialize functions from libomptarget */
7094 __kmp_init_omptarget();
7095#endif
7096
7097 /* Initialize internal memory allocator */
7099
7100 /* Register the library startup via an environment variable or via mapped
7101 shared memory file and check to see whether another copy of the library is
7102 already registered. Since forked child process is often terminated, we
7103 postpone the registration till middle initialization in the child */
7106
7107 /* TODO reinitialization of library */
7108 if (TCR_4(__kmp_global.g.g_done)) {
7109 KA_TRACE(10, ("__kmp_do_serial_initialize: reinitialization of library\n"));
7110 }
7111
7112 __kmp_global.g.g_abort = 0;
7113 TCW_SYNC_4(__kmp_global.g.g_done, FALSE);
7114
7115/* initialize the locks */
7116#if KMP_USE_ADAPTIVE_LOCKS
7117#if KMP_DEBUG_ADAPTIVE_LOCKS
7118 __kmp_init_speculative_stats();
7119#endif
7120#endif
7121#if KMP_STATS_ENABLED
7123#endif
7140#if KMP_USE_MONITOR
7141 __kmp_init_bootstrap_lock(&__kmp_monitor_lock);
7142#endif
7144
7145 /* conduct initialization and initial setup of configuration */
7146
7148
7149#if KMP_MIC_SUPPORTED
7150 __kmp_check_mic_type();
7151#endif
7152#if ENABLE_LIBOMPTARGET
7153 __kmp_target_init();
7154#endif /* ENABLE_LIBOMPTARGET */
7155
7156// Some global variable initialization moved here from kmp_env_initialize()
7157#ifdef KMP_DEBUG
7158 kmp_diag = 0;
7159#endif
7161
7162 // From __kmp_init_dflt_team_nth()
7163 /* assume the entire machine will be used */
7167 }
7170 }
7173 __kmp_teams_max_nth = __kmp_xproc; // set a "reasonable" default
7176 }
7177
7178 // Three vars below moved here from __kmp_env_initialize() "KMP_BLOCKTIME"
7179 // part
7181#if KMP_USE_MONITOR
7182 __kmp_monitor_wakeups =
7183 KMP_WAKEUPS_FROM_BLOCKTIME(__kmp_dflt_blocktime, __kmp_monitor_wakeups);
7184 __kmp_bt_intervals =
7185 KMP_INTERVALS_FROM_BLOCKTIME(__kmp_dflt_blocktime, __kmp_monitor_wakeups);
7186#endif
7187 // From "KMP_LIBRARY" part of __kmp_env_initialize()
7189 // From KMP_SCHEDULE initialization
7191// AC: do not use analytical here, because it is non-monotonous
7192//__kmp_guided = kmp_sch_guided_iterative_chunked;
7193//__kmp_auto = kmp_sch_guided_analytical_chunked; // AC: it is the default, no
7194// need to repeat assignment
7195// Barrier initialization. Moved here from __kmp_env_initialize() Barrier branch
7196// bit control and barrier method control parts
7197#if KMP_FAST_REDUCTION_BARRIER
7198#define kmp_reduction_barrier_gather_bb ((int)1)
7199#define kmp_reduction_barrier_release_bb ((int)1)
7200#define kmp_reduction_barrier_gather_pat __kmp_barrier_gather_pat_dflt
7201#define kmp_reduction_barrier_release_pat __kmp_barrier_release_pat_dflt
7202#endif // KMP_FAST_REDUCTION_BARRIER
7203 for (i = bs_plain_barrier; i < bs_last_barrier; i++) {
7208#if KMP_FAST_REDUCTION_BARRIER
7209 if (i == bs_reduction_barrier) { // tested and confirmed on ALTIX only (
7210 // lin_64 ): hyper,1
7211 __kmp_barrier_gather_branch_bits[i] = kmp_reduction_barrier_gather_bb;
7212 __kmp_barrier_release_branch_bits[i] = kmp_reduction_barrier_release_bb;
7213 __kmp_barrier_gather_pattern[i] = kmp_reduction_barrier_gather_pat;
7214 __kmp_barrier_release_pattern[i] = kmp_reduction_barrier_release_pat;
7215 }
7216#endif // KMP_FAST_REDUCTION_BARRIER
7217 }
7218#if KMP_FAST_REDUCTION_BARRIER
7219#undef kmp_reduction_barrier_release_pat
7220#undef kmp_reduction_barrier_gather_pat
7221#undef kmp_reduction_barrier_release_bb
7222#undef kmp_reduction_barrier_gather_bb
7223#endif // KMP_FAST_REDUCTION_BARRIER
7224#if KMP_MIC_SUPPORTED
7225 if (__kmp_mic_type == mic2) { // KNC
7226 // AC: plane=3,2, forkjoin=2,1 are optimal for 240 threads on KNC
7229 1; // forkjoin release
7232 }
7233#if KMP_FAST_REDUCTION_BARRIER
7234 if (__kmp_mic_type == mic2) { // KNC
7237 }
7238#endif // KMP_FAST_REDUCTION_BARRIER
7239#endif // KMP_MIC_SUPPORTED
7240
7241// From KMP_CHECKS initialization
7242#ifdef KMP_DEBUG
7243 __kmp_env_checks = TRUE; /* development versions have the extra checks */
7244#else
7245 __kmp_env_checks = FALSE; /* port versions do not have the extra checks */
7246#endif
7247
7248 // From "KMP_FOREIGN_THREADS_THREADPRIVATE" initialization
7250
7251 __kmp_global.g.g_dynamic = FALSE;
7252 __kmp_global.g.g_dynamic_mode = dynamic_default;
7253
7255
7257
7258#if KMP_HAVE_MWAIT || KMP_HAVE_UMWAIT
7259 __kmp_user_level_mwait_init();
7260#endif
7261// Print all messages in message catalog for testing purposes.
7262#ifdef KMP_DEBUG
7263 char const *val = __kmp_env_get("KMP_DUMP_CATALOG");
7265 kmp_str_buf_t buffer;
7266 __kmp_str_buf_init(&buffer);
7267 __kmp_i18n_dump_catalog(&buffer);
7268 __kmp_printf("%s", buffer.str);
7269 __kmp_str_buf_free(&buffer);
7270 }
7272#endif
7273
7276 // Moved here from __kmp_env_initialize() "KMP_ALL_THREADPRIVATE" part
7279
7280 // If the library is shut down properly, both pools must be NULL. Just in
7281 // case, set them to NULL -- some memory may leak, but subsequent code will
7282 // work even if pools are not freed.
7286 __kmp_thread_pool = NULL;
7288 __kmp_team_pool = NULL;
7289
7290 /* Allocate all of the variable sized records */
7291 /* NOTE: __kmp_threads_capacity entries are allocated, but the arrays are
7292 * expandable */
7293 /* Since allocation is cache-aligned, just add extra padding at the end */
7294 size =
7295 (sizeof(kmp_info_t *) + sizeof(kmp_root_t *)) * __kmp_threads_capacity +
7296 CACHE_LINE;
7298 __kmp_root = (kmp_root_t **)((char *)__kmp_threads +
7300
7301 /* init thread counts */
7303 0); // Asserts fail if the library is reinitializing and
7304 KMP_DEBUG_ASSERT(__kmp_nth == 0); // something was wrong in termination.
7305 __kmp_all_nth = 0;
7306 __kmp_nth = 0;
7307
7308 /* setup the uber master thread and hierarchy */
7309 gtid = __kmp_register_root(TRUE);
7310 KA_TRACE(10, ("__kmp_do_serial_initialize T#%d\n", gtid));
7313
7314 KMP_MB(); /* Flush all pending memory write invalidates. */
7315
7317
7318#if KMP_OS_UNIX
7319 /* invoke the child fork handler */
7321#endif
7322
7323#if !KMP_DYNAMIC_LIB || \
7324 ((KMP_COMPILER_ICC || KMP_COMPILER_ICX) && KMP_OS_DARWIN)
7325 {
7326 /* Invoke the exit handler when the program finishes, only for static
7327 library and macOS* dynamic. For other dynamic libraries, we already
7328 have _fini and DllMain. */
7329 int rc = atexit(__kmp_internal_end_atexit);
7330 if (rc != 0) {
7331 __kmp_fatal(KMP_MSG(FunctionError, "atexit()"), KMP_ERR(rc),
7333 }
7334 }
7335#endif
7336
7337#if KMP_HANDLE_SIGNALS
7338#if KMP_OS_UNIX
7339 /* NOTE: make sure that this is called before the user installs their own
7340 signal handlers so that the user handlers are called first. this way they
7341 can return false, not call our handler, avoid terminating the library, and
7342 continue execution where they left off. */
7343 __kmp_install_signals(FALSE);
7344#endif /* KMP_OS_UNIX */
7345#if KMP_OS_WINDOWS
7346 __kmp_install_signals(TRUE);
7347#endif /* KMP_OS_WINDOWS */
7348#endif
7349
7350 /* we have finished the serial initialization */
7352
7354
7355 if (__kmp_version) {
7357 }
7358
7359 if (__kmp_settings) {
7361 }
7362
7365 }
7366
7367#if OMPT_SUPPORT
7369#endif
7370
7371 KMP_MB();
7372
7373 KA_TRACE(10, ("__kmp_do_serial_initialize: exit\n"));
7374}
7375
7388
7390 int i, j;
7391 int prev_dflt_team_nth;
7392
7393 if (!__kmp_init_serial) {
7395 }
7396
7397 KA_TRACE(10, ("__kmp_middle_initialize: enter\n"));
7398
7400 // We are in a forked child process. The registration was skipped during
7401 // serial initialization in __kmp_atfork_child handler. Do it here.
7403 }
7404
7405 // Save the previous value for the __kmp_dflt_team_nth so that
7406 // we can avoid some reinitialization if it hasn't changed.
7407 prev_dflt_team_nth = __kmp_dflt_team_nth;
7408
7409#if KMP_AFFINITY_SUPPORTED
7410 // __kmp_affinity_initialize() will try to set __kmp_ncores to the
7411 // number of cores on the machine.
7412 __kmp_affinity_initialize(__kmp_affinity);
7413
7414#endif /* KMP_AFFINITY_SUPPORTED */
7415
7417 if (__kmp_avail_proc == 0) {
7419 }
7420
7421 // If there were empty places in num_threads list (OMP_NUM_THREADS=,,2,3),
7422 // correct them now
7423 j = 0;
7424 while ((j < __kmp_nested_nth.used) && !__kmp_nested_nth.nth[j]) {
7427 j++;
7428 }
7429
7430 if (__kmp_dflt_team_nth == 0) {
7431#ifdef KMP_DFLT_NTH_CORES
7432 // Default #threads = #cores
7434 KA_TRACE(20, ("__kmp_middle_initialize: setting __kmp_dflt_team_nth = "
7435 "__kmp_ncores (%d)\n",
7437#else
7438 // Default #threads = #available OS procs
7440 KA_TRACE(20, ("__kmp_middle_initialize: setting __kmp_dflt_team_nth = "
7441 "__kmp_avail_proc(%d)\n",
7443#endif /* KMP_DFLT_NTH_CORES */
7444 }
7445
7448 }
7451 }
7452
7453 if (__kmp_nesting_mode > 0)
7455
7456 // There's no harm in continuing if the following check fails,
7457 // but it indicates an error in the previous logic.
7459
7460 if (__kmp_dflt_team_nth != prev_dflt_team_nth) {
7461 // Run through the __kmp_threads array and set the num threads icv for each
7462 // root thread that is currently registered with the RTL (which has not
7463 // already explicitly set its nthreads-var with a call to
7464 // omp_set_num_threads()).
7465 for (i = 0; i < __kmp_threads_capacity; i++) {
7466 kmp_info_t *thread = __kmp_threads[i];
7467 if (thread == NULL)
7468 continue;
7469 if (thread->th.th_current_task->td_icvs.nproc != 0)
7470 continue;
7471
7473 }
7474 }
7475 KA_TRACE(
7476 20,
7477 ("__kmp_middle_initialize: final value for __kmp_dflt_team_nth = %d\n",
7479
7480#ifdef KMP_ADJUST_BLOCKTIME
7481 /* Adjust blocktime to zero if necessary now that __kmp_avail_proc is set */
7482 if (!__kmp_env_blocktime && (__kmp_avail_proc > 0)) {
7485 __kmp_zero_bt = TRUE;
7486 }
7487 }
7488#endif /* KMP_ADJUST_BLOCKTIME */
7489
7490 /* we have finished middle initialization */
7492
7493 KA_TRACE(10, ("__kmp_do_middle_initialize: exit\n"));
7494}
7495
7508
7510 int gtid = __kmp_entry_gtid(); // this might be a new root
7511
7512 /* synchronize parallel initialization (for sibling) */
7514 return;
7518 return;
7519 }
7520
7521 /* TODO reinitialization after we have already shut down */
7522 if (TCR_4(__kmp_global.g.g_done)) {
7523 KA_TRACE(
7524 10,
7525 ("__kmp_parallel_initialize: attempt to init while shutting down\n"));
7527 }
7528
7529 /* jc: The lock __kmp_initz_lock is already held, so calling
7530 __kmp_serial_initialize would cause a deadlock. So we call
7531 __kmp_do_serial_initialize directly. */
7532 if (!__kmp_init_middle) {
7534 }
7537
7538 /* begin initialization */
7539 KA_TRACE(10, ("__kmp_parallel_initialize: enter\n"));
7541
7542#if KMP_ARCH_X86 || KMP_ARCH_X86_64
7543 // Save the FP control regs.
7544 // Worker threads will set theirs to these values at thread startup.
7545 __kmp_store_x87_fpu_control_word(&__kmp_init_x87_fpu_control_word);
7546 __kmp_store_mxcsr(&__kmp_init_mxcsr);
7547 __kmp_init_mxcsr &= KMP_X86_MXCSR_MASK;
7548#endif /* KMP_ARCH_X86 || KMP_ARCH_X86_64 */
7549
7550#if KMP_OS_UNIX
7551#if KMP_HANDLE_SIGNALS
7552 /* must be after __kmp_serial_initialize */
7553 __kmp_install_signals(TRUE);
7554#endif
7555#endif
7556
7558
7559#if defined(USE_LOAD_BALANCE)
7560 if (__kmp_global.g.g_dynamic_mode == dynamic_default) {
7561 __kmp_global.g.g_dynamic_mode = dynamic_load_balance;
7562 }
7563#else
7564 if (__kmp_global.g.g_dynamic_mode == dynamic_default) {
7565 __kmp_global.g.g_dynamic_mode = dynamic_thread_limit;
7566 }
7567#endif
7568
7569 if (__kmp_version) {
7571 }
7572
7573 /* we have finished parallel initialization */
7575
7576 KMP_MB();
7577 KA_TRACE(10, ("__kmp_parallel_initialize: exit\n"));
7578
7580}
7581
7584 return;
7585
7586 // __kmp_parallel_initialize is required before we initialize hidden helper
7589
7590 // Double check. Note that this double check should not be placed before
7591 // __kmp_parallel_initialize as it will cause dead lock.
7595 return;
7596 }
7597
7598#if KMP_AFFINITY_SUPPORTED
7599 // Initialize hidden helper affinity settings.
7600 // The above __kmp_parallel_initialize() will initialize
7601 // regular affinity (and topology) if not already done.
7602 if (!__kmp_hh_affinity.flags.initialized)
7603 __kmp_affinity_initialize(__kmp_hh_affinity);
7604#endif
7605
7606 // Set the count of hidden helper tasks to be executed to zero
7608
7609 // Set the global variable indicating that we're initializing hidden helper
7610 // team/threads
7612
7613 // Platform independent initialization
7615
7616 // Wait here for the finish of initialization of hidden helper teams
7618
7619 // We have finished hidden helper initialization
7621
7623}
7624
7625/* ------------------------------------------------------------------------ */
7626
7627void __kmp_run_before_invoked_task(int gtid, int tid, kmp_info_t *this_thr,
7628 kmp_team_t *team) {
7629 kmp_disp_t *dispatch;
7630
7631 KMP_MB();
7632
7633 /* none of the threads have encountered any constructs, yet. */
7634 this_thr->th.th_local.this_construct = 0;
7635#if KMP_CACHE_MANAGE
7636 KMP_CACHE_PREFETCH(&this_thr->th.th_bar[bs_forkjoin_barrier].bb.b_arrived);
7637#endif /* KMP_CACHE_MANAGE */
7638 dispatch = (kmp_disp_t *)TCR_PTR(this_thr->th.th_dispatch);
7639 KMP_DEBUG_ASSERT(dispatch);
7640 KMP_DEBUG_ASSERT(team->t.t_dispatch);
7641 // KMP_DEBUG_ASSERT( this_thr->th.th_dispatch == &team->t.t_dispatch[
7642 // this_thr->th.th_info.ds.ds_tid ] );
7643
7644 dispatch->th_disp_index = 0; /* reset the dispatch buffer counter */
7645 dispatch->th_doacross_buf_idx = 0; // reset doacross dispatch buffer counter
7647 __kmp_push_parallel(gtid, team->t.t_ident);
7648
7649 KMP_MB(); /* Flush all pending memory write invalidates. */
7650}
7651
7652void __kmp_run_after_invoked_task(int gtid, int tid, kmp_info_t *this_thr,
7653 kmp_team_t *team) {
7655 __kmp_pop_parallel(gtid, team->t.t_ident);
7656
7658}
7659
7661 int rc;
7662 int tid = __kmp_tid_from_gtid(gtid);
7663 kmp_info_t *this_thr = __kmp_threads[gtid];
7664 kmp_team_t *team = this_thr->th.th_team;
7665
7666 __kmp_run_before_invoked_task(gtid, tid, this_thr, team);
7667#if USE_ITT_BUILD
7668 if (__itt_stack_caller_create_ptr) {
7669 // inform ittnotify about entering user's code
7670 if (team->t.t_stack_id != NULL) {
7671 __kmp_itt_stack_callee_enter((__itt_caller)team->t.t_stack_id);
7672 } else {
7673 KMP_DEBUG_ASSERT(team->t.t_parent->t.t_stack_id != NULL);
7674 __kmp_itt_stack_callee_enter(
7675 (__itt_caller)team->t.t_parent->t.t_stack_id);
7676 }
7677 }
7678#endif /* USE_ITT_BUILD */
7679#if INCLUDE_SSC_MARKS
7680 SSC_MARK_INVOKING();
7681#endif
7682
7683#if OMPT_SUPPORT
7684 void *dummy;
7685 void **exit_frame_p;
7686 ompt_data_t *my_task_data;
7687 ompt_data_t *my_parallel_data;
7688 int ompt_team_size;
7689
7690 if (ompt_enabled.enabled) {
7691 exit_frame_p = &(team->t.t_implicit_task_taskdata[tid]
7692 .ompt_task_info.frame.exit_frame.ptr);
7693 } else {
7694 exit_frame_p = &dummy;
7695 }
7696
7697 my_task_data =
7698 &(team->t.t_implicit_task_taskdata[tid].ompt_task_info.task_data);
7699 my_parallel_data = &(team->t.ompt_team_info.parallel_data);
7700 if (ompt_enabled.ompt_callback_implicit_task) {
7701 ompt_team_size = team->t.t_nproc;
7702 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
7703 ompt_scope_begin, my_parallel_data, my_task_data, ompt_team_size,
7704 __kmp_tid_from_gtid(gtid), ompt_task_implicit);
7705 OMPT_CUR_TASK_INFO(this_thr)->thread_num = __kmp_tid_from_gtid(gtid);
7706 }
7707#endif
7708
7709#if KMP_STATS_ENABLED
7710 stats_state_e previous_state = KMP_GET_THREAD_STATE();
7711 if (previous_state == stats_state_e::TEAMS_REGION) {
7712 KMP_PUSH_PARTITIONED_TIMER(OMP_teams);
7713 } else {
7714 KMP_PUSH_PARTITIONED_TIMER(OMP_parallel);
7715 }
7716 KMP_SET_THREAD_STATE(IMPLICIT_TASK);
7717#endif
7718
7719 rc = __kmp_invoke_microtask((microtask_t)TCR_SYNC_PTR(team->t.t_pkfn), gtid,
7720 tid, (int)team->t.t_argc, (void **)team->t.t_argv
7721#if OMPT_SUPPORT
7722 ,
7723 exit_frame_p
7724#endif
7725 );
7726#if OMPT_SUPPORT
7727 *exit_frame_p = NULL;
7728 this_thr->th.ompt_thread_info.parallel_flags = ompt_parallel_team;
7729#endif
7730
7731#if KMP_STATS_ENABLED
7732 if (previous_state == stats_state_e::TEAMS_REGION) {
7733 KMP_SET_THREAD_STATE(previous_state);
7734 }
7736#endif
7737
7738#if USE_ITT_BUILD
7739 if (__itt_stack_caller_create_ptr) {
7740 // inform ittnotify about leaving user's code
7741 if (team->t.t_stack_id != NULL) {
7742 __kmp_itt_stack_callee_leave((__itt_caller)team->t.t_stack_id);
7743 } else {
7744 KMP_DEBUG_ASSERT(team->t.t_parent->t.t_stack_id != NULL);
7745 __kmp_itt_stack_callee_leave(
7746 (__itt_caller)team->t.t_parent->t.t_stack_id);
7747 }
7748 }
7749#endif /* USE_ITT_BUILD */
7750 __kmp_run_after_invoked_task(gtid, tid, this_thr, team);
7751
7752 return rc;
7753}
7754
7755void __kmp_teams_master(int gtid) {
7756 // This routine is called by all primary threads in teams construct
7757 kmp_info_t *thr = __kmp_threads[gtid];
7758 kmp_team_t *team = thr->th.th_team;
7759 ident_t *loc = team->t.t_ident;
7760 thr->th.th_set_nproc = thr->th.th_teams_size.nth;
7761 KMP_DEBUG_ASSERT(thr->th.th_teams_microtask);
7762 KMP_DEBUG_ASSERT(thr->th.th_set_nproc);
7763 KA_TRACE(20, ("__kmp_teams_master: T#%d, Tid %d, microtask %p\n", gtid,
7764 __kmp_tid_from_gtid(gtid), thr->th.th_teams_microtask));
7765
7766 // This thread is a new CG root. Set up the proper variables.
7768 tmp->cg_root = thr; // Make thr the CG root
7769 // Init to thread limit stored when league primary threads were forked
7770 tmp->cg_thread_limit = thr->th.th_current_task->td_icvs.thread_limit;
7771 tmp->cg_nthreads = 1; // Init counter to one active thread, this one
7772 KA_TRACE(100, ("__kmp_teams_master: Thread %p created node %p and init"
7773 " cg_nthreads to 1\n",
7774 thr, tmp));
7775 tmp->up = thr->th.th_cg_roots;
7776 thr->th.th_cg_roots = tmp;
7777
7778// Launch league of teams now, but not let workers execute
7779// (they hang on fork barrier until next parallel)
7780#if INCLUDE_SSC_MARKS
7781 SSC_MARK_FORKING();
7782#endif
7783 __kmp_fork_call(loc, gtid, fork_context_intel, team->t.t_argc,
7784 (microtask_t)thr->th.th_teams_microtask, // "wrapped" task
7786#if INCLUDE_SSC_MARKS
7787 SSC_MARK_JOINING();
7788#endif
7789 // If the team size was reduced from the limit, set it to the new size
7790 if (thr->th.th_team_nproc < thr->th.th_teams_size.nth)
7791 thr->th.th_teams_size.nth = thr->th.th_team_nproc;
7792 // AC: last parameter "1" eliminates join barrier which won't work because
7793 // worker threads are in a fork barrier waiting for more parallel regions
7794 __kmp_join_call(loc, gtid
7795#if OMPT_SUPPORT
7796 ,
7798#endif
7799 ,
7800 1);
7801}
7802
7804 kmp_info_t *this_thr = __kmp_threads[gtid];
7805 kmp_team_t *team = this_thr->th.th_team;
7806#if KMP_DEBUG
7807 if (!__kmp_threads[gtid]->th.th_team->t.t_serialized)
7808 KMP_DEBUG_ASSERT((void *)__kmp_threads[gtid]->th.th_team->t.t_pkfn ==
7809 (void *)__kmp_teams_master);
7810#endif
7811 __kmp_run_before_invoked_task(gtid, 0, this_thr, team);
7812#if OMPT_SUPPORT
7813 int tid = __kmp_tid_from_gtid(gtid);
7814 ompt_data_t *task_data =
7815 &team->t.t_implicit_task_taskdata[tid].ompt_task_info.task_data;
7816 ompt_data_t *parallel_data = &team->t.ompt_team_info.parallel_data;
7817 if (ompt_enabled.ompt_callback_implicit_task) {
7818 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
7819 ompt_scope_begin, parallel_data, task_data, team->t.t_nproc, tid,
7820 ompt_task_initial);
7821 OMPT_CUR_TASK_INFO(this_thr)->thread_num = tid;
7822 }
7823#endif
7824 __kmp_teams_master(gtid);
7825#if OMPT_SUPPORT
7826 this_thr->th.ompt_thread_info.parallel_flags = ompt_parallel_league;
7827#endif
7828 __kmp_run_after_invoked_task(gtid, 0, this_thr, team);
7829 return 1;
7830}
7831
7832/* this sets the requested number of threads for the next parallel region
7833 encountered by this team. since this should be enclosed in the forkjoin
7834 critical section it should avoid race conditions with asymmetrical nested
7835 parallelism */
7836void __kmp_push_num_threads(ident_t *id, int gtid, int num_threads) {
7837 kmp_info_t *thr = __kmp_threads[gtid];
7838
7839 if (num_threads > 0)
7840 thr->th.th_set_nproc = num_threads;
7841}
7842
7843void __kmp_push_num_threads_list(ident_t *id, int gtid, kmp_uint32 list_length,
7844 int *num_threads_list) {
7845 kmp_info_t *thr = __kmp_threads[gtid];
7846
7847 KMP_DEBUG_ASSERT(list_length > 1);
7848
7849 if (num_threads_list[0] > 0)
7850 thr->th.th_set_nproc = num_threads_list[0];
7851 thr->th.th_set_nested_nth =
7852 (int *)KMP_INTERNAL_MALLOC(list_length * sizeof(int));
7853 for (kmp_uint32 i = 0; i < list_length; ++i)
7854 thr->th.th_set_nested_nth[i] = num_threads_list[i];
7855 thr->th.th_set_nested_nth_sz = list_length;
7856}
7857
7859 const char *msg) {
7860 kmp_info_t *thr = __kmp_threads[gtid];
7861 thr->th.th_nt_strict = true;
7862 thr->th.th_nt_loc = loc;
7863 // if sev is unset make fatal
7864 if (sev == severity_warning)
7865 thr->th.th_nt_sev = sev;
7866 else
7867 thr->th.th_nt_sev = severity_fatal;
7868 // if msg is unset, use an appropriate message
7869 if (msg)
7870 thr->th.th_nt_msg = msg;
7871 else
7872 thr->th.th_nt_msg = "Cannot form team with number of threads specified by "
7873 "strict num_threads clause.";
7874}
7875
7876static void __kmp_push_thread_limit(kmp_info_t *thr, int num_teams,
7877 int num_threads) {
7878 KMP_DEBUG_ASSERT(thr);
7879 // Remember the number of threads for inner parallel regions
7881 __kmp_middle_initialize(); // get internal globals calculated
7885
7886 if (num_threads == 0) {
7887 if (__kmp_teams_thread_limit > 0) {
7888 num_threads = __kmp_teams_thread_limit;
7889 } else {
7890 num_threads = __kmp_avail_proc / num_teams;
7891 }
7892 // adjust num_threads w/o warning as it is not user setting
7893 // num_threads = min(num_threads, nthreads-var, thread-limit-var)
7894 // no thread_limit clause specified - do not change thread-limit-var ICV
7895 if (num_threads > __kmp_dflt_team_nth) {
7896 num_threads = __kmp_dflt_team_nth; // honor nthreads-var ICV
7897 }
7898 if (num_threads > thr->th.th_current_task->td_icvs.thread_limit) {
7899 num_threads = thr->th.th_current_task->td_icvs.thread_limit;
7900 } // prevent team size to exceed thread-limit-var
7901 if (num_teams * num_threads > __kmp_teams_max_nth) {
7902 num_threads = __kmp_teams_max_nth / num_teams;
7903 }
7904 if (num_threads == 0) {
7905 num_threads = 1;
7906 }
7907 } else {
7908 if (num_threads < 0) {
7909 __kmp_msg(kmp_ms_warning, KMP_MSG(CantFormThrTeam, num_threads, 1),
7911 num_threads = 1;
7912 }
7913 // This thread will be the primary thread of the league primary threads
7914 // Store new thread limit; old limit is saved in th_cg_roots list
7915 thr->th.th_current_task->td_icvs.thread_limit = num_threads;
7916 // num_threads = min(num_threads, nthreads-var)
7917 if (num_threads > __kmp_dflt_team_nth) {
7918 num_threads = __kmp_dflt_team_nth; // honor nthreads-var ICV
7919 }
7920 if (num_teams * num_threads > __kmp_teams_max_nth) {
7921 int new_threads = __kmp_teams_max_nth / num_teams;
7922 if (new_threads == 0) {
7923 new_threads = 1;
7924 }
7925 if (new_threads != num_threads) {
7926 if (!__kmp_reserve_warn) { // user asked for too many threads
7927 __kmp_reserve_warn = 1; // conflicts with KMP_TEAMS_THREAD_LIMIT
7929 KMP_MSG(CantFormThrTeam, num_threads, new_threads),
7930 KMP_HNT(Unset_ALL_THREADS), __kmp_msg_null);
7931 }
7932 }
7933 num_threads = new_threads;
7934 }
7935 }
7936 thr->th.th_teams_size.nth = num_threads;
7937}
7938
7939/* this sets the requested number of teams for the teams region and/or
7940 the number of threads for the next parallel region encountered */
7941void __kmp_push_num_teams(ident_t *id, int gtid, int num_teams,
7942 int num_threads) {
7943 kmp_info_t *thr = __kmp_threads[gtid];
7944 if (num_teams < 0) {
7945 // OpenMP specification requires requested values to be positive,
7946 // but people can send us any value, so we'd better check
7947 __kmp_msg(kmp_ms_warning, KMP_MSG(NumTeamsNotPositive, num_teams, 1),
7949 num_teams = 1;
7950 }
7951 if (num_teams == 0) {
7952 if (__kmp_nteams > 0) {
7953 num_teams = __kmp_nteams;
7954 } else {
7955 num_teams = 1; // default number of teams is 1.
7956 }
7957 }
7958 if (num_teams > __kmp_teams_max_nth) { // if too many teams requested?
7959 if (!__kmp_reserve_warn) {
7962 KMP_MSG(CantFormThrTeam, num_teams, __kmp_teams_max_nth),
7963 KMP_HNT(Unset_ALL_THREADS), __kmp_msg_null);
7964 }
7965 num_teams = __kmp_teams_max_nth;
7966 }
7967 // Set number of teams (number of threads in the outer "parallel" of the
7968 // teams)
7969 thr->th.th_set_nproc = thr->th.th_teams_size.nteams = num_teams;
7970
7971 __kmp_push_thread_limit(thr, num_teams, num_threads);
7972}
7973
7974/* This sets the requested number of teams for the teams region and/or
7975 the number of threads for the next parallel region encountered */
7976void __kmp_push_num_teams_51(ident_t *id, int gtid, int num_teams_lb,
7977 int num_teams_ub, int num_threads) {
7978 kmp_info_t *thr = __kmp_threads[gtid];
7979 KMP_DEBUG_ASSERT(num_teams_lb >= 0 && num_teams_ub >= 0);
7980 KMP_DEBUG_ASSERT(num_teams_ub >= num_teams_lb);
7981 KMP_DEBUG_ASSERT(num_threads >= 0);
7982
7983 if (num_teams_lb > num_teams_ub) {
7984 __kmp_fatal(KMP_MSG(FailedToCreateTeam, num_teams_lb, num_teams_ub),
7986 }
7987
7988 int num_teams = 1; // defalt number of teams is 1.
7989
7990 if (num_teams_lb == 0 && num_teams_ub > 0)
7991 num_teams_lb = num_teams_ub;
7992
7993 if (num_teams_lb == 0 && num_teams_ub == 0) { // no num_teams clause
7994 num_teams = (__kmp_nteams > 0) ? __kmp_nteams : num_teams;
7995 if (num_teams > __kmp_teams_max_nth) {
7996 if (!__kmp_reserve_warn) {
7999 KMP_MSG(CantFormThrTeam, num_teams, __kmp_teams_max_nth),
8000 KMP_HNT(Unset_ALL_THREADS), __kmp_msg_null);
8001 }
8002 num_teams = __kmp_teams_max_nth;
8003 }
8004 } else if (num_teams_lb == num_teams_ub) { // requires exact number of teams
8005 num_teams = num_teams_ub;
8006 } else { // num_teams_lb <= num_teams <= num_teams_ub
8007 if (num_threads <= 0) {
8008 if (num_teams_ub > __kmp_teams_max_nth) {
8009 num_teams = num_teams_lb;
8010 } else {
8011 num_teams = num_teams_ub;
8012 }
8013 } else {
8014 num_teams = (num_threads > __kmp_teams_max_nth)
8015 ? num_teams
8016 : __kmp_teams_max_nth / num_threads;
8017 if (num_teams < num_teams_lb) {
8018 num_teams = num_teams_lb;
8019 } else if (num_teams > num_teams_ub) {
8020 num_teams = num_teams_ub;
8021 }
8022 }
8023 }
8024 // Set number of teams (number of threads in the outer "parallel" of the
8025 // teams)
8026 thr->th.th_set_nproc = thr->th.th_teams_size.nteams = num_teams;
8027
8028 __kmp_push_thread_limit(thr, num_teams, num_threads);
8029}
8030
8031// Set the proc_bind var to use in the following parallel region.
8032void __kmp_push_proc_bind(ident_t *id, int gtid, kmp_proc_bind_t proc_bind) {
8033 kmp_info_t *thr = __kmp_threads[gtid];
8034 thr->th.th_set_proc_bind = proc_bind;
8035}
8036
8037/* Launch the worker threads into the microtask. */
8038
8039void __kmp_internal_fork(ident_t *id, int gtid, kmp_team_t *team) {
8040 kmp_info_t *this_thr = __kmp_threads[gtid];
8041
8042#ifdef KMP_DEBUG
8043 int f;
8044#endif /* KMP_DEBUG */
8045
8046 KMP_DEBUG_ASSERT(team);
8047 KMP_DEBUG_ASSERT(this_thr->th.th_team == team);
8049 KMP_MB(); /* Flush all pending memory write invalidates. */
8050
8051 team->t.t_construct = 0; /* no single directives seen yet */
8052 team->t.t_ordered.dt.t_value =
8053 0; /* thread 0 enters the ordered section first */
8054
8055 /* Reset the identifiers on the dispatch buffer */
8056 KMP_DEBUG_ASSERT(team->t.t_disp_buffer);
8057 if (team->t.t_max_nproc > 1) {
8058 int i;
8059 for (i = 0; i < __kmp_dispatch_num_buffers; ++i) {
8060 team->t.t_disp_buffer[i].buffer_index = i;
8061 team->t.t_disp_buffer[i].doacross_buf_idx = i;
8062 }
8063 } else {
8064 team->t.t_disp_buffer[0].buffer_index = 0;
8065 team->t.t_disp_buffer[0].doacross_buf_idx = 0;
8066 }
8067
8068 KMP_MB(); /* Flush all pending memory write invalidates. */
8069 KMP_ASSERT(this_thr->th.th_team == team);
8070
8071#ifdef KMP_DEBUG
8072 for (f = 0; f < team->t.t_nproc; f++) {
8073 KMP_DEBUG_ASSERT(team->t.t_threads[f] &&
8074 team->t.t_threads[f]->th.th_team_nproc == team->t.t_nproc);
8075 }
8076#endif /* KMP_DEBUG */
8077
8078 /* release the worker threads so they may begin working */
8079 __kmp_fork_barrier(gtid, 0);
8080}
8081
8082void __kmp_internal_join(ident_t *id, int gtid, kmp_team_t *team) {
8083 kmp_info_t *this_thr = __kmp_threads[gtid];
8084
8085 KMP_DEBUG_ASSERT(team);
8086 KMP_DEBUG_ASSERT(this_thr->th.th_team == team);
8088 KMP_MB(); /* Flush all pending memory write invalidates. */
8089
8090 /* Join barrier after fork */
8091
8092#ifdef KMP_DEBUG
8093 if (__kmp_threads[gtid] &&
8094 __kmp_threads[gtid]->th.th_team_nproc != team->t.t_nproc) {
8095 __kmp_printf("GTID: %d, __kmp_threads[%d]=%p\n", gtid, gtid,
8096 __kmp_threads[gtid]);
8097 __kmp_printf("__kmp_threads[%d]->th.th_team_nproc=%d, TEAM: %p, "
8098 "team->t.t_nproc=%d\n",
8099 gtid, __kmp_threads[gtid]->th.th_team_nproc, team,
8100 team->t.t_nproc);
8102 }
8104 __kmp_threads[gtid]->th.th_team_nproc == team->t.t_nproc);
8105#endif /* KMP_DEBUG */
8106
8107 __kmp_join_barrier(gtid); /* wait for everyone */
8108#if OMPT_SUPPORT
8109 ompt_state_t ompt_state = this_thr->th.ompt_thread_info.state;
8110 if (ompt_enabled.enabled &&
8111 (ompt_state == ompt_state_wait_barrier_teams ||
8112 ompt_state == ompt_state_wait_barrier_implicit_parallel)) {
8113 int ds_tid = this_thr->th.th_info.ds.ds_tid;
8114 ompt_data_t *task_data = OMPT_CUR_TASK_DATA(this_thr);
8115 this_thr->th.ompt_thread_info.state = ompt_state_overhead;
8116#if OMPT_OPTIONAL
8117 void *codeptr = NULL;
8118 if (KMP_MASTER_TID(ds_tid) &&
8119 (ompt_callbacks.ompt_callback(ompt_callback_sync_region_wait) ||
8120 ompt_callbacks.ompt_callback(ompt_callback_sync_region)))
8121 codeptr = OMPT_CUR_TEAM_INFO(this_thr)->master_return_address;
8122
8123 ompt_sync_region_t sync_kind = ompt_sync_region_barrier_implicit_parallel;
8124 if (this_thr->th.ompt_thread_info.parallel_flags & ompt_parallel_league)
8125 sync_kind = ompt_sync_region_barrier_teams;
8126 if (ompt_enabled.ompt_callback_sync_region_wait) {
8127 ompt_callbacks.ompt_callback(ompt_callback_sync_region_wait)(
8128 sync_kind, ompt_scope_end, NULL, task_data, codeptr);
8129 }
8130 if (ompt_enabled.ompt_callback_sync_region) {
8131 ompt_callbacks.ompt_callback(ompt_callback_sync_region)(
8132 sync_kind, ompt_scope_end, NULL, task_data, codeptr);
8133 }
8134#endif
8135 if (!KMP_MASTER_TID(ds_tid) && ompt_enabled.ompt_callback_implicit_task) {
8136 ompt_callbacks.ompt_callback(ompt_callback_implicit_task)(
8137 ompt_scope_end, NULL, task_data, 0, ds_tid,
8138 ompt_task_implicit); // TODO: Can this be ompt_task_initial?
8139 }
8140 }
8141#endif
8142
8143 KMP_MB(); /* Flush all pending memory write invalidates. */
8144 KMP_ASSERT(this_thr->th.th_team == team);
8145}
8146
8147/* ------------------------------------------------------------------------ */
8148
8149#ifdef USE_LOAD_BALANCE
8150
8151// Return the worker threads actively spinning in the hot team, if we
8152// are at the outermost level of parallelism. Otherwise, return 0.
8153static int __kmp_active_hot_team_nproc(kmp_root_t *root) {
8154 int i;
8155 int retval;
8156 kmp_team_t *hot_team;
8157
8158 if (root->r.r_active) {
8159 return 0;
8160 }
8161 hot_team = root->r.r_hot_team;
8163 return hot_team->t.t_nproc - 1; // Don't count primary thread
8164 }
8165
8166 // Skip the primary thread - it is accounted for elsewhere.
8167 retval = 0;
8168 for (i = 1; i < hot_team->t.t_nproc; i++) {
8169 if (hot_team->t.t_threads[i]->th.th_active) {
8170 retval++;
8171 }
8172 }
8173 return retval;
8174}
8175
8176// Perform an automatic adjustment to the number of
8177// threads used by the next parallel region.
8178static int __kmp_load_balance_nproc(kmp_root_t *root, int set_nproc) {
8179 int retval;
8180 int pool_active;
8181 int hot_team_active;
8182 int team_curr_active;
8183 int system_active;
8184
8185 KB_TRACE(20, ("__kmp_load_balance_nproc: called root:%p set_nproc:%d\n", root,
8186 set_nproc));
8187 KMP_DEBUG_ASSERT(root);
8188 KMP_DEBUG_ASSERT(root->r.r_root_team->t.t_threads[0]
8189 ->th.th_current_task->td_icvs.dynamic == TRUE);
8190 KMP_DEBUG_ASSERT(set_nproc > 1);
8191
8192 if (set_nproc == 1) {
8193 KB_TRACE(20, ("__kmp_load_balance_nproc: serial execution.\n"));
8194 return 1;
8195 }
8196
8197 // Threads that are active in the thread pool, active in the hot team for this
8198 // particular root (if we are at the outer par level), and the currently
8199 // executing thread (to become the primary thread) are available to add to the
8200 // new team, but are currently contributing to the system load, and must be
8201 // accounted for.
8202 pool_active = __kmp_thread_pool_active_nth;
8203 hot_team_active = __kmp_active_hot_team_nproc(root);
8204 team_curr_active = pool_active + hot_team_active + 1;
8205
8206 // Check the system load.
8207 system_active = __kmp_get_load_balance(__kmp_avail_proc + team_curr_active);
8208 KB_TRACE(30, ("__kmp_load_balance_nproc: system active = %d pool active = %d "
8209 "hot team active = %d\n",
8210 system_active, pool_active, hot_team_active));
8211
8212 if (system_active < 0) {
8213 // There was an error reading the necessary info from /proc, so use the
8214 // thread limit algorithm instead. Once we set __kmp_global.g.g_dynamic_mode
8215 // = dynamic_thread_limit, we shouldn't wind up getting back here.
8216 __kmp_global.g.g_dynamic_mode = dynamic_thread_limit;
8217 KMP_WARNING(CantLoadBalUsing, "KMP_DYNAMIC_MODE=thread limit");
8218
8219 // Make this call behave like the thread limit algorithm.
8220 retval = __kmp_avail_proc - __kmp_nth +
8221 (root->r.r_active ? 1 : root->r.r_hot_team->t.t_nproc);
8222 if (retval > set_nproc) {
8223 retval = set_nproc;
8224 }
8225 if (retval < KMP_MIN_NTH) {
8226 retval = KMP_MIN_NTH;
8227 }
8228
8229 KB_TRACE(20, ("__kmp_load_balance_nproc: thread limit exit. retval:%d\n",
8230 retval));
8231 return retval;
8232 }
8233
8234 // There is a slight delay in the load balance algorithm in detecting new
8235 // running procs. The real system load at this instant should be at least as
8236 // large as the #active omp thread that are available to add to the team.
8237 if (system_active < team_curr_active) {
8238 system_active = team_curr_active;
8239 }
8240 retval = __kmp_avail_proc - system_active + team_curr_active;
8241 if (retval > set_nproc) {
8242 retval = set_nproc;
8243 }
8244 if (retval < KMP_MIN_NTH) {
8245 retval = KMP_MIN_NTH;
8246 }
8247
8248 KB_TRACE(20, ("__kmp_load_balance_nproc: exit. retval:%d\n", retval));
8249 return retval;
8250} // __kmp_load_balance_nproc()
8251
8252#endif /* USE_LOAD_BALANCE */
8253
8254/* ------------------------------------------------------------------------ */
8255
8256/* NOTE: this is called with the __kmp_init_lock held */
8257void __kmp_cleanup(void) {
8258 int f;
8259
8260 KA_TRACE(10, ("__kmp_cleanup: enter\n"));
8261
8263#if KMP_HANDLE_SIGNALS
8264 __kmp_remove_signals();
8265#endif
8267 }
8268
8269 if (TCR_4(__kmp_init_middle)) {
8270#if KMP_AFFINITY_SUPPORTED
8271 __kmp_affinity_uninitialize();
8272#endif /* KMP_AFFINITY_SUPPORTED */
8275 }
8276
8277 KA_TRACE(10, ("__kmp_cleanup: go serial cleanup\n"));
8278
8279 if (__kmp_init_serial) {
8282 }
8283
8285
8286 for (f = 0; f < __kmp_threads_capacity; f++) {
8287 if (__kmp_root[f] != NULL) {
8289 __kmp_root[f] = NULL;
8290 }
8291 }
8293 // __kmp_threads and __kmp_root were allocated at once, as single block, so
8294 // there is no need in freeing __kmp_root.
8295 __kmp_threads = NULL;
8296 __kmp_root = NULL;
8298
8299 // Free old __kmp_threads arrays if they exist.
8301 while (ptr) {
8302 kmp_old_threads_list_t *next = ptr->next;
8303 __kmp_free(ptr->threads);
8304 __kmp_free(ptr);
8305 ptr = next;
8306 }
8308
8309#if KMP_USE_DYNAMIC_LOCK
8310 __kmp_cleanup_indirect_user_locks();
8311#else
8313#endif
8314#if OMPD_SUPPORT
8315 if (ompd_env_block) {
8316 __kmp_free(ompd_env_block);
8317 ompd_env_block = NULL;
8318 ompd_env_block_size = 0;
8319 }
8320#endif
8321
8322#if KMP_AFFINITY_SUPPORTED
8323 KMP_INTERNAL_FREE(CCAST(char *, __kmp_cpuinfo_file));
8324 __kmp_cpuinfo_file = NULL;
8325#endif /* KMP_AFFINITY_SUPPORTED */
8326
8327#if KMP_USE_ADAPTIVE_LOCKS
8328#if KMP_DEBUG_ADAPTIVE_LOCKS
8329 __kmp_print_speculative_stats();
8330#endif
8331#endif
8333 __kmp_nested_nth.nth = NULL;
8334 __kmp_nested_nth.size = 0;
8335 __kmp_nested_nth.used = 0;
8336
8338 __kmp_nested_proc_bind.bind_types = NULL;
8339 __kmp_nested_proc_bind.size = 0;
8340 __kmp_nested_proc_bind.used = 0;
8345 __kmp_affinity_format = NULL;
8346 }
8347
8349
8351
8354
8355#if KMP_USE_HIER_SCHED
8356 __kmp_hier_scheds.deallocate();
8357#endif
8358
8359#if KMP_STATS_ENABLED
8361#endif
8362
8365
8366 KA_TRACE(10, ("__kmp_cleanup: exit\n"));
8367}
8368
8369/* ------------------------------------------------------------------------ */
8370
8372 char *env;
8373
8374 if ((env = getenv("KMP_IGNORE_MPPBEG")) != NULL) {
8375 if (__kmp_str_match_false(env))
8376 return FALSE;
8377 }
8378 // By default __kmpc_begin() is no-op.
8379 return TRUE;
8380}
8381
8383 char *env;
8384
8385 if ((env = getenv("KMP_IGNORE_MPPEND")) != NULL) {
8386 if (__kmp_str_match_false(env))
8387 return FALSE;
8388 }
8389 // By default __kmpc_end() is no-op.
8390 return TRUE;
8391}
8392
8394 int gtid;
8395 kmp_root_t *root;
8396
8397 /* this is a very important step as it will register new sibling threads
8398 and assign these new uber threads a new gtid */
8399 gtid = __kmp_entry_gtid();
8400 root = __kmp_threads[gtid]->th.th_root;
8402
8403 if (root->r.r_begin)
8404 return;
8405 __kmp_acquire_lock(&root->r.r_begin_lock, gtid);
8406 if (root->r.r_begin) {
8407 __kmp_release_lock(&root->r.r_begin_lock, gtid);
8408 return;
8409 }
8410
8411 root->r.r_begin = TRUE;
8412
8413 __kmp_release_lock(&root->r.r_begin_lock, gtid);
8414}
8415
8416/* ------------------------------------------------------------------------ */
8417
8419 int gtid;
8420 kmp_root_t *root;
8421 kmp_info_t *thread;
8422
8423 /* first, make sure we are initialized so we can get our gtid */
8424
8425 gtid = __kmp_entry_gtid();
8426 thread = __kmp_threads[gtid];
8427
8428 root = thread->th.th_root;
8429
8430 KA_TRACE(20, ("__kmp_user_set_library: enter T#%d, arg: %d, %d\n", gtid, arg,
8432 if (root->r.r_in_parallel) { /* Must be called in serial section of top-level
8433 thread */
8434 KMP_WARNING(SetLibraryIncorrectCall);
8435 return;
8436 }
8437
8438 switch (arg) {
8439 case library_serial:
8440 thread->th.th_set_nproc = 0;
8441 set__nproc(thread, 1);
8442 break;
8443 case library_turnaround:
8444 thread->th.th_set_nproc = 0;
8447 break;
8448 case library_throughput:
8449 thread->th.th_set_nproc = 0;
8452 break;
8453 default:
8454 KMP_FATAL(UnknownLibraryType, arg);
8455 }
8456
8458}
8459
8460void __kmp_aux_set_stacksize(size_t arg) {
8461 if (!__kmp_init_serial)
8463
8464#if KMP_OS_DARWIN
8465 if (arg & (0x1000 - 1)) {
8466 arg &= ~(0x1000 - 1);
8467 if (arg + 0x1000) /* check for overflow if we round up */
8468 arg += 0x1000;
8469 }
8470#endif
8472
8473 /* only change the default stacksize before the first parallel region */
8474 if (!TCR_4(__kmp_init_parallel)) {
8475 size_t value = arg; /* argument is in bytes */
8476
8479 else if (value > KMP_MAX_STKSIZE)
8481
8483
8484 __kmp_env_stksize = TRUE; /* was KMP_STACKSIZE specified? */
8485 }
8486
8488}
8489
8490/* set the behaviour of the runtime library */
8491/* TODO this can cause some odd behaviour with sibling parallelism... */
8493 __kmp_library = arg;
8494
8495 switch (__kmp_library) {
8496 case library_serial: {
8497 KMP_INFORM(LibraryIsSerial);
8498 } break;
8499 case library_turnaround:
8501 __kmp_use_yield = 2; // only yield when oversubscribed
8502 break;
8503 case library_throughput:
8506 break;
8507 default:
8508 KMP_FATAL(UnknownLibraryType, arg);
8509 }
8510}
8511
8512/* Getting team information common for all team API */
8513// Returns NULL if not in teams construct
8514static kmp_team_t *__kmp_aux_get_team_info(int &teams_serialized) {
8516 teams_serialized = 0;
8517 if (thr->th.th_teams_microtask) {
8518 kmp_team_t *team = thr->th.th_team;
8519 int tlevel = thr->th.th_teams_level; // the level of the teams construct
8520 int ii = team->t.t_level;
8521 teams_serialized = team->t.t_serialized;
8522 int level = tlevel + 1;
8523 KMP_DEBUG_ASSERT(ii >= tlevel);
8524 while (ii > level) {
8525 for (teams_serialized = team->t.t_serialized;
8526 (teams_serialized > 0) && (ii > level); teams_serialized--, ii--) {
8527 }
8528 if (team->t.t_serialized && (!teams_serialized)) {
8529 team = team->t.t_parent;
8530 continue;
8531 }
8532 if (ii > level) {
8533 team = team->t.t_parent;
8534 ii--;
8535 }
8536 }
8537 return team;
8538 }
8539 return NULL;
8540}
8541
8543 int serialized;
8544 kmp_team_t *team = __kmp_aux_get_team_info(serialized);
8545 if (team) {
8546 if (serialized > 1) {
8547 return 0; // teams region is serialized ( 1 team of 1 thread ).
8548 } else {
8549 return team->t.t_master_tid;
8550 }
8551 }
8552 return 0;
8553}
8554
8556 int serialized;
8557 kmp_team_t *team = __kmp_aux_get_team_info(serialized);
8558 if (team) {
8559 if (serialized > 1) {
8560 return 1;
8561 } else {
8562 return team->t.t_parent->t.t_nproc;
8563 }
8564 }
8565 return 1;
8566}
8567
8568/* ------------------------------------------------------------------------ */
8569
8570/*
8571 * Affinity Format Parser
8572 *
8573 * Field is in form of: %[[[0].]size]type
8574 * % and type are required (%% means print a literal '%')
8575 * type is either single char or long name surrounded by {},
8576 * e.g., N or {num_threads}
8577 * 0 => leading zeros
8578 * . => right justified when size is specified
8579 * by default output is left justified
8580 * size is the *minimum* field length
8581 * All other characters are printed as is
8582 *
8583 * Available field types:
8584 * L {thread_level} - omp_get_level()
8585 * n {thread_num} - omp_get_thread_num()
8586 * h {host} - name of host machine
8587 * P {process_id} - process id (integer)
8588 * T {thread_identifier} - native thread identifier (integer)
8589 * N {num_threads} - omp_get_num_threads()
8590 * A {ancestor_tnum} - omp_get_ancestor_thread_num(omp_get_level()-1)
8591 * a {thread_affinity} - comma separated list of integers or integer ranges
8592 * (values of affinity mask)
8593 *
8594 * Implementation-specific field types can be added
8595 * If a type is unknown, print "undefined"
8596 */
8597
8598// Structure holding the short name, long name, and corresponding data type
8599// for snprintf. A table of these will represent the entire valid keyword
8600// field types.
8602 char short_name; // from spec e.g., L -> thread level
8603 const char *long_name; // from spec thread_level -> thread level
8604 char field_format; // data type for snprintf (typically 'd' or 's'
8605 // for integer or string)
8607
8609#if KMP_AFFINITY_SUPPORTED
8610 {'A', "thread_affinity", 's'},
8611#endif
8612 {'t', "team_num", 'd'},
8613 {'T', "num_teams", 'd'},
8614 {'L', "nesting_level", 'd'},
8615 {'n', "thread_num", 'd'},
8616 {'N', "num_threads", 'd'},
8617 {'a', "ancestor_tnum", 'd'},
8618 {'H', "host", 's'},
8619 {'P', "process_id", 'd'},
8620 {'i', "native_thread_id", 'd'}};
8621
8622// Return the number of characters it takes to hold field
8623static int __kmp_aux_capture_affinity_field(int gtid, const kmp_info_t *th,
8624 const char **ptr,
8625 kmp_str_buf_t *field_buffer) {
8626 int rc, format_index, field_value;
8627 const char *width_left, *width_right;
8628 bool pad_zeros, right_justify, parse_long_name, found_valid_name;
8629 static const int FORMAT_SIZE = 20;
8630 char format[FORMAT_SIZE] = {0};
8631 char absolute_short_name = 0;
8632
8633 KMP_DEBUG_ASSERT(gtid >= 0);
8634 KMP_DEBUG_ASSERT(th);
8635 KMP_DEBUG_ASSERT(**ptr == '%');
8636 KMP_DEBUG_ASSERT(field_buffer);
8637
8638 __kmp_str_buf_clear(field_buffer);
8639
8640 // Skip the initial %
8641 (*ptr)++;
8642
8643 // Check for %% first
8644 if (**ptr == '%') {
8645 __kmp_str_buf_cat(field_buffer, "%", 1);
8646 (*ptr)++; // skip over the second %
8647 return 1;
8648 }
8649
8650 // Parse field modifiers if they are present
8651 pad_zeros = false;
8652 if (**ptr == '0') {
8653 pad_zeros = true;
8654 (*ptr)++; // skip over 0
8655 }
8656 right_justify = false;
8657 if (**ptr == '.') {
8658 right_justify = true;
8659 (*ptr)++; // skip over .
8660 }
8661 // Parse width of field: [width_left, width_right)
8662 width_left = width_right = NULL;
8663 if (**ptr >= '0' && **ptr <= '9') {
8664 width_left = *ptr;
8665 SKIP_DIGITS(*ptr);
8666 width_right = *ptr;
8667 }
8668
8669 // Create the format for KMP_SNPRINTF based on flags parsed above
8670 format_index = 0;
8671 format[format_index++] = '%';
8672 if (!right_justify)
8673 format[format_index++] = '-';
8674 if (pad_zeros)
8675 format[format_index++] = '0';
8676 if (width_left && width_right) {
8677 int i = 0;
8678 // Only allow 8 digit number widths.
8679 // This also prevents overflowing format variable
8680 while (i < 8 && width_left < width_right) {
8681 format[format_index++] = *width_left;
8682 width_left++;
8683 i++;
8684 }
8685 }
8686
8687 // Parse a name (long or short)
8688 // Canonicalize the name into absolute_short_name
8689 found_valid_name = false;
8690 parse_long_name = (**ptr == '{');
8691 if (parse_long_name)
8692 (*ptr)++; // skip initial left brace
8693 for (size_t i = 0; i < sizeof(__kmp_affinity_format_table) /
8694 sizeof(__kmp_affinity_format_table[0]);
8695 ++i) {
8696 char short_name = __kmp_affinity_format_table[i].short_name;
8697 const char *long_name = __kmp_affinity_format_table[i].long_name;
8698 char field_format = __kmp_affinity_format_table[i].field_format;
8699 if (parse_long_name) {
8700 size_t length = KMP_STRLEN(long_name);
8701 if (strncmp(*ptr, long_name, length) == 0) {
8702 found_valid_name = true;
8703 (*ptr) += length; // skip the long name
8704 }
8705 } else if (**ptr == short_name) {
8706 found_valid_name = true;
8707 (*ptr)++; // skip the short name
8708 }
8709 if (found_valid_name) {
8710 format[format_index++] = field_format;
8711 format[format_index++] = '\0';
8712 absolute_short_name = short_name;
8713 break;
8714 }
8715 }
8716 if (parse_long_name) {
8717 if (**ptr != '}') {
8718 absolute_short_name = 0;
8719 } else {
8720 (*ptr)++; // skip over the right brace
8721 }
8722 }
8723
8724 // Attempt to fill the buffer with the requested
8725 // value using snprintf within __kmp_str_buf_print()
8726 switch (absolute_short_name) {
8727 case 't':
8728 rc = __kmp_str_buf_print(field_buffer, format, __kmp_aux_get_team_num());
8729 break;
8730 case 'T':
8731 rc = __kmp_str_buf_print(field_buffer, format, __kmp_aux_get_num_teams());
8732 break;
8733 case 'L':
8734 rc = __kmp_str_buf_print(field_buffer, format, th->th.th_team->t.t_level);
8735 break;
8736 case 'n':
8737 rc = __kmp_str_buf_print(field_buffer, format, __kmp_tid_from_gtid(gtid));
8738 break;
8739 case 'H': {
8740 static const int BUFFER_SIZE = 256;
8741 char buf[BUFFER_SIZE];
8743 rc = __kmp_str_buf_print(field_buffer, format, buf);
8744 } break;
8745 case 'P':
8746 rc = __kmp_str_buf_print(field_buffer, format, getpid());
8747 break;
8748 case 'i':
8749 rc = __kmp_str_buf_print(field_buffer, format, __kmp_gettid());
8750 break;
8751 case 'N':
8752 rc = __kmp_str_buf_print(field_buffer, format, th->th.th_team->t.t_nproc);
8753 break;
8754 case 'a':
8755 field_value =
8756 __kmp_get_ancestor_thread_num(gtid, th->th.th_team->t.t_level - 1);
8757 rc = __kmp_str_buf_print(field_buffer, format, field_value);
8758 break;
8759#if KMP_AFFINITY_SUPPORTED
8760 case 'A': {
8761 if (th->th.th_affin_mask) {
8764 __kmp_affinity_str_buf_mask(&buf, th->th.th_affin_mask);
8765 rc = __kmp_str_buf_print(field_buffer, format, buf.str);
8767 } else {
8768 rc = __kmp_str_buf_print(field_buffer, "%s", "disabled");
8769 }
8770 } break;
8771#endif
8772 default:
8773 // According to spec, If an implementation does not have info for field
8774 // type, then "undefined" is printed
8775 rc = __kmp_str_buf_print(field_buffer, "%s", "undefined");
8776 // Skip the field
8777 if (parse_long_name) {
8778 SKIP_TOKEN(*ptr);
8779 if (**ptr == '}')
8780 (*ptr)++;
8781 } else {
8782 (*ptr)++;
8783 }
8784 }
8785
8786 KMP_ASSERT(format_index <= FORMAT_SIZE);
8787 return rc;
8788}
8789
8790/*
8791 * Return number of characters needed to hold the affinity string
8792 * (not including null byte character)
8793 * The resultant string is printed to buffer, which the caller can then
8794 * handle afterwards
8795 */
8796size_t __kmp_aux_capture_affinity(int gtid, const char *format,
8797 kmp_str_buf_t *buffer) {
8798 const char *parse_ptr;
8799 size_t retval;
8800 const kmp_info_t *th;
8801 kmp_str_buf_t field;
8802
8803 KMP_DEBUG_ASSERT(buffer);
8804 KMP_DEBUG_ASSERT(gtid >= 0);
8805
8806 __kmp_str_buf_init(&field);
8807 __kmp_str_buf_clear(buffer);
8808
8809 th = __kmp_threads[gtid];
8810 retval = 0;
8811
8812 // If format is NULL or zero-length string, then we use
8813 // affinity-format-var ICV
8814 parse_ptr = format;
8815 if (parse_ptr == NULL || *parse_ptr == '\0') {
8816 parse_ptr = __kmp_affinity_format;
8817 }
8818 KMP_DEBUG_ASSERT(parse_ptr);
8819
8820 while (*parse_ptr != '\0') {
8821 // Parse a field
8822 if (*parse_ptr == '%') {
8823 // Put field in the buffer
8824 int rc = __kmp_aux_capture_affinity_field(gtid, th, &parse_ptr, &field);
8825 __kmp_str_buf_catbuf(buffer, &field);
8826 retval += rc;
8827 } else {
8828 // Put literal character in buffer
8829 __kmp_str_buf_cat(buffer, parse_ptr, 1);
8830 retval++;
8831 parse_ptr++;
8832 }
8833 }
8834 __kmp_str_buf_free(&field);
8835 return retval;
8836}
8837
8838// Displays the affinity string to stdout
8839void __kmp_aux_display_affinity(int gtid, const char *format) {
8842 __kmp_aux_capture_affinity(gtid, format, &buf);
8843 __kmp_fprintf(kmp_out, "%s" KMP_END_OF_LINE, buf.str);
8845}
8846
8847/* ------------------------------------------------------------------------ */
8848void __kmp_aux_set_blocktime(int arg, kmp_info_t *thread, int tid) {
8849 int blocktime = arg; /* argument is in microseconds */
8850#if KMP_USE_MONITOR
8851 int bt_intervals;
8852#endif
8853 kmp_int8 bt_set;
8854
8856
8857 /* Normalize and set blocktime for the teams */
8858 if (blocktime < KMP_MIN_BLOCKTIME)
8859 blocktime = KMP_MIN_BLOCKTIME;
8860 else if (blocktime > KMP_MAX_BLOCKTIME)
8861 blocktime = KMP_MAX_BLOCKTIME;
8862
8863 set__blocktime_team(thread->th.th_team, tid, blocktime);
8864 set__blocktime_team(thread->th.th_serial_team, 0, blocktime);
8865
8866#if KMP_USE_MONITOR
8867 /* Calculate and set blocktime intervals for the teams */
8868 bt_intervals = KMP_INTERVALS_FROM_BLOCKTIME(blocktime, __kmp_monitor_wakeups);
8869
8870 set__bt_intervals_team(thread->th.th_team, tid, bt_intervals);
8871 set__bt_intervals_team(thread->th.th_serial_team, 0, bt_intervals);
8872#endif
8873
8874 /* Set whether blocktime has been set to "TRUE" */
8875 bt_set = TRUE;
8876
8877 set__bt_set_team(thread->th.th_team, tid, bt_set);
8878 set__bt_set_team(thread->th.th_serial_team, 0, bt_set);
8879#if KMP_USE_MONITOR
8880 KF_TRACE(10, ("kmp_set_blocktime: T#%d(%d:%d), blocktime=%d, "
8881 "bt_intervals=%d, monitor_updates=%d\n",
8882 __kmp_gtid_from_tid(tid, thread->th.th_team),
8883 thread->th.th_team->t.t_id, tid, blocktime, bt_intervals,
8884 __kmp_monitor_wakeups));
8885#else
8886 KF_TRACE(10, ("kmp_set_blocktime: T#%d(%d:%d), blocktime=%d\n",
8887 __kmp_gtid_from_tid(tid, thread->th.th_team),
8888 thread->th.th_team->t.t_id, tid, blocktime));
8889#endif
8890}
8891
8892void __kmp_aux_set_defaults(char const *str, size_t len) {
8893 if (!__kmp_init_serial) {
8895 }
8897
8900 }
8901} // __kmp_aux_set_defaults
8902
8903/* ------------------------------------------------------------------------ */
8904/* internal fast reduction routines */
8905
8908 ident_t *loc, kmp_int32 global_tid, kmp_int32 num_vars, size_t reduce_size,
8909 void *reduce_data, void (*reduce_func)(void *lhs_data, void *rhs_data),
8911
8912 // Default reduction method: critical construct ( lck != NULL, like in current
8913 // PAROPT )
8914 // If ( reduce_data!=NULL && reduce_func!=NULL ): the tree-reduction method
8915 // can be selected by RTL
8916 // If loc->flags contains KMP_IDENT_ATOMIC_REDUCE, the atomic reduce method
8917 // can be selected by RTL
8918 // Finally, it's up to OpenMP RTL to make a decision on which method to select
8919 // among generated by PAROPT.
8920
8922
8923 int team_size;
8924
8925 KMP_DEBUG_ASSERT(lck); // it would be nice to test ( lck != 0 )
8926
8927#define FAST_REDUCTION_ATOMIC_METHOD_GENERATED \
8928 (loc && \
8929 ((loc->flags & (KMP_IDENT_ATOMIC_REDUCE)) == (KMP_IDENT_ATOMIC_REDUCE)))
8930#define FAST_REDUCTION_TREE_METHOD_GENERATED ((reduce_data) && (reduce_func))
8931
8932 retval = critical_reduce_block;
8933
8934 // another choice of getting a team size (with 1 dynamic deference) is slower
8935 team_size = __kmp_get_team_num_threads(global_tid);
8936 if (team_size == 1) {
8937
8938 retval = empty_reduce_block;
8939
8940 } else {
8941
8942 int atomic_available = FAST_REDUCTION_ATOMIC_METHOD_GENERATED;
8943
8944#if KMP_ARCH_X86_64 || KMP_ARCH_PPC64 || KMP_ARCH_AARCH64 || \
8945 KMP_ARCH_MIPS64 || KMP_ARCH_RISCV64 || KMP_ARCH_LOONGARCH64 || \
8946 KMP_ARCH_VE || KMP_ARCH_S390X || KMP_ARCH_WASM32 || KMP_ARCH_WASM64 || \
8947 KMP_ARCH_ARM64EC
8948
8949#if KMP_OS_LINUX || KMP_OS_DRAGONFLY || KMP_OS_FREEBSD || KMP_OS_NETBSD || \
8950 KMP_OS_OPENBSD || KMP_OS_WINDOWS || KMP_OS_DARWIN || KMP_OS_HAIKU || \
8951 KMP_OS_HURD || KMP_OS_SOLARIS || KMP_OS_WASI || KMP_OS_AIX
8952
8953 int teamsize_cutoff = 4;
8954
8955#if KMP_MIC_SUPPORTED
8956 if (__kmp_mic_type != non_mic) {
8957 teamsize_cutoff = 8;
8958 }
8959#endif
8960 int tree_available = FAST_REDUCTION_TREE_METHOD_GENERATED;
8961 if (tree_available) {
8962 if (team_size <= teamsize_cutoff) {
8963 if (atomic_available) {
8964 retval = atomic_reduce_block;
8965 }
8966 } else {
8967 retval = TREE_REDUCE_BLOCK_WITH_REDUCTION_BARRIER;
8968 }
8969 } else if (atomic_available) {
8970 retval = atomic_reduce_block;
8971 }
8972#else
8973#error "Unknown or unsupported OS"
8974#endif // KMP_OS_LINUX || KMP_OS_DRAGONFLY || KMP_OS_FREEBSD || KMP_OS_NETBSD ||
8975 // KMP_OS_OPENBSD || KMP_OS_WINDOWS || KMP_OS_DARWIN || KMP_OS_HAIKU ||
8976 // KMP_OS_HURD || KMP_OS_SOLARIS || KMP_OS_WASI || KMP_OS_AIX
8977
8978#elif KMP_ARCH_X86 || KMP_ARCH_ARM || KMP_ARCH_AARCH || KMP_ARCH_MIPS || \
8979 KMP_ARCH_PPC || KMP_ARCH_AARCH64_32 || KMP_ARCH_SPARC
8980
8981#if KMP_OS_LINUX || KMP_OS_DRAGONFLY || KMP_OS_FREEBSD || KMP_OS_NETBSD || \
8982 KMP_OS_OPENBSD || KMP_OS_WINDOWS || KMP_OS_HAIKU || KMP_OS_HURD || \
8983 KMP_OS_SOLARIS || KMP_OS_WASI || KMP_OS_AIX
8984
8985 // basic tuning
8986
8987 if (atomic_available) {
8988 if (num_vars <= 2) { // && ( team_size <= 8 ) due to false-sharing ???
8989 retval = atomic_reduce_block;
8990 }
8991 } // otherwise: use critical section
8992
8993#elif KMP_OS_DARWIN
8994
8995 int tree_available = FAST_REDUCTION_TREE_METHOD_GENERATED;
8996 if (atomic_available && (num_vars <= 3)) {
8997 retval = atomic_reduce_block;
8998 } else if (tree_available) {
8999 if ((reduce_size > (9 * sizeof(kmp_real64))) &&
9000 (reduce_size < (2000 * sizeof(kmp_real64)))) {
9001 retval = TREE_REDUCE_BLOCK_WITH_PLAIN_BARRIER;
9002 }
9003 } // otherwise: use critical section
9004
9005#else
9006#error "Unknown or unsupported OS"
9007#endif
9008
9009#else
9010#error "Unknown or unsupported architecture"
9011#endif
9012 }
9013
9014 // KMP_FORCE_REDUCTION
9015
9016 // If the team is serialized (team_size == 1), ignore the forced reduction
9017 // method and stay with the unsynchronized method (empty_reduce_block)
9019 team_size != 1) {
9020
9022
9023 int atomic_available, tree_available;
9024
9025 switch ((forced_retval = __kmp_force_reduction_method)) {
9027 KMP_ASSERT(lck); // lck should be != 0
9028 break;
9029
9031 atomic_available = FAST_REDUCTION_ATOMIC_METHOD_GENERATED;
9032 if (!atomic_available) {
9033 KMP_WARNING(RedMethodNotSupported, "atomic");
9034 forced_retval = critical_reduce_block;
9035 }
9036 break;
9037
9038 case tree_reduce_block:
9039 tree_available = FAST_REDUCTION_TREE_METHOD_GENERATED;
9040 if (!tree_available) {
9041 KMP_WARNING(RedMethodNotSupported, "tree");
9042 forced_retval = critical_reduce_block;
9043 } else {
9044#if KMP_FAST_REDUCTION_BARRIER
9045 forced_retval = TREE_REDUCE_BLOCK_WITH_REDUCTION_BARRIER;
9046#endif
9047 }
9048 break;
9049
9050 default:
9051 KMP_ASSERT(0); // "unsupported method specified"
9052 }
9053
9054 retval = forced_retval;
9055 }
9056
9057 KA_TRACE(10, ("reduction method selected=%08x\n", retval));
9058
9059#undef FAST_REDUCTION_TREE_METHOD_GENERATED
9060#undef FAST_REDUCTION_ATOMIC_METHOD_GENERATED
9061
9062 return (retval);
9063}
9064// this function is for testing set/get/determine reduce method
9066 return ((__kmp_entry_thread()->th.th_local.packed_reduction_method) >> 8);
9067}
9068
9069// Soft pause sets up threads to ignore blocktime and just go to sleep.
9070// Spin-wait code checks __kmp_pause_status and reacts accordingly.
9072
9073// Hard pause shuts down the runtime completely. Resume happens naturally when
9074// OpenMP is used subsequently.
9079
9080// Soft resume sets __kmp_pause_status, and wakes up all threads.
9084
9085 for (int gtid = 1; gtid < __kmp_threads_capacity; ++gtid) {
9086 kmp_info_t *thread = __kmp_threads[gtid];
9087 if (thread) { // Wake it if sleeping
9088 kmp_flag_64<> fl(&thread->th.th_bar[bs_forkjoin_barrier].bb.b_go,
9089 thread);
9090 if (fl.is_sleeping())
9091 fl.resume(gtid);
9092 else if (__kmp_try_suspend_mx(thread)) { // got suspend lock
9093 __kmp_unlock_suspend_mx(thread); // unlock it; it won't sleep
9094 } else { // thread holds the lock and may sleep soon
9095 do { // until either the thread sleeps, or we can get the lock
9096 if (fl.is_sleeping()) {
9097 fl.resume(gtid);
9098 break;
9099 } else if (__kmp_try_suspend_mx(thread)) {
9101 break;
9102 }
9103 } while (1);
9104 }
9105 }
9106 }
9107 }
9108}
9109
9110// This function is called via __kmpc_pause_resource. Returns 0 if successful.
9111// TODO: add warning messages
9113 if (level == kmp_not_paused) { // requesting resume
9115 // error message about runtime not being paused, so can't resume
9116 return 1;
9117 } else {
9121 return 0;
9122 }
9123 } else if (level == kmp_soft_paused) { // requesting soft pause
9125 // error message about already being paused
9126 return 1;
9127 } else {
9129 return 0;
9130 }
9131 } else if (level == kmp_hard_paused || level == kmp_stop_tool_paused) {
9132 // requesting hard pause or stop_tool pause
9134 // error message about already being paused
9135 return 1;
9136 } else {
9138 return 0;
9139 }
9140 } else {
9141 // error message about invalid level
9142 return 1;
9143 }
9144}
9145
9153
9154// The team size is changing, so distributed barrier must be modified
9155void __kmp_resize_dist_barrier(kmp_team_t *team, int old_nthreads,
9156 int new_nthreads) {
9158 bp_dist_bar);
9159 kmp_info_t **other_threads = team->t.t_threads;
9160
9161 // We want all the workers to stop waiting on the barrier while we adjust the
9162 // size of the team.
9163 for (int f = 1; f < old_nthreads; ++f) {
9164 KMP_DEBUG_ASSERT(other_threads[f] != NULL);
9165 // Ignore threads that are already inactive or not present in the team
9166 if (team->t.t_threads[f]->th.th_used_in_team.load() == 0) {
9167 // teams construct causes thread_limit to get passed in, and some of
9168 // those could be inactive; just ignore them
9169 continue;
9170 }
9171 // If thread is transitioning still to in_use state, wait for it
9172 if (team->t.t_threads[f]->th.th_used_in_team.load() == 3) {
9173 while (team->t.t_threads[f]->th.th_used_in_team.load() == 3)
9174 KMP_CPU_PAUSE();
9175 }
9176 // The thread should be in_use now
9177 KMP_DEBUG_ASSERT(team->t.t_threads[f]->th.th_used_in_team.load() == 1);
9178 // Transition to unused state
9179 team->t.t_threads[f]->th.th_used_in_team.store(2);
9180 KMP_DEBUG_ASSERT(team->t.t_threads[f]->th.th_used_in_team.load() == 2);
9181 }
9182 // Release all the workers
9183 team->t.b->go_release();
9184
9185 KMP_MFENCE();
9186
9187 // Workers should see transition status 2 and move to 0; but may need to be
9188 // woken up first
9189 int count = old_nthreads - 1;
9190 while (count > 0) {
9191 count = old_nthreads - 1;
9192 for (int f = 1; f < old_nthreads; ++f) {
9193 if (other_threads[f]->th.th_used_in_team.load() != 0) {
9194 if (__kmp_dflt_blocktime != KMP_MAX_BLOCKTIME) { // Wake up the workers
9196 void *, other_threads[f]->th.th_sleep_loc);
9197 __kmp_atomic_resume_64(other_threads[f]->th.th_info.ds.ds_gtid, flag);
9198 }
9199 } else {
9200 KMP_DEBUG_ASSERT(team->t.t_threads[f]->th.th_used_in_team.load() == 0);
9201 count--;
9202 }
9203 }
9204 }
9205 // Now update the barrier size
9206 team->t.b->update_num_threads(new_nthreads);
9207 team->t.b->go_reset();
9208}
9209
9210void __kmp_add_threads_to_team(kmp_team_t *team, int new_nthreads) {
9211 // Add the threads back to the team
9212 KMP_DEBUG_ASSERT(team);
9213 // Threads were paused and pointed at th_used_in_team temporarily during a
9214 // resize of the team. We're going to set th_used_in_team to 3 to indicate to
9215 // the thread that it should transition itself back into the team. Then, if
9216 // blocktime isn't infinite, the thread could be sleeping, so we send a resume
9217 // to wake it up.
9218 for (int f = 1; f < new_nthreads; ++f) {
9219 KMP_DEBUG_ASSERT(team->t.t_threads[f]);
9221 &(team->t.t_threads[f]->th.th_used_in_team), 0, 3);
9222 if (__kmp_dflt_blocktime != KMP_MAX_BLOCKTIME) { // Wake up sleeping threads
9223 __kmp_resume_32(team->t.t_threads[f]->th.th_info.ds.ds_gtid,
9225 }
9226 }
9227 // The threads should be transitioning to the team; when they are done, they
9228 // should have set th_used_in_team to 1. This loop forces master to wait until
9229 // all threads have moved into the team and are waiting in the barrier.
9230 int count = new_nthreads - 1;
9231 while (count > 0) {
9232 count = new_nthreads - 1;
9233 for (int f = 1; f < new_nthreads; ++f) {
9234 if (team->t.t_threads[f]->th.th_used_in_team.load() == 1) {
9235 count--;
9236 }
9237 }
9238 }
9239}
9240
9241// Globals and functions for hidden helper task
9245#if KMP_OS_LINUX
9248#else
9251#endif
9252
9253namespace {
9254std::atomic<kmp_int32> __kmp_hit_hidden_helper_threads_num;
9255
9256void __kmp_hidden_helper_wrapper_fn(int *gtid, int *, ...) {
9257 // This is an explicit synchronization on all hidden helper threads in case
9258 // that when a regular thread pushes a hidden helper task to one hidden
9259 // helper thread, the thread has not been awaken once since they're released
9260 // by the main thread after creating the team.
9261 KMP_ATOMIC_INC(&__kmp_hit_hidden_helper_threads_num);
9262 while (KMP_ATOMIC_LD_ACQ(&__kmp_hit_hidden_helper_threads_num) !=
9264 ;
9265
9266 // If main thread, then wait for signal
9267 if (__kmpc_master(nullptr, *gtid)) {
9268 // First, unset the initial state and release the initial thread
9272 // Now wake up all worker threads
9273 for (int i = 1; i < __kmp_hit_hidden_helper_threads_num; ++i) {
9275 }
9276 }
9277}
9278} // namespace
9279
9281 // Create a new root for hidden helper team/threads
9282 const int gtid = __kmp_register_root(TRUE);
9285 __kmp_hidden_helper_main_thread->th.th_set_nproc =
9287
9288 KMP_ATOMIC_ST_REL(&__kmp_hit_hidden_helper_threads_num, 0);
9289
9290 __kmpc_fork_call(nullptr, 0, __kmp_hidden_helper_wrapper_fn);
9291
9292 // Set the initialization flag to FALSE
9294
9296}
9297
9298/* Nesting Mode:
9299 Set via KMP_NESTING_MODE, which takes an integer.
9300 Note: we skip duplicate topology levels, and skip levels with only
9301 one entity.
9302 KMP_NESTING_MODE=0 is the default, and doesn't use nesting mode.
9303 KMP_NESTING_MODE=1 sets as many nesting levels as there are distinct levels
9304 in the topology, and initializes the number of threads at each of those
9305 levels to the number of entities at each level, respectively, below the
9306 entity at the parent level.
9307 KMP_NESTING_MODE=N, where N>1, attempts to create up to N nesting levels,
9308 but starts with nesting OFF -- max-active-levels-var is 1 -- and requires
9309 the user to turn nesting on explicitly. This is an even more experimental
9310 option to this experimental feature, and may change or go away in the
9311 future.
9312*/
9313
9314// Allocate space to store nesting levels
9316 int levels = KMP_HW_LAST;
9318 __kmp_nesting_nth_level = (int *)KMP_INTERNAL_MALLOC(levels * sizeof(int));
9319 for (int i = 0; i < levels; ++i)
9321 if (__kmp_nested_nth.size < levels) {
9322 __kmp_nested_nth.nth =
9323 (int *)KMP_INTERNAL_REALLOC(__kmp_nested_nth.nth, levels * sizeof(int));
9324 __kmp_nested_nth.size = levels;
9325 }
9326}
9327
9328// Set # threads for top levels of nesting; must be called after topology set
9331
9332 if (__kmp_nesting_mode == 1)
9334 else if (__kmp_nesting_mode > 1)
9336
9337 if (__kmp_topology) { // use topology info
9338 int loc, hw_level;
9339 for (loc = 0, hw_level = 0; hw_level < __kmp_topology->get_depth() &&
9341 loc++, hw_level++) {
9342 __kmp_nesting_nth_level[loc] = __kmp_topology->get_ratio(hw_level);
9343 if (__kmp_nesting_nth_level[loc] == 1)
9344 loc--;
9345 }
9346 // Make sure all cores are used
9347 if (__kmp_nesting_mode > 1 && loc > 1) {
9348 int core_level = __kmp_topology->get_level(KMP_HW_CORE);
9349 int num_cores = __kmp_topology->get_count(core_level);
9350 int upper_levels = 1;
9351 for (int level = 0; level < loc - 1; ++level)
9352 upper_levels *= __kmp_nesting_nth_level[level];
9353 if (upper_levels * __kmp_nesting_nth_level[loc - 1] < num_cores)
9355 num_cores / __kmp_nesting_nth_level[loc - 2];
9356 }
9359 } else { // no topology info available; provide a reasonable guesstimation
9360 if (__kmp_avail_proc >= 4) {
9364 } else {
9367 }
9369 }
9370 for (int i = 0; i < __kmp_nesting_mode_nlevels; ++i) {
9372 }
9376 if (get__max_active_levels(thread) > 1) {
9377 // if max levels was set, set nesting mode levels to same
9379 }
9380 if (__kmp_nesting_mode == 1) // turn on nesting for this case only
9382}
9383
9384#if ENABLE_LIBOMPTARGET
9385void (*kmp_target_sync_cb)(ident_t *loc_ref, int gtid, void *current_task,
9386 void *event) = NULL;
9387void __kmp_target_init() {
9388 // Look for hooks in the libomptarget library
9389 *(void **)(&kmp_target_sync_cb) = KMP_DLSYM("__tgt_target_sync");
9390}
9391#endif // ENABLE_LIBOMPTARGET
9392
9393// Empty symbols to export (see exports_so.txt) when feature is disabled
9394extern "C" {
9395#if !KMP_STATS_ENABLED
9397#endif
9398#if !USE_DEBUGGER
9401#endif
9402#if !USE_ITT_BUILD || !USE_ITT_NOTIFY
9405#endif
9406}
9407
9408// end of file
char buf[BUFFER_SIZE]
#define BUFFER_SIZE
uint8_t kmp_uint8
A simple pure header implementation of VLA that aims to replace uses of actual VLA,...
Definition kmp_utils.h:26
static void deallocate(distributedBarrier *db)
static distributedBarrier * allocate(int nThreads)
void resume(int th_gtid)
bool is_sleeping()
Test whether there are threads sleeping on the flag.
int64_t kmp_int64
Definition common.h:10
@ KMP_IDENT_AUTOPAR
Entry point generated by auto-parallelization.
Definition kmp.h:182
KMP_EXPORT void __kmpc_serialized_parallel(ident_t *, kmp_int32 global_tid)
KMP_EXPORT void __kmpc_fork_call(ident_t *, kmp_int32 nargs, kmpc_micro microtask,...)
KMP_EXPORT void __kmpc_end_serialized_parallel(ident_t *, kmp_int32 global_tid)
sched_type
Describes the loop schedule to be used for a parallel for loop.
Definition kmp.h:340
KMP_EXPORT kmp_int32 __kmpc_master(ident_t *, kmp_int32 global_tid)
@ kmp_sch_auto
auto
Definition kmp.h:347
@ kmp_sch_static
static unspecialized
Definition kmp.h:343
@ kmp_sch_guided_chunked
guided unspecialized
Definition kmp.h:345
@ kmp_sch_dynamic_chunked
Definition kmp.h:344
@ kmp_sch_guided_analytical_chunked
Definition kmp.h:355
@ kmp_sch_static_balanced
Definition kmp.h:352
@ kmp_sch_static_greedy
Definition kmp.h:351
@ kmp_sch_static_chunked
Definition kmp.h:342
@ kmp_sch_trapezoidal
Definition kmp.h:348
@ kmp_sch_guided_iterative_chunked
Definition kmp.h:354
@ kmp_sch_static_steal
Definition kmp.h:357
__itt_string_handle * name
Definition ittnotify.h:3305
void
Definition ittnotify.h:3324
void const char const char int ITT_FORMAT __itt_group_sync s
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t ITT_FORMAT d no args const wchar_t const wchar_t ITT_FORMAT s __itt_heap_function void size_t int ITT_FORMAT d __itt_heap_function void ITT_FORMAT p __itt_heap_function void void size_t new_size
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t count
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t ITT_FORMAT d no args const wchar_t const wchar_t ITT_FORMAT s __itt_heap_function void size_t int ITT_FORMAT d __itt_heap_function void ITT_FORMAT p __itt_heap_function void void size_t int ITT_FORMAT d no args no args unsigned int ITT_FORMAT u const __itt_domain __itt_id ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain __itt_id ITT_FORMAT p const __itt_domain __itt_id __itt_timestamp __itt_timestamp ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain ITT_FORMAT p const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_id __itt_string_handle __itt_metadata_type size_t void ITT_FORMAT p const __itt_domain __itt_id __itt_string_handle const wchar_t size_t length
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t ITT_FORMAT d no args const wchar_t const wchar_t ITT_FORMAT s __itt_heap_function void size_t int ITT_FORMAT d __itt_heap_function void ITT_FORMAT p __itt_heap_function void void size_t int ITT_FORMAT d no args no args unsigned int ITT_FORMAT u const __itt_domain __itt_id ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain __itt_id ITT_FORMAT p const __itt_domain __itt_id __itt_timestamp __itt_timestamp ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain ITT_FORMAT p const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_id __itt_string_handle __itt_metadata_type size_t void ITT_FORMAT p const __itt_domain __itt_id __itt_string_handle const wchar_t size_t ITT_FORMAT lu const __itt_domain __itt_id __itt_relation __itt_id ITT_FORMAT p const wchar_t int ITT_FORMAT __itt_group_mark S
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t ITT_FORMAT d no args const wchar_t const wchar_t ITT_FORMAT s __itt_heap_function void size_t int ITT_FORMAT d __itt_heap_function void ITT_FORMAT p __itt_heap_function void void size_t int ITT_FORMAT d no args no args unsigned int ITT_FORMAT u const __itt_domain __itt_id ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain __itt_id ITT_FORMAT p const __itt_domain __itt_id __itt_timestamp __itt_timestamp ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain ITT_FORMAT p const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_id __itt_string_handle __itt_metadata_type size_t void ITT_FORMAT p const __itt_domain __itt_id __itt_string_handle const wchar_t size_t ITT_FORMAT lu const __itt_domain __itt_id __itt_relation __itt_id ITT_FORMAT p const wchar_t int ITT_FORMAT __itt_group_mark d __itt_event event
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t ITT_FORMAT d no args const wchar_t const wchar_t ITT_FORMAT s __itt_heap_function void size_t int ITT_FORMAT d __itt_heap_function void ITT_FORMAT p __itt_heap_function void void size_t int ITT_FORMAT d no args no args unsigned int ITT_FORMAT u const __itt_domain __itt_id ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain __itt_id ITT_FORMAT p const __itt_domain __itt_id __itt_timestamp __itt_timestamp ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain ITT_FORMAT p const __itt_domain __itt_string_handle unsigned long long value
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t size
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t ITT_FORMAT d no args const wchar_t const wchar_t ITT_FORMAT s __itt_heap_function void size_t int ITT_FORMAT d __itt_heap_function void ITT_FORMAT p __itt_heap_function void void size_t int ITT_FORMAT d no args no args unsigned int ITT_FORMAT u const __itt_domain __itt_id ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain __itt_id ITT_FORMAT p const __itt_domain __itt_id __itt_timestamp __itt_timestamp ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain ITT_FORMAT p const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_id __itt_string_handle __itt_metadata_type size_t void ITT_FORMAT p const __itt_domain __itt_id __itt_string_handle const wchar_t size_t ITT_FORMAT lu const __itt_domain __itt_id __itt_relation __itt_id tail
void __kmp_free_task_team(kmp_info_t *thread, kmp_task_team_t *task_team)
struct kmp_disp kmp_disp_t
int __kmp_hot_teams_max_level
void __kmp_finish_implicit_task(kmp_info_t *this_thr)
volatile kmp_team_t * __kmp_team_pool
#define get__dynamic_2(xteam, xtid)
Definition kmp.h:2390
#define __kmp_free(ptr)
Definition kmp.h:3749
kmp_bar_pat_e __kmp_barrier_release_pat_dflt
kmp_info_t * __kmp_hidden_helper_main_thread
int __kmp_generate_warnings
int __kmp_cg_max_nth
int __kmp_abort_delay
kmp_proc_bind_t __kmp_teams_proc_bind
#define KMP_INTERNAL_MALLOC(sz)
Definition kmp.h:103
#define KMP_CPU_PAUSE()
Definition kmp.h:1585
#define KMP_DEFAULT_CHUNK
Definition kmp.h:1305
bool __kmp_detect_shm()
int __kmp_version
kmp_bootstrap_lock_t __kmp_initz_lock
#define KMP_MAX_STKSIZE
Definition kmp.h:1207
#define KMP_MAX_STKPADDING
Definition kmp.h:1240
int __kmp_display_env_verbose
kmp_global_t __kmp_global
void __kmp_init_target_mem()
union kmp_task_team kmp_task_team_t
Definition kmp.h:243
@ ct_psingle
Definition kmp.h:1688
@ ct_ordered_in_parallel
Definition kmp.h:1690
void __kmp_hidden_helper_worker_thread_signal()
void __kmp_teams_master(int gtid)
void __kmp_common_initialize(void)
void __kmp_release_64(kmp_flag_64<> *flag)
kmp_pause_status_t __kmp_pause_status
#define KMP_MAX_BLOCKTIME
Definition kmp.h:1245
int __kmp_teams_max_nth
void __kmp_read_system_time(double *delta)
kmp_bootstrap_lock_t __kmp_tp_cached_lock
void __kmp_reap_task_teams(void)
kmp_int32 __kmp_use_yield
kmp_pause_status_t
Definition kmp.h:4538
@ kmp_hard_paused
Definition kmp.h:4541
@ kmp_stop_tool_paused
Definition kmp.h:4542
@ kmp_soft_paused
Definition kmp.h:4540
@ kmp_not_paused
Definition kmp.h:4539
int __kmp_dflt_team_nth_ub
void __kmp_hidden_helper_threads_initz_wait()
struct dispatch_shared_info dispatch_shared_info_t
void __kmp_fini_target_mem()
Finalize target memory support.
#define KMP_INTERNAL_REALLOC(p, sz)
Definition kmp.h:105
#define get__nproc_2(xteam, xtid)
Definition kmp.h:2392
void __kmp_wait_to_unref_task_teams(void)
union kmp_team kmp_team_p
Definition kmp.h:244
struct KMP_ALIGN_CACHE dispatch_private_info dispatch_private_info_t
#define __kmp_assign_root_init_mask()
Definition kmp.h:3949
int __kmp_dflt_max_active_levels
struct kmp_hot_team_ptr kmp_hot_team_ptr_t
#define KMP_NOT_SAFE_TO_REAP
Definition kmp.h:2139
int __kmp_xproc
int __kmp_debug_buf
void __kmp_unlock_suspend_mx(kmp_info_t *th)
kmp_bar_pat_e __kmp_barrier_gather_pat_dflt
#define KMP_HIDDEN_HELPER_TEAM(team)
Definition kmp.h:4599
static kmp_team_t * __kmp_team_from_gtid(int gtid)
Definition kmp.h:3632
void __kmp_do_initialize_hidden_helper_threads()
struct kmp_local kmp_local_t
kmp_bar_pat_e __kmp_barrier_gather_pattern[bs_last_barrier]
kmp_tasking_mode_t __kmp_tasking_mode
char * __kmp_affinity_format
int __kmp_dflt_blocktime
volatile kmp_info_t * __kmp_thread_pool
void __kmp_internal_end_atexit(void)
volatile int __kmp_init_gtid
omp_allocator_handle_t __kmp_def_allocator
static void __kmp_resume_if_hard_paused()
Definition kmp.h:4554
size_t __kmp_stksize
int __kmp_env_checks
#define get__max_active_levels(xthread)
Definition kmp.h:2424
kmp_nested_proc_bind_t __kmp_nested_proc_bind
void __kmp_free_implicit_task(kmp_info_t *this_thr)
void __kmp_hidden_helper_main_thread_release()
fork_context_e
Tell the fork call which compiler generated the fork call, and therefore how to deal with the call.
Definition kmp.h:4055
@ fork_context_gnu
Called from GNU generated code, so must not invoke the microtask internally.
Definition kmp.h:4056
@ fork_context_intel
Called from Intel generated code.
Definition kmp.h:4058
@ fork_context_last
Definition kmp.h:4059
void __kmp_suspend_initialize(void)
kmp_nested_nthreads_t __kmp_nested_nth
int __kmp_max_nth
omp_allocator_handle_t const omp_default_mem_alloc
int __kmp_chunk
#define KMP_GTID_SHUTDOWN
Definition kmp.h:1000
@ flag_unset
Definition kmp.h:2149
void __kmp_internal_end_dtor(void)
volatile int __kmp_all_nth
#define set__nproc(xthread, xval)
Definition kmp.h:2415
#define KMP_MIN_NTH
Definition kmp.h:1181
int __kmp_is_address_mapped(void *addr)
kmp_lock_t __kmp_global_lock
@ severity_warning
Definition kmp.h:4628
@ severity_fatal
Definition kmp.h:4629
void __kmpc_destroy_allocator(int gtid, omp_allocator_handle_t al)
union KMP_ALIGN_CACHE kmp_root kmp_root_t
int __kmp_adjust_gtid_mode
int __kmp_env_blocktime
#define __kmp_entry_gtid()
Definition kmp.h:3594
kmp_old_threads_list_t * __kmp_old_threads_list
struct kmp_internal_control kmp_internal_control_t
#define KMP_GTID_MONITOR
Definition kmp.h:1001
volatile int __kmp_init_common
kmp_info_t __kmp_monitor
static int __kmp_tid_from_gtid(int gtid)
Definition kmp.h:3612
static bool KMP_UBER_GTID(int gtid)
Definition kmp.h:3605
int __kmp_display_env
kmp_int32 __kmp_use_yield_exp_set
int __kmp_tp_cached
volatile int __kmp_init_hidden_helper
#define KMP_DEBUG_ASSERT_TASKTEAM_INVARIANT(team, thr)
Definition kmp.h:4147
int __kmp_gtid_get_specific(void)
volatile int __kmp_init_middle
void __kmp_hidden_helper_threads_deinitz_wait()
void __kmpc_error(ident_t *loc, int severity, const char *message)
static kmp_sched_t __kmp_sched_without_mods(kmp_sched_t kind)
Definition kmp.h:471
@ cancel_noreq
Definition kmp.h:970
int __kmp_reserve_warn
#define KMP_CHECK_UPDATE(a, b)
Definition kmp.h:2374
int __kmp_storage_map_verbose
int __kmp_allThreadsSpecified
enum sched_type __kmp_static
#define KMP_INITIAL_GTID(gtid)
Definition kmp.h:1337
volatile int __kmp_nth
int PACKED_REDUCTION_METHOD_T
Definition kmp.h:568
std::atomic< int > __kmp_thread_pool_active_nth
#define KMP_MASTER_TID(tid)
Definition kmp.h:1332
int __kmp_duplicate_library_ok
volatile int __kmp_need_register_serial
kmp_bootstrap_lock_t __kmp_forkjoin_lock
struct kmp_cg_root kmp_cg_root_t
kmp_uint32 __kmp_barrier_release_branch_bits[bs_last_barrier]
static kmp_info_t * __kmp_entry_thread()
Definition kmp.h:3724
void __kmp_init_memkind()
void __kmp_hidden_helper_main_thread_wait()
#define KMP_GEN_TEAM_ID()
Definition kmp.h:3674
void __kmp_init_implicit_task(ident_t *loc_ref, kmp_info_t *this_thr, kmp_team_t *team, int tid, int set_curr_task)
kmp_int32 __kmp_default_device
#define get__sched_2(xteam, xtid)
Definition kmp.h:2394
void __kmp_cleanup_threadprivate_caches()
static void copy_icvs(kmp_internal_control_t *dst, kmp_internal_control_t *src)
Definition kmp.h:2205
kmp_bootstrap_lock_t __kmp_exit_lock
kmp_info_t ** __kmp_threads
void __kmp_hidden_helper_initz_release()
enum sched_type __kmp_sched
#define KMP_BARRIER_PARENT_FLAG
Definition kmp.h:2132
void __kmp_suspend_uninitialize_thread(kmp_info_t *th)
void __kmp_finalize_bget(kmp_info_t *th)
#define KMP_BARRIER_SWITCH_TO_OWN_FLAG
Definition kmp.h:2134
static void __kmp_reset_root_init_mask(int gtid)
Definition kmp.h:3950
kmp_uint32 __kmp_barrier_gather_bb_dflt
kmp_uint32 __kmp_barrier_release_bb_dflt
int __kmp_dispatch_num_buffers
#define SCHEDULE_WITHOUT_MODIFIERS(s)
Definition kmp.h:433
union kmp_team kmp_team_t
Definition kmp.h:241
int __kmp_nesting_mode
#define set__max_active_levels(xthread, xval)
Definition kmp.h:2421
#define __kmp_get_team_num_threads(gtid)
Definition kmp.h:3602
#define KMP_MIN_MALLOC_ARGV_ENTRIES
Definition kmp.h:3105
#define KMP_MASTER_GTID(gtid)
Definition kmp.h:1335
void __kmp_lock_suspend_mx(kmp_info_t *th)
bool __kmp_detect_tmp()
int __kmp_nesting_mode_nlevels
int __kmp_nteams
int __kmp_storage_map
#define KMP_YIELD(cond)
Definition kmp.h:1603
int(* launch_t)(int gtid)
Definition kmp.h:3102
void __kmp_create_worker(int gtid, kmp_info_t *th, size_t stack_size)
int * __kmp_nesting_nth_level
volatile int __kmp_init_parallel
int __kmp_init_counter
int __kmp_sys_max_nth
union kmp_barrier_union kmp_balign_t
Definition kmp.h:2250
kmp_root_t ** __kmp_root
kmp_int32 __kmp_enable_hidden_helper
#define KMP_DEFAULT_BLOCKTIME
Definition kmp.h:1249
#define set__blocktime_team(xteam, xtid, xval)
Definition kmp.h:2397
#define __kmp_allocate(size)
Definition kmp.h:3747
enum kmp_sched kmp_sched_t
#define TRUE
Definition kmp.h:1341
enum library_type __kmp_library
#define FALSE
Definition kmp.h:1340
int __kmp_tp_capacity
int __kmp_settings
@ tskm_immediate_exec
Definition kmp.h:2438
kmp_info_t * __kmp_thread_pool_insert_pt
#define UNLIKELY(x)
Definition kmp.h:140
int __kmp_env_consistency_check
#define bs_reduction_barrier
Definition kmp.h:2164
void __kmp_runtime_destroy(void)
union KMP_ALIGN_CACHE kmp_desc kmp_desc_t
static void __kmp_sched_apply_mods_intkind(kmp_sched_t kind, enum sched_type *internal_kind)
Definition kmp.h:462
volatile int __kmp_hidden_helper_team_done
static void __kmp_sched_apply_mods_stdkind(kmp_sched_t *kind, enum sched_type internal_kind)
Definition kmp.h:453
union kmp_barrier_team_union kmp_balign_team_t
Definition kmp.h:2269
int __kmp_hot_teams_mode
std::atomic< kmp_int32 > __kmp_unexecuted_hidden_helper_tasks
#define KMP_INIT_BARRIER_STATE
Definition kmp.h:2112
size_t __kmp_sys_min_stksize
@ kmp_sched_upper
Definition kmp.h:330
@ kmp_sched_lower
Definition kmp.h:318
@ kmp_sched_trapezoidal
Definition kmp.h:326
@ kmp_sched_upper_std
Definition kmp.h:324
@ kmp_sched_dynamic
Definition kmp.h:321
@ kmp_sched_auto
Definition kmp.h:323
@ kmp_sched_guided
Definition kmp.h:322
@ kmp_sched_lower_ext
Definition kmp.h:325
@ kmp_sched_default
Definition kmp.h:331
@ kmp_sched_static
Definition kmp.h:320
union kmp_info kmp_info_p
Definition kmp.h:245
#define set__bt_set_team(xteam, xtid, xval)
Definition kmp.h:2407
int __kmp_invoke_task_func(int gtid)
kmp_uint32 __kmp_barrier_gather_branch_bits[bs_last_barrier]
#define KMP_BARRIER_NOT_WAITING
Definition kmp.h:2129
int __kmp_task_max_nth
#define KMP_INTERNAL_FREE(p)
Definition kmp.h:104
int __kmp_threads_capacity
kmp_info_t ** __kmp_hidden_helper_threads
void __kmp_push_current_task_to_thread(kmp_info_t *this_thr, kmp_team_t *team, int tid)
int __kmp_foreign_tp
static int __kmp_gtid_from_tid(int tid, const kmp_team_t *team)
Definition kmp.h:3617
#define KMP_SAFE_TO_REAP
Definition kmp.h:2141
void __kmp_push_task_team_node(kmp_info_t *thread, kmp_team_t *team)
void __kmp_threadprivate_resize_cache(int newCapacity)
union kmp_r_sched kmp_r_sched_t
void __kmp_runtime_initialize(void)
@ bs_plain_barrier
Definition kmp.h:2153
@ bs_last_barrier
Definition kmp.h:2159
@ bs_forkjoin_barrier
Definition kmp.h:2155
volatile int __kmp_init_hidden_helper_threads
void __kmp_common_destroy_gtid(int gtid)
int __kmp_try_suspend_mx(kmp_info_t *th)
int __kmp_display_affinity
enum sched_type __kmp_guided
void __kmp_resume_32(int target_gtid, kmp_flag_32< C, S > *flag)
#define KMP_INLINE_ARGV_ENTRIES
Definition kmp.h:3121
#define __kmp_get_gtid()
Definition kmp.h:3593
#define SCHEDULE_GET_MODIFIERS(s)
Definition kmp.h:440
PACKED_REDUCTION_METHOD_T __kmp_force_reduction_method
int __kmp_avail_proc
#define __kmp_page_allocate(size)
Definition kmp.h:3748
#define SKIP_DIGITS(_x)
Definition kmp.h:270
void __kmp_initialize_bget(kmp_info_t *th)
int __kmp_teams_thread_limit
int __kmp_stkpadding
void __kmp_cleanup_hierarchy()
int __kmp_dflt_team_nth
void __kmp_pop_current_task_from_thread(kmp_info_t *this_thr)
void __kmp_gtid_set_specific(int gtid)
kmp_proc_bind_t
Definition kmp.h:930
@ proc_bind_false
Definition kmp.h:931
@ proc_bind_close
Definition kmp.h:934
@ proc_bind_primary
Definition kmp.h:933
@ proc_bind_spread
Definition kmp.h:935
@ proc_bind_default
Definition kmp.h:937
@ KMP_HW_CORE
Definition kmp.h:603
@ KMP_HW_LAST
Definition kmp.h:605
void __kmp_atomic_resume_64(int target_gtid, kmp_atomic_flag_64< C, S > *flag)
int __kmp_root_counter
static int __kmp_gtid_from_thread(const kmp_info_t *thr)
Definition kmp.h:3622
int __kmp_gtid_mode
#define KMP_MIN_BLOCKTIME
Definition kmp.h:1244
#define SCHEDULE_SET_MODIFIERS(s, m)
Definition kmp.h:443
void __kmp_suspend_initialize_thread(kmp_info_t *th)
library_type
Definition kmp.h:487
@ library_turnaround
Definition kmp.h:490
@ library_throughput
Definition kmp.h:491
@ library_serial
Definition kmp.h:489
volatile int __kmp_init_serial
@ empty_reduce_block
Definition kmp.h:522
@ critical_reduce_block
Definition kmp.h:519
@ tree_reduce_block
Definition kmp.h:521
@ reduction_method_not_defined
Definition kmp.h:518
@ atomic_reduce_block
Definition kmp.h:520
#define KMP_CHECK_UPDATE_SYNC(a, b)
Definition kmp.h:2377
int __kmp_invoke_microtask(microtask_t pkfn, int gtid, int npr, int argc, void *argv[])
kmp_int32 __kmp_hidden_helper_threads_num
#define KMP_MAX_ACTIVE_LEVELS_LIMIT
Definition kmp.h:1317
static void __kmp_type_convert(T1 src, T2 *dest)
Definition kmp.h:4879
#define SKIP_TOKEN(_x)
Definition kmp.h:275
void __kmp_fini_memkind()
struct kmp_taskdata kmp_taskdata_t
Definition kmp.h:242
kmp_bar_pat_e __kmp_barrier_release_pattern[bs_last_barrier]
void __kmp_reap_worker(kmp_info_t *th)
int __kmp_env_stksize
#define KMP_GTID_DNE
Definition kmp.h:999
@ bp_dist_bar
Definition kmp.h:2175
@ bp_hierarchical_bar
Definition kmp.h:2174
@ dynamic_thread_limit
Definition kmp.h:309
@ dynamic_default
Definition kmp.h:304
@ dynamic_random
Definition kmp.h:308
void __kmp_hidden_helper_threads_deinitz_release()
void __kmp_expand_host_name(char *buffer, size_t size)
union KMP_ALIGN_CACHE kmp_info kmp_info_t
enum sched_type __kmp_sch_map[]
int __kmp_tls_gtid_min
void __kmp_task_team_wait(kmp_info_t *this_thr, kmp_team_t *team, int wait=1)
#define __kmp_thread_free(th, ptr)
Definition kmp.h:3775
kmp_topology_t * __kmp_topology
static int __kmp_ncores
kmp_atomic_lock_t __kmp_atomic_lock_8c
kmp_atomic_lock_t __kmp_atomic_lock_8r
kmp_atomic_lock_t __kmp_atomic_lock_4i
KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 kmp_int16
kmp_atomic_lock_t __kmp_atomic_lock_20c
kmp_atomic_lock_t __kmp_atomic_lock_16c
KMP_ARCH_X86 short
kmp_atomic_lock_t __kmp_atomic_lock_2i
kmp_atomic_lock_t __kmp_atomic_lock_32c
kmp_atomic_lock_t __kmp_atomic_lock_8i
kmp_atomic_lock_t __kmp_atomic_lock
kmp_atomic_lock_t __kmp_atomic_lock_10r
KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 kmp_int8
kmp_atomic_lock_t __kmp_atomic_lock_1i
kmp_atomic_lock_t __kmp_atomic_lock_16r
KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86<<, 2i, 1, KMP_ARCH_X86) ATOMIC_CMPXCHG(fixed2, shr, kmp_int16, 16, > KMP_ARCH_X86 KMP_ARCH_X86 kmp_uint32
kmp_atomic_lock_t __kmp_atomic_lock_4r
static void __kmp_init_atomic_lock(kmp_atomic_lock_t *lck)
Definition kmp_atomic.h:405
void __kmp_print_structure(void)
#define MAX_MESSAGE
void __kmp_dump_debug_buffer(void)
Definition kmp_debug.cpp:84
#define KA_TRACE(d, x)
Definition kmp_debug.h:157
#define KMP_ASSERT(cond)
Definition kmp_debug.h:59
#define KMP_BUILD_ASSERT(expr)
Definition kmp_debug.h:26
#define KF_TRACE(d, x)
Definition kmp_debug.h:162
#define KD_TRACE(d, x)
Definition kmp_debug.h:160
#define KC_TRACE(d, x)
Definition kmp_debug.h:159
#define KMP_DEBUG_ASSERT(cond)
Definition kmp_debug.h:61
#define KB_TRACE(d, x)
Definition kmp_debug.h:158
#define KMP_ASSERT2(cond, msg)
Definition kmp_debug.h:60
unsigned long long kmp_uint64
void __kmp_device_env_reset(void)
kmp_hier_sched_env_t __kmp_hier_scheds
void __kmp_dispatch_free_hierarchies(kmp_team_t *team)
void __kmp_env_free(char const **value)
char * __kmp_env_get(char const *name)
void __kmp_env_set(char const *name, char const *value, int overwrite)
void __kmp_env_unset(char const *name)
void __kmp_push_sync(int gtid, enum cons_type ct, ident_t const *ident, kmp_user_lock_p lck)
void __kmp_push_parallel(int gtid, ident_t const *ident)
void __kmp_check_workshare(int gtid, enum cons_type ct, ident_t const *ident)
void __kmp_push_workshare(int gtid, enum cons_type ct, ident_t const *ident)
enum cons_type __kmp_pop_workshare(int gtid, enum cons_type ct, ident_t const *ident)
void __kmp_pop_sync(int gtid, enum cons_type ct, ident_t const *ident)
struct cons_header * __kmp_allocate_cons_stack(int gtid)
void __kmp_pop_parallel(int gtid, ident_t const *ident)
void __kmp_free_cons_stack(void *ptr)
static volatile kmp_i18n_cat_status_t status
Definition kmp_i18n.cpp:48
kmp_msg_t __kmp_msg_null
Definition kmp_i18n.cpp:36
static void __kmp_msg(kmp_msg_severity_t severity, kmp_msg_t message, va_list ap)
Definition kmp_i18n.cpp:787
void __kmp_i18n_dump_catalog(kmp_str_buf_t *buffer)
Definition kmp_i18n.cpp:593
void __kmp_fatal(kmp_msg_t message,...)
Definition kmp_i18n.cpp:873
#define KMP_INFORM(...)
Definition kmp_i18n.h:142
#define KMP_WARNING(...)
Definition kmp_i18n.h:144
#define KMP_MSG(...)
Definition kmp_i18n.h:121
@ kmp_ms_warning
Definition kmp_i18n.h:130
#define KMP_I18N_STR(id)
Definition kmp_i18n.h:46
#define KMP_FATAL(...)
Definition kmp_i18n.h:146
#define KMP_HNT(...)
Definition kmp_i18n.h:122
void __kmp_i18n_catclose()
#define KMP_ERR
Definition kmp_i18n.h:125
kmp_bootstrap_lock_t __kmp_stdio_lock
Definition kmp_io.cpp:41
void __kmp_fprintf(enum kmp_io stream, char const *format,...)
Definition kmp_io.cpp:206
void __kmp_vprintf(enum kmp_io out_stream, char const *format, va_list ap)
Definition kmp_io.cpp:115
void __kmp_printf(char const *format,...)
Definition kmp_io.cpp:186
void __kmp_printf_no_lock(char const *format,...)
Definition kmp_io.cpp:197
@ kmp_out
Definition kmp_io.h:22
@ kmp_err
Definition kmp_io.h:22
void __kmp_close_console(void)
#define USE_ITT_BUILD_ARG(x)
Definition kmp_itt.h:346
void __kmp_cleanup_user_locks(void)
void __kmp_validate_locks(void)
Definition kmp_lock.cpp:43
static void __kmp_release_bootstrap_lock(kmp_bootstrap_lock_t *lck)
Definition kmp_lock.h:535
static int __kmp_acquire_lock(kmp_lock_t *lck, kmp_int32 gtid)
Definition kmp_lock.h:559
static void __kmp_init_lock(kmp_lock_t *lck)
Definition kmp_lock.h:571
static int __kmp_acquire_bootstrap_lock(kmp_bootstrap_lock_t *lck)
Definition kmp_lock.h:527
static void __kmp_release_lock(kmp_lock_t *lck, kmp_int32 gtid)
Definition kmp_lock.h:567
static void __kmp_init_bootstrap_lock(kmp_bootstrap_lock_t *lck)
Definition kmp_lock.h:539
#define TCW_PTR(a, b)
Definition kmp_os.h:1171
void(* microtask_t)(int *gtid, int *npr,...)
Definition kmp_os.h:1189
#define kmp_va_deref(ap)
Definition kmp_os.h:231
#define KMP_WAIT
Definition kmp_os.h:1197
#define TCW_SYNC_PTR(a, b)
Definition kmp_os.h:1173
#define KMP_ATOMIC_ST_REL(p, v)
Definition kmp_os.h:1265
double kmp_real64
Definition kmp_os.h:201
long kmp_intptr_t
Definition kmp_os.h:205
#define TCR_SYNC_PTR(a)
Definition kmp_os.h:1172
#define TCR_PTR(a)
Definition kmp_os.h:1170
#define KMP_UINTPTR_SPEC
Definition kmp_os.h:208
@ kmp_warnings_off
Definition kmp_os.h:1245
#define RCAST(type, var)
Definition kmp_os.h:294
#define KMP_THREAD_LOCAL
Definition kmp_os.h:396
#define CACHE_LINE
Definition kmp_os.h:342
#define KMP_CACHE_PREFETCH(ADDR)
Definition kmp_os.h:350
#define KMP_ATOMIC_LD_ACQ(p)
Definition kmp_os.h:1263
#define TCW_SYNC_4(a, b)
Definition kmp_os.h:1150
#define VOLATILE_CAST(x)
Definition kmp_os.h:1194
#define CCAST(type, var)
Definition kmp_os.h:293
#define KMP_MB()
Definition kmp_os.h:1070
#define KMP_EQ
Definition kmp_os.h:1199
bool __kmp_atomic_compare_store_acq(std::atomic< T > *p, T expected, T desired)
Definition kmp_os.h:1286
#define TCR_4(a)
Definition kmp_os.h:1141
#define KMP_FALLTHROUGH()
Definition kmp_os.h:366
#define KMP_ATOMIC_DEC(p)
Definition kmp_os.h:1274
#define KMP_GET_PAGE_SIZE()
Definition kmp_os.h:324
#define KMP_ATOMIC_LD_RLX(p)
Definition kmp_os.h:1264
#define KMP_MFENCE()
Definition kmp_os.h:1103
#define KMP_COMPARE_AND_STORE_ACQ32(p, cv, sv)
Definition kmp_os.h:818
#define TCW_4(a, b)
Definition kmp_os.h:1142
#define KMP_WEAK_ATTRIBUTE_EXTERNAL
Definition kmp_os.h:403
va_list kmp_va_list
Definition kmp_os.h:230
#define KMP_DLSYM(name)
Definition kmp_os.h:1306
#define KMP_ATOMIC_INC(p)
Definition kmp_os.h:1273
#define KMP_COMPARE_AND_STORE_PTR(p, cv, sv)
Definition kmp_os.h:824
int __kmp_pause_resource(kmp_pause_status_t level)
void __kmp_warn(char const *format,...)
void __kmp_set_schedule(int gtid, kmp_sched_t kind, int chunk)
static void __kmp_initialize_team(kmp_team_t *team, int new_nproc, kmp_internal_control_t *new_icvs, ident_t *loc)
static void __kmp_fini_allocator()
void __kmp_soft_pause()
static void __kmp_init_allocator()
void __kmp_aux_set_defaults(char const *str, size_t len)
static int __kmp_free_hot_teams(kmp_root_t *root, kmp_info_t *thr, int level, const int max_level)
static kmp_team_t * __kmp_aux_get_team_info(int &teams_serialized)
static int __kmp_expand_threads(int nNeed)
void __kmp_teams_master(int gtid)
static void __kmp_itthash_clean(kmp_info_t *th)
#define propagateFPControl(x)
void __kmp_itt_init_ittlib()
void __kmp_infinite_loop(void)
void __kmp_push_num_teams_51(ident_t *id, int gtid, int num_teams_lb, int num_teams_ub, int num_threads)
int __kmp_aux_get_num_teams()
kmp_team_t * __kmp_allocate_team(kmp_root_t *root, int new_nproc, int max_nproc, kmp_proc_bind_t new_proc_bind, kmp_internal_control_t *new_icvs, int argc, kmp_info_t *master)
kmp_info_t * __kmp_allocate_thread(kmp_root_t *root, kmp_team_t *team, int new_tid)
void __kmp_run_before_invoked_task(int gtid, int tid, kmp_info_t *this_thr, kmp_team_t *team)
static long __kmp_registration_flag
int __kmp_get_max_active_levels(int gtid)
void __kmp_aux_set_library(enum library_type arg)
void __kmp_print_storage_map_gtid(int gtid, void *p1, void *p2, size_t size, char const *format,...)
void __kmp_free_team(kmp_root_t *root, kmp_team_t *team, kmp_info_t *master)
unsigned short __kmp_get_random(kmp_info_t *thread)
int __kmp_register_root(int initial_thread)
static void __kmp_internal_end(void)
void __kmp_set_max_active_levels(int gtid, int max_active_levels)
void __kmp_abort_thread(void)
void __kmp_setup_icv_copy(kmp_team_t *team, int new_nproc, kmp_internal_control_t *new_icvs, ident_t *loc)
void __kmp_internal_end_atexit(void)
static void __kmp_fork_team_threads(kmp_root_t *root, kmp_team_t *team, kmp_info_t *master_th, int master_gtid, int fork_teams_workers)
void __kmp_push_proc_bind(ident_t *id, int gtid, kmp_proc_bind_t proc_bind)
kmp_team_t * __kmp_reap_team(kmp_team_t *team)
void __kmp_exit_single(int gtid)
void __kmp_check_stack_overlap(kmp_info_t *th)
void __kmp_push_num_teams(ident_t *id, int gtid, int num_teams, int num_threads)
int __kmp_get_team_size(int gtid, int level)
static void __kmp_allocate_team_arrays(kmp_team_t *team, int max_nth)
static void __kmp_do_middle_initialize(void)
int __kmp_get_max_teams(void)
static void __kmp_free_team_arrays(kmp_team_t *team)
static void __kmp_initialize_root(kmp_root_t *root)
static void __kmp_reinitialize_team(kmp_team_t *team, kmp_internal_control_t *new_icvs, ident_t *loc)
int __kmp_fork_call(ident_t *loc, int gtid, enum fork_context_e call_context, kmp_int32 argc, microtask_t microtask, launch_t invoker, kmp_va_list ap)
void __kmp_parallel_dxo(int *gtid_ref, int *cid_ref, ident_t *loc_ref)
void * __kmp_launch_thread(kmp_info_t *this_thr)
void __kmp_set_teams_thread_limit(int limit)
static int __kmp_serial_fork_call(ident_t *loc, int gtid, enum fork_context_e call_context, kmp_int32 argc, microtask_t microtask, launch_t invoker, kmp_info_t *master_th, kmp_team_t *parent_team, kmp_va_list ap)
void __kmp_join_barrier(int gtid)
static kmp_internal_control_t __kmp_get_x_global_icvs(const kmp_team_t *team)
void __kmp_init_random(kmp_info_t *thread)
static void __kmp_push_thread_limit(kmp_info_t *thr, int num_teams, int num_threads)
void __kmp_push_num_threads(ident_t *id, int gtid, int num_threads)
void __kmp_user_set_library(enum library_type arg)
#define updateHWFPControl(x)
#define FAST_REDUCTION_ATOMIC_METHOD_GENERATED
void __kmp_internal_end_dest(void *specific_gtid)
int __kmp_aux_get_team_num()
void __kmp_set_num_threads(int new_nth, int gtid)
void __kmp_internal_end_thread(int gtid_req)
static bool __kmp_is_fork_in_teams(kmp_info_t *master_th, microtask_t microtask, int level, int teams_level, kmp_va_list ap)
PACKED_REDUCTION_METHOD_T __kmp_determine_reduction_method(ident_t *loc, kmp_int32 global_tid, kmp_int32 num_vars, size_t reduce_size, void *reduce_data, void(*reduce_func)(void *lhs_data, void *rhs_data), kmp_critical_name *lck)
void __kmp_hidden_helper_threads_initz_routine()
int __kmp_enter_single(int gtid, ident_t *id_ref, int push_ws)
static void __kmp_initialize_info(kmp_info_t *, kmp_team_t *, int tid, int gtid)
void __kmp_internal_join(ident_t *id, int gtid, kmp_team_t *team)
void __kmp_join_call(ident_t *loc, int gtid, int exit_teams)
static int __kmp_reset_root(int gtid, kmp_root_t *root)
int __kmp_get_ancestor_thread_num(int gtid, int level)
void __kmp_itt_fini_ittlib()
void __kmp_omp_display_env(int verbose)
void __kmp_reset_stats()
void __kmp_middle_initialize(void)
void __kmp_unregister_root_current_thread(int gtid)
int __kmp_debugging
static void __kmp_reap_thread(kmp_info_t *thread, int is_root)
static const unsigned __kmp_primes[]
int __kmp_get_teams_thread_limit(void)
#define FAST_REDUCTION_TREE_METHOD_GENERATED
void __kmp_parallel_deo(int *gtid_ref, int *cid_ref, ident_t *loc_ref)
kmp_r_sched_t __kmp_get_schedule_global()
void __kmp_run_after_invoked_task(int gtid, int tid, kmp_info_t *this_thr, kmp_team_t *team)
static kmp_internal_control_t __kmp_get_global_icvs(void)
void __kmp_parallel_initialize(void)
void __kmp_set_nesting_mode_threads()
void __kmp_unregister_library(void)
char const __kmp_version_omp_api[]
static char * __kmp_registration_str
int __kmp_ignore_mppbeg(void)
void __kmp_internal_fork(ident_t *id, int gtid, kmp_team_t *team)
void __kmp_aux_set_stacksize(size_t arg)
void __kmp_internal_end_library(int gtid_req)
size_t __kmp_aux_capture_affinity(int gtid, const char *format, kmp_str_buf_t *buffer)
void __kmp_hard_pause()
void __kmp_resize_dist_barrier(kmp_team_t *team, int old_nthreads, int new_nthreads)
int __kmp_omp_debug_struct_info
static void __kmp_print_thread_storage_map(kmp_info_t *thr, int gtid)
void __kmp_aux_display_affinity(int gtid, const char *format)
void __kmp_init_nesting_mode()
void __kmp_register_library_startup(void)
void __kmp_free_thread(kmp_info_t *this_th)
int __kmp_invoke_task_func(int gtid)
void __kmp_get_schedule(int gtid, kmp_sched_t *kind, int *chunk)
void __kmp_set_strict_num_threads(ident_t *loc, int gtid, int sev, const char *msg)
void __kmp_abort_process()
static const kmp_affinity_format_field_t __kmp_affinity_format_table[]
void __kmp_set_num_teams(int num_teams)
static void __kmp_alloc_argv_entries(int argc, kmp_team_t *team, int realloc)
void __kmp_save_internal_controls(kmp_info_t *thread)
int __kmp_invoke_teams_master(int gtid)
void __kmp_hidden_helper_initialize()
void __kmp_add_threads_to_team(kmp_team_t *team, int new_nthreads)
void __kmp_push_num_threads_list(ident_t *id, int gtid, kmp_uint32 list_length, int *num_threads_list)
static void __kmp_reallocate_team_arrays(kmp_team_t *team, int max_nth)
static int __kmp_reserve_threads(kmp_root_t *root, kmp_team_t *parent_team, int master_tid, int set_nthreads, int enter_teams)
void __kmp_serial_initialize(void)
static bool __kmp_is_entering_teams(int active_level, int level, int teams_level, kmp_va_list ap)
void __kmp_resume_if_soft_paused()
void __kmp_serialized_parallel(ident_t *loc, kmp_int32 global_tid)
int __kmp_get_global_thread_id()
void __kmp_internal_begin(void)
static char * __kmp_reg_status_name()
static void __kmp_print_team_storage_map(const char *header, kmp_team_t *team, int team_id, int num_thr)
static void __kmp_do_serial_initialize(void)
void __kmp_fork_barrier(int gtid, int tid)
int __kmp_get_global_thread_id_reg()
int __kmp_ignore_mppend(void)
static kmp_nested_nthreads_t * __kmp_override_nested_nth(kmp_info_t *thr, int level)
kmp_int32 __kmp_get_reduce_method(void)
static int __kmp_aux_capture_affinity_field(int gtid, const kmp_info_t *th, const char **ptr, kmp_str_buf_t *field_buffer)
void __kmp_cleanup(void)
static int __kmp_fork_in_teams(ident_t *loc, int gtid, kmp_team_t *parent_team, kmp_int32 argc, kmp_info_t *master_th, kmp_root_t *root, enum fork_context_e call_context, microtask_t microtask, launch_t invoker, int master_set_numthreads, int level, kmp_va_list ap)
void __kmp_aux_set_blocktime(int arg, kmp_info_t *thread, int tid)
#define KMP_ALLOCA
#define KMP_STRCPY_S(dst, bsz, src)
#define KMP_SNPRINTF
#define KMP_SSCANF
#define KMP_MEMCPY
#define KMP_STRLEN
void __kmp_env_print_2()
int __kmp_default_tp_capacity(int req_nproc, int max_nth, int all_threads_specified)
int __kmp_initial_threads_capacity(int req_nproc)
void __kmp_env_initialize(char const *string)
void __kmp_display_env_impl(int display_env, int display_env_verbose)
void __kmp_env_print()
void __kmp_stats_init(void)
void __kmp_stats_fini(void)
Functions for collecting statistics.
#define KMP_COUNT_VALUE(n, v)
Definition kmp_stats.h:1000
#define KMP_PUSH_PARTITIONED_TIMER(name)
Definition kmp_stats.h:1014
#define KMP_GET_THREAD_STATE()
Definition kmp_stats.h:1017
#define KMP_POP_PARTITIONED_TIMER()
Definition kmp_stats.h:1015
#define KMP_INIT_PARTITIONED_TIMERS(name)
Definition kmp_stats.h:1012
#define KMP_SET_THREAD_STATE_BLOCK(state_name)
Definition kmp_stats.h:1018
#define KMP_TIME_PARTITIONED_BLOCK(name)
Definition kmp_stats.h:1013
#define KMP_TIME_DEVELOPER_PARTITIONED_BLOCK(n)
Definition kmp_stats.h:1008
#define KMP_SET_THREAD_STATE(state_name)
Definition kmp_stats.h:1016
void __kmp_str_split(char *str, char delim, char **head, char **tail)
Definition kmp_str.cpp:571
void __kmp_str_buf_clear(kmp_str_buf_t *buffer)
Definition kmp_str.cpp:71
void __kmp_str_buf_free(kmp_str_buf_t *buffer)
Definition kmp_str.cpp:123
char * __kmp_str_format(char const *format,...)
Definition kmp_str.cpp:448
int __kmp_str_match_true(char const *data)
Definition kmp_str.cpp:552
void __kmp_str_buf_cat(kmp_str_buf_t *buffer, char const *str, size_t len)
Definition kmp_str.cpp:134
void __kmp_str_buf_catbuf(kmp_str_buf_t *dest, const kmp_str_buf_t *src)
Definition kmp_str.cpp:146
#define args
int __kmp_str_buf_print(kmp_str_buf_t *buffer, char const *format,...)
Definition kmp_str.cpp:221
int __kmp_str_match_false(char const *data)
Definition kmp_str.cpp:543
struct kmp_str_buf kmp_str_buf_t
Definition kmp_str.h:39
#define __kmp_str_buf_init(b)
Definition kmp_str.h:41
#define i
Definition kmp_stub.cpp:88
void __kmp_print_version_1(void)
void __kmp_print_version_2(void)
#define KMP_VERSION_PREFIX
Definition kmp_version.h:34
char const __kmp_version_alt_comp[]
char const __kmp_version_lock[]
static void __kmp_null_resume_wrapper(kmp_info_t *thr)
void microtask(int *global_tid, int *bound_tid)
int32_t kmp_int32
omp_lock_t lck
Definition omp_lock.c:7
static int ii
#define res
void ompt_fini()
ompt_callbacks_active_t ompt_enabled
void ompt_pre_init()
void ompt_post_init()
ompt_callbacks_internal_t ompt_callbacks
#define OMPT_INVOKER(x)
struct ompt_lw_taskteam_s ompt_lw_taskteam_t
#define OMPT_GET_FRAME_ADDRESS(level)
void __ompt_lw_taskteam_init(ompt_lw_taskteam_t *lwt, kmp_info_t *thr, int gtid, ompt_data_t *ompt_pid, void *codeptr)
int __ompt_get_task_info_internal(int ancestor_level, int *type, ompt_data_t **task_data, ompt_frame_t **task_frame, ompt_data_t **parallel_data, int *thread_num)
void __ompt_lw_taskteam_link(ompt_lw_taskteam_t *lwt, kmp_info_t *thr, int on_heap, bool always)
ompt_task_info_t * __ompt_get_task_info_object(int depth)
void __ompt_team_assign_id(kmp_team_t *team, ompt_data_t ompt_pid)
void __ompt_lw_taskteam_unlink(kmp_info_t *thr)
ompt_data_t * __ompt_get_thread_data_internal()
static id loc
volatile int flag
__attribute__((noinline))
kmp_int32 tt_found_proxy_tasks
Definition kmp.h:2861
kmp_int32 tt_hidden_helper_task_encountered
Definition kmp.h:2866
kmp_info_p * cg_root
Definition kmp.h:2921
kmp_int32 cg_nthreads
Definition kmp.h:2925
kmp_int32 cg_thread_limit
Definition kmp.h:2924
struct kmp_cg_root * up
Definition kmp.h:2926
void(* th_dxo_fcn)(int *gtid, int *cid, ident_t *)
Definition kmp.h:2093
kmp_int32 th_doacross_buf_idx
Definition kmp.h:2100
dispatch_private_info_t * th_dispatch_pr_current
Definition kmp.h:2096
kmp_uint32 th_disp_index
Definition kmp.h:2099
dispatch_private_info_t * th_disp_buffer
Definition kmp.h:2098
void(* th_deo_fcn)(int *gtid, int *cid, ident_t *)
Definition kmp.h:2091
dispatch_shared_info_t * th_dispatch_sh_current
Definition kmp.h:2095
kmp_team_p * hot_team
Definition kmp.h:2901
kmp_int32 hot_team_nth
Definition kmp.h:2902
kmp_proc_bind_t proc_bind
Definition kmp.h:2200
kmp_r_sched_t sched
Definition kmp.h:2199
struct kmp_internal_control * next
Definition kmp.h:2202
int serial_nesting_level
Definition kmp.h:2183
struct kmp_old_threads_list_t * next
Definition kmp.h:3292
kmp_info_t ** threads
Definition kmp.h:3291
char * str
Definition kmp_str.h:34
ompt_task_info_t ompt_task_info
ompt_data_t task_data
ompt_frame_t frame
kmp_bstate_t bb
Definition kmp.h:2247
kmp_base_info_t th
Definition kmp.h:3081
int chunk
Definition kmp.h:479
enum sched_type r_sched_type
Definition kmp.h:478
kmp_int64 sched
Definition kmp.h:481
kmp_base_task_team_t tt
Definition kmp.h:2877
kmp_base_team_t t
Definition kmp.h:3227
void __kmp_reap_monitor(kmp_info_t *th)
void __kmp_register_atfork(void)
void __kmp_free_handle(kmp_thread_t tHandle)
int __kmp_get_load_balance(int max)
int __kmp_still_running(kmp_info_t *th)
void __kmp_initialize_system_tick(void)
int __kmp_is_thread_alive(kmp_info_t *th, DWORD *exit_val)