LLVM OpenMP
kmp_affinity.h
Go to the documentation of this file.
1/*
2 * kmp_affinity.h -- header for affinity management
3 */
4
5//===----------------------------------------------------------------------===//
6//
7// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
8// See https://llvm.org/LICENSE.txt for license information.
9// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
10//
11//===----------------------------------------------------------------------===//
12
13#ifndef KMP_AFFINITY_H
14#define KMP_AFFINITY_H
15
16#include "kmp.h"
17#include "kmp_os.h"
18#include <limits>
19
20#if KMP_AFFINITY_SUPPORTED
21#if KMP_HWLOC_ENABLED
22class KMPHwlocAffinity : public KMPAffinity {
23public:
24 class Mask : public KMPAffinity::Mask {
25 hwloc_cpuset_t mask;
26
27 public:
28 Mask() {
29 mask = hwloc_bitmap_alloc();
30 this->zero();
31 }
32 Mask(const Mask &other) = delete;
33 Mask &operator=(const Mask &other) = delete;
34 ~Mask() { hwloc_bitmap_free(mask); }
35 void set(int i) override { hwloc_bitmap_set(mask, i); }
36 bool is_set(int i) const override { return hwloc_bitmap_isset(mask, i); }
37 void clear(int i) override { hwloc_bitmap_clr(mask, i); }
38 void zero() override { hwloc_bitmap_zero(mask); }
39 bool empty() const override { return hwloc_bitmap_iszero(mask); }
40 void copy(const KMPAffinity::Mask *src) override {
41 const Mask *convert = static_cast<const Mask *>(src);
42 hwloc_bitmap_copy(mask, convert->mask);
43 }
44 void bitwise_and(const KMPAffinity::Mask *rhs) override {
45 const Mask *convert = static_cast<const Mask *>(rhs);
46 hwloc_bitmap_and(mask, mask, convert->mask);
47 }
48 void bitwise_or(const KMPAffinity::Mask *rhs) override {
49 const Mask *convert = static_cast<const Mask *>(rhs);
50 hwloc_bitmap_or(mask, mask, convert->mask);
51 }
52 void bitwise_not() override { hwloc_bitmap_not(mask, mask); }
53 bool is_equal(const KMPAffinity::Mask *rhs) const override {
54 const Mask *convert = static_cast<const Mask *>(rhs);
55 return hwloc_bitmap_isequal(mask, convert->mask);
56 }
57 int begin() const override { return hwloc_bitmap_first(mask); }
58 int end() const override { return -1; }
59 int next(int previous) const override {
60 return hwloc_bitmap_next(mask, previous);
61 }
62 int get_system_affinity(bool abort_on_error) override {
63 KMP_ASSERT2(KMP_AFFINITY_CAPABLE(),
64 "Illegal get affinity operation when not capable");
65 long retval =
66 hwloc_get_cpubind(__kmp_hwloc_topology, mask, HWLOC_CPUBIND_THREAD);
67 if (retval >= 0) {
68 return 0;
69 }
70 int error = errno;
71 if (abort_on_error) {
72 __kmp_fatal(KMP_MSG(FunctionError, "hwloc_get_cpubind()"),
73 KMP_ERR(error), __kmp_msg_null);
74 }
75 return error;
76 }
77 int set_system_affinity(bool abort_on_error) const override {
78 KMP_ASSERT2(KMP_AFFINITY_CAPABLE(),
79 "Illegal set affinity operation when not capable");
80 long retval =
81 hwloc_set_cpubind(__kmp_hwloc_topology, mask, HWLOC_CPUBIND_THREAD);
82 if (retval >= 0) {
83 return 0;
84 }
85 int error = errno;
86 if (abort_on_error) {
87 __kmp_fatal(KMP_MSG(FunctionError, "hwloc_set_cpubind()"),
88 KMP_ERR(error), __kmp_msg_null);
89 }
90 return error;
91 }
92#if KMP_OS_WINDOWS
93 int set_process_affinity(bool abort_on_error) const override {
94 KMP_ASSERT2(KMP_AFFINITY_CAPABLE(),
95 "Illegal set process affinity operation when not capable");
96 int error = 0;
97 const hwloc_topology_support *support =
98 hwloc_topology_get_support(__kmp_hwloc_topology);
99 if (support->cpubind->set_proc_cpubind) {
100 int retval;
101 retval = hwloc_set_cpubind(__kmp_hwloc_topology, mask,
102 HWLOC_CPUBIND_PROCESS);
103 if (retval >= 0)
104 return 0;
105 error = errno;
106 if (abort_on_error)
107 __kmp_fatal(KMP_MSG(FunctionError, "hwloc_set_cpubind()"),
108 KMP_ERR(error), __kmp_msg_null);
109 }
110 return error;
111 }
112#endif // KMP_OS_WINDOWS
113 int get_proc_group() const override {
114 int group = -1;
115#if KMP_OS_WINDOWS
116 if (__kmp_num_proc_groups == 1) {
117 return 1;
118 }
119 for (int i = 0; i < __kmp_num_proc_groups; i++) {
120 // On windows, the long type is always 32 bits
121 unsigned long first_32_bits = hwloc_bitmap_to_ith_ulong(mask, i * 2);
122 unsigned long second_32_bits =
123 hwloc_bitmap_to_ith_ulong(mask, i * 2 + 1);
124 if (first_32_bits == 0 && second_32_bits == 0) {
125 continue;
126 }
127 if (group >= 0) {
128 return -1;
129 }
130 group = i;
131 }
132#endif /* KMP_OS_WINDOWS */
133 return group;
134 }
135 };
136 void determine_capable(const char *var) override {
137 const hwloc_topology_support *topology_support;
138 if (__kmp_hwloc_topology == NULL) {
139 if (hwloc_topology_init(&__kmp_hwloc_topology) < 0) {
140 __kmp_hwloc_error = TRUE;
141 if (__kmp_affinity.flags.verbose) {
142 KMP_WARNING(AffHwlocErrorOccurred, var, "hwloc_topology_init()");
143 }
144 }
145 if (hwloc_topology_load(__kmp_hwloc_topology) < 0) {
146 __kmp_hwloc_error = TRUE;
147 if (__kmp_affinity.flags.verbose) {
148 KMP_WARNING(AffHwlocErrorOccurred, var, "hwloc_topology_load()");
149 }
150 }
151 }
152 topology_support = hwloc_topology_get_support(__kmp_hwloc_topology);
153 // Is the system capable of setting/getting this thread's affinity?
154 // Also, is topology discovery possible? (pu indicates ability to discover
155 // processing units). And finally, were there no errors when calling any
156 // hwloc_* API functions?
157 if (topology_support && topology_support->cpubind->set_thisthread_cpubind &&
158 topology_support->cpubind->get_thisthread_cpubind &&
159 topology_support->discovery->pu && !__kmp_hwloc_error) {
160 // enables affinity according to KMP_AFFINITY_CAPABLE() macro
161 KMP_AFFINITY_ENABLE(TRUE);
162 } else {
163 // indicate that hwloc didn't work and disable affinity
164 __kmp_hwloc_error = TRUE;
165 KMP_AFFINITY_DISABLE();
166 }
167 }
168 void bind_thread(int which) override {
169 KMP_ASSERT2(KMP_AFFINITY_CAPABLE(),
170 "Illegal set affinity operation when not capable");
171 KMPAffinity::Mask *mask;
172 KMP_CPU_ALLOC_ON_STACK(mask);
173 KMP_CPU_ZERO(mask);
174 KMP_CPU_SET(which, mask);
175 __kmp_set_system_affinity(mask, TRUE);
176 KMP_CPU_FREE_FROM_STACK(mask);
177 }
178 KMPAffinity::Mask *allocate_mask() override { return new Mask(); }
179 void deallocate_mask(KMPAffinity::Mask *m) override { delete m; }
180 KMPAffinity::Mask *allocate_mask_array(int num) override {
181 return new Mask[num];
182 }
183 void deallocate_mask_array(KMPAffinity::Mask *array) override {
184 Mask *hwloc_array = static_cast<Mask *>(array);
185 delete[] hwloc_array;
186 }
187 KMPAffinity::Mask *index_mask_array(KMPAffinity::Mask *array,
188 int index) override {
189 Mask *hwloc_array = static_cast<Mask *>(array);
190 return &(hwloc_array[index]);
191 }
192 api_type get_api_type() const override { return HWLOC; }
193};
194#endif /* KMP_HWLOC_ENABLED */
195
196#if KMP_OS_LINUX || KMP_OS_FREEBSD || KMP_OS_NETBSD || KMP_OS_DRAGONFLY || \
197 KMP_OS_AIX
198#if KMP_OS_LINUX
199/* On some of the older OS's that we build on, these constants aren't present
200 in <asm/unistd.h> #included from <sys.syscall.h>. They must be the same on
201 all systems of the same arch where they are defined, and they cannot change.
202 stone forever. */
203#include <sys/syscall.h>
204#if KMP_ARCH_X86 || KMP_ARCH_ARM
205#ifndef __NR_sched_setaffinity
206#define __NR_sched_setaffinity 241
207#elif __NR_sched_setaffinity != 241
208#error Wrong code for setaffinity system call.
209#endif /* __NR_sched_setaffinity */
210#ifndef __NR_sched_getaffinity
211#define __NR_sched_getaffinity 242
212#elif __NR_sched_getaffinity != 242
213#error Wrong code for getaffinity system call.
214#endif /* __NR_sched_getaffinity */
215#elif KMP_ARCH_AARCH64
216#ifndef __NR_sched_setaffinity
217#define __NR_sched_setaffinity 122
218#elif __NR_sched_setaffinity != 122
219#error Wrong code for setaffinity system call.
220#endif /* __NR_sched_setaffinity */
221#ifndef __NR_sched_getaffinity
222#define __NR_sched_getaffinity 123
223#elif __NR_sched_getaffinity != 123
224#error Wrong code for getaffinity system call.
225#endif /* __NR_sched_getaffinity */
226#elif KMP_ARCH_X86_64
227#ifndef __NR_sched_setaffinity
228#define __NR_sched_setaffinity 203
229#elif __NR_sched_setaffinity != 203
230#error Wrong code for setaffinity system call.
231#endif /* __NR_sched_setaffinity */
232#ifndef __NR_sched_getaffinity
233#define __NR_sched_getaffinity 204
234#elif __NR_sched_getaffinity != 204
235#error Wrong code for getaffinity system call.
236#endif /* __NR_sched_getaffinity */
237#elif KMP_ARCH_PPC64
238#ifndef __NR_sched_setaffinity
239#define __NR_sched_setaffinity 222
240#elif __NR_sched_setaffinity != 222
241#error Wrong code for setaffinity system call.
242#endif /* __NR_sched_setaffinity */
243#ifndef __NR_sched_getaffinity
244#define __NR_sched_getaffinity 223
245#elif __NR_sched_getaffinity != 223
246#error Wrong code for getaffinity system call.
247#endif /* __NR_sched_getaffinity */
248#elif KMP_ARCH_MIPS
249#ifndef __NR_sched_setaffinity
250#define __NR_sched_setaffinity 4239
251#elif __NR_sched_setaffinity != 4239
252#error Wrong code for setaffinity system call.
253#endif /* __NR_sched_setaffinity */
254#ifndef __NR_sched_getaffinity
255#define __NR_sched_getaffinity 4240
256#elif __NR_sched_getaffinity != 4240
257#error Wrong code for getaffinity system call.
258#endif /* __NR_sched_getaffinity */
259#elif KMP_ARCH_MIPS64
260#ifndef __NR_sched_setaffinity
261#define __NR_sched_setaffinity 5195
262#elif __NR_sched_setaffinity != 5195
263#error Wrong code for setaffinity system call.
264#endif /* __NR_sched_setaffinity */
265#ifndef __NR_sched_getaffinity
266#define __NR_sched_getaffinity 5196
267#elif __NR_sched_getaffinity != 5196
268#error Wrong code for getaffinity system call.
269#endif /* __NR_sched_getaffinity */
270#elif KMP_ARCH_LOONGARCH64
271#ifndef __NR_sched_setaffinity
272#define __NR_sched_setaffinity 122
273#elif __NR_sched_setaffinity != 122
274#error Wrong code for setaffinity system call.
275#endif /* __NR_sched_setaffinity */
276#ifndef __NR_sched_getaffinity
277#define __NR_sched_getaffinity 123
278#elif __NR_sched_getaffinity != 123
279#error Wrong code for getaffinity system call.
280#endif /* __NR_sched_getaffinity */
281#elif KMP_ARCH_RISCV64
282#ifndef __NR_sched_setaffinity
283#define __NR_sched_setaffinity 122
284#elif __NR_sched_setaffinity != 122
285#error Wrong code for setaffinity system call.
286#endif /* __NR_sched_setaffinity */
287#ifndef __NR_sched_getaffinity
288#define __NR_sched_getaffinity 123
289#elif __NR_sched_getaffinity != 123
290#error Wrong code for getaffinity system call.
291#endif /* __NR_sched_getaffinity */
292#elif KMP_ARCH_VE
293#ifndef __NR_sched_setaffinity
294#define __NR_sched_setaffinity 203
295#elif __NR_sched_setaffinity != 203
296#error Wrong code for setaffinity system call.
297#endif /* __NR_sched_setaffinity */
298#ifndef __NR_sched_getaffinity
299#define __NR_sched_getaffinity 204
300#elif __NR_sched_getaffinity != 204
301#error Wrong code for getaffinity system call.
302#endif /* __NR_sched_getaffinity */
303#elif KMP_ARCH_S390X
304#ifndef __NR_sched_setaffinity
305#define __NR_sched_setaffinity 239
306#elif __NR_sched_setaffinity != 239
307#error Wrong code for setaffinity system call.
308#endif /* __NR_sched_setaffinity */
309#ifndef __NR_sched_getaffinity
310#define __NR_sched_getaffinity 240
311#elif __NR_sched_getaffinity != 240
312#error Wrong code for getaffinity system call.
313#endif /* __NR_sched_getaffinity */
314#elif KMP_ARCH_SPARC
315#ifndef __NR_sched_setaffinity
316#define __NR_sched_setaffinity 261
317#elif __NR_sched_setaffinity != 261
318#error Wrong code for setaffinity system call.
319#endif /* __NR_sched_setaffinity */
320#ifndef __NR_sched_getaffinity
321#define __NR_sched_getaffinity 260
322#elif __NR_sched_getaffinity != 260
323#error Wrong code for getaffinity system call.
324#endif /* __NR_sched_getaffinity */
325#else
326#error Unknown or unsupported architecture
327#endif /* KMP_ARCH_* */
328#elif KMP_OS_FREEBSD || KMP_OS_DRAGONFLY
329#include <pthread.h>
330#include <pthread_np.h>
331#elif KMP_OS_NETBSD
332#include <pthread.h>
333#include <sched.h>
334#elif KMP_OS_AIX
335#include <sys/dr.h>
336#include <sys/rset.h>
337#define VMI_MAXRADS 64 // Maximum number of RADs allowed by AIX.
338#define GET_NUMBER_SMT_SETS 0x0004
339extern "C" int syssmt(int flags, int, int, int *);
340#endif
341class KMPNativeAffinity : public KMPAffinity {
342 class Mask : public KMPAffinity::Mask {
343 typedef unsigned long mask_t;
344 typedef decltype(__kmp_affin_mask_size) mask_size_type;
345 static const unsigned int BITS_PER_MASK_T = sizeof(mask_t) * CHAR_BIT;
346 static const mask_t ONE = 1;
347 mask_size_type get_num_mask_types() const {
348 return __kmp_affin_mask_size / sizeof(mask_t);
349 }
350
351 public:
352 mask_t *mask;
353 Mask()
354 : mask(__kmp_affin_mask_size == 0
355 ? nullptr
356 : (mask_t *)__kmp_allocate(__kmp_affin_mask_size)) {}
357 ~Mask() {
358 if (mask)
360 }
361 void set(int i) override {
362 mask[i / BITS_PER_MASK_T] |= (ONE << (i % BITS_PER_MASK_T));
363 }
364 bool is_set(int i) const override {
365 return (mask[i / BITS_PER_MASK_T] & (ONE << (i % BITS_PER_MASK_T)));
366 }
367 void clear(int i) override {
368 mask[i / BITS_PER_MASK_T] &= ~(ONE << (i % BITS_PER_MASK_T));
369 }
370 void zero() override {
371 mask_size_type e = get_num_mask_types();
372 for (mask_size_type i = 0; i < e; ++i)
373 mask[i] = (mask_t)0;
374 }
375 bool empty() const override {
376 mask_size_type e = get_num_mask_types();
377 for (mask_size_type i = 0; i < e; ++i)
378 if (mask[i] != (mask_t)0)
379 return false;
380 return true;
381 }
382 void copy(const KMPAffinity::Mask *src) override {
383 const Mask *convert = static_cast<const Mask *>(src);
384 mask_size_type e = get_num_mask_types();
385 for (mask_size_type i = 0; i < e; ++i)
386 mask[i] = convert->mask[i];
387 }
388 void bitwise_and(const KMPAffinity::Mask *rhs) override {
389 const Mask *convert = static_cast<const Mask *>(rhs);
390 mask_size_type e = get_num_mask_types();
391 for (mask_size_type i = 0; i < e; ++i)
392 mask[i] &= convert->mask[i];
393 }
394 void bitwise_or(const KMPAffinity::Mask *rhs) override {
395 const Mask *convert = static_cast<const Mask *>(rhs);
396 mask_size_type e = get_num_mask_types();
397 for (mask_size_type i = 0; i < e; ++i)
398 mask[i] |= convert->mask[i];
399 }
400 void bitwise_not() override {
401 mask_size_type e = get_num_mask_types();
402 for (mask_size_type i = 0; i < e; ++i)
403 mask[i] = ~(mask[i]);
404 }
405 bool is_equal(const KMPAffinity::Mask *rhs) const override {
406 const Mask *convert = static_cast<const Mask *>(rhs);
407 mask_size_type e = get_num_mask_types();
408 for (mask_size_type i = 0; i < e; ++i)
409 if (mask[i] != convert->mask[i])
410 return false;
411 return true;
412 }
413 int begin() const override {
414 int retval = 0;
415 while (retval < end() && !is_set(retval))
416 ++retval;
417 return retval;
418 }
419 int end() const override {
420 int e;
421 __kmp_type_convert(get_num_mask_types() * BITS_PER_MASK_T, &e);
422 return e;
423 }
424 int next(int previous) const override {
425 int retval = previous + 1;
426 while (retval < end() && !is_set(retval))
427 ++retval;
428 return retval;
429 }
430#if KMP_OS_AIX
431 // On AIX, we don't have a way to get CPU(s) a thread is bound to.
432 // This routine is only used to get the full mask.
433 int get_system_affinity(bool abort_on_error) override {
434 KMP_ASSERT2(KMP_AFFINITY_CAPABLE(),
435 "Illegal get affinity operation when not capable");
436
437 (void)abort_on_error;
438
439 // Set the mask with all CPUs that are available.
440 for (int i = 0; i < __kmp_xproc; ++i)
441 KMP_CPU_SET(i, this);
442 return 0;
443 }
444 int set_system_affinity(bool abort_on_error) const override {
445 KMP_ASSERT2(KMP_AFFINITY_CAPABLE(),
446
447 "Illegal set affinity operation when not capable");
448
449 int location;
450 int gtid = __kmp_entry_gtid();
451 int tid = thread_self();
452
453 // Unbind the thread if it was bound to any processors before so that
454 // we can bind the thread to CPUs specified by the mask not others.
455 int retval = bindprocessor(BINDTHREAD, tid, PROCESSOR_CLASS_ANY);
456
457 // On AIX, we can only bind to one instead of a set of CPUs with the
458 // bindprocessor() system call.
459 KMP_CPU_SET_ITERATE(location, this) {
460 if (KMP_CPU_ISSET(location, this)) {
461 retval = bindprocessor(BINDTHREAD, tid, location);
462 if (retval == -1 && errno == 1) {
463 rsid_t rsid;
464 rsethandle_t rsh;
465 // Put something in rsh to prevent compiler warning
466 // about uninitalized use
467 rsh = rs_alloc(RS_EMPTY);
468 rsid.at_pid = getpid();
469 if (RS_DEFAULT_RSET != ra_getrset(R_PROCESS, rsid, 0, rsh)) {
470 retval = ra_detachrset(R_PROCESS, rsid, 0);
471 retval = bindprocessor(BINDTHREAD, tid, location);
472 }
473 }
474 if (retval == 0) {
475 KA_TRACE(10, ("__kmp_set_system_affinity: Done binding "
476 "T#%d to cpu=%d.\n",
477 gtid, location));
478 continue;
479 }
480 int error = errno;
481 if (abort_on_error) {
482 __kmp_fatal(KMP_MSG(FunctionError, "bindprocessor()"),
483 KMP_ERR(error), __kmp_msg_null);
484 KA_TRACE(10, ("__kmp_set_system_affinity: Error binding "
485 "T#%d to cpu=%d, errno=%d.\n",
486 gtid, location, error));
487 return error;
488 }
489 }
490 }
491 return 0;
492 }
493#else // !KMP_OS_AIX
494 int get_system_affinity(bool abort_on_error) override {
495 KMP_ASSERT2(KMP_AFFINITY_CAPABLE(),
496 "Illegal get affinity operation when not capable");
497#if KMP_OS_LINUX
498 long retval =
499 syscall(__NR_sched_getaffinity, 0, __kmp_affin_mask_size, mask);
500#elif KMP_OS_FREEBSD || KMP_OS_NETBSD || KMP_OS_DRAGONFLY
501 int r = pthread_getaffinity_np(pthread_self(), __kmp_affin_mask_size,
502 reinterpret_cast<cpuset_t *>(mask));
503 int retval = (r == 0 ? 0 : -1);
504#endif
505 if (retval >= 0) {
506 return 0;
507 }
508 int error = errno;
509 if (abort_on_error) {
510 __kmp_fatal(KMP_MSG(FunctionError, "pthread_getaffinity_np()"),
511 KMP_ERR(error), __kmp_msg_null);
512 }
513 return error;
514 }
515 int set_system_affinity(bool abort_on_error) const override {
516 KMP_ASSERT2(KMP_AFFINITY_CAPABLE(),
517 "Illegal set affinity operation when not capable");
518#if KMP_OS_LINUX
519 long retval =
520 syscall(__NR_sched_setaffinity, 0, __kmp_affin_mask_size, mask);
521#elif KMP_OS_FREEBSD || KMP_OS_NETBSD || KMP_OS_DRAGONFLY
522 int r = pthread_setaffinity_np(pthread_self(), __kmp_affin_mask_size,
523 reinterpret_cast<cpuset_t *>(mask));
524 int retval = (r == 0 ? 0 : -1);
525#endif
526 if (retval >= 0) {
527 return 0;
528 }
529 int error = errno;
530 if (abort_on_error) {
531 __kmp_fatal(KMP_MSG(FunctionError, "pthread_setaffinity_np()"),
532 KMP_ERR(error), __kmp_msg_null);
533 }
534 return error;
535 }
536#endif // KMP_OS_AIX
537 };
538 void determine_capable(const char *env_var) override {
540 }
541 void bind_thread(int which) override { __kmp_affinity_bind_thread(which); }
542 KMPAffinity::Mask *allocate_mask() override {
543 KMPNativeAffinity::Mask *retval = new Mask();
544 return retval;
545 }
546 void deallocate_mask(KMPAffinity::Mask *m) override {
547 KMPNativeAffinity::Mask *native_mask =
548 static_cast<KMPNativeAffinity::Mask *>(m);
549 delete native_mask;
550 }
551 KMPAffinity::Mask *allocate_mask_array(int num) override {
552 return new Mask[num];
553 }
554 void deallocate_mask_array(KMPAffinity::Mask *array) override {
555 Mask *linux_array = static_cast<Mask *>(array);
556 delete[] linux_array;
557 }
558 KMPAffinity::Mask *index_mask_array(KMPAffinity::Mask *array,
559 int index) override {
560 Mask *linux_array = static_cast<Mask *>(array);
561 return &(linux_array[index]);
562 }
563 api_type get_api_type() const override { return NATIVE_OS; }
564};
565#endif /* KMP_OS_LINUX || KMP_OS_FREEBSD || KMP_OS_NETBSD || KMP_OS_DRAGONFLY \
566 || KMP_OS_AIX */
567
568#if KMP_OS_WINDOWS
569class KMPNativeAffinity : public KMPAffinity {
570 class Mask : public KMPAffinity::Mask {
571 typedef ULONG_PTR mask_t;
572 static const int BITS_PER_MASK_T = sizeof(mask_t) * CHAR_BIT;
573 mask_t *mask;
574
575 public:
576 Mask() {
577 mask = (mask_t *)__kmp_allocate(sizeof(mask_t) * __kmp_num_proc_groups);
578 }
579 ~Mask() {
580 if (mask)
582 }
583 void set(int i) override {
584 mask[i / BITS_PER_MASK_T] |= ((mask_t)1 << (i % BITS_PER_MASK_T));
585 }
586 bool is_set(int i) const override {
587 return (mask[i / BITS_PER_MASK_T] & ((mask_t)1 << (i % BITS_PER_MASK_T)));
588 }
589 void clear(int i) override {
590 mask[i / BITS_PER_MASK_T] &= ~((mask_t)1 << (i % BITS_PER_MASK_T));
591 }
592 void zero() override {
593 for (int i = 0; i < __kmp_num_proc_groups; ++i)
594 mask[i] = 0;
595 }
596 bool empty() const override {
597 for (size_t i = 0; i < __kmp_num_proc_groups; ++i)
598 if (mask[i])
599 return false;
600 return true;
601 }
602 void copy(const KMPAffinity::Mask *src) override {
603 const Mask *convert = static_cast<const Mask *>(src);
604 for (int i = 0; i < __kmp_num_proc_groups; ++i)
605 mask[i] = convert->mask[i];
606 }
607 void bitwise_and(const KMPAffinity::Mask *rhs) override {
608 const Mask *convert = static_cast<const Mask *>(rhs);
609 for (int i = 0; i < __kmp_num_proc_groups; ++i)
610 mask[i] &= convert->mask[i];
611 }
612 void bitwise_or(const KMPAffinity::Mask *rhs) override {
613 const Mask *convert = static_cast<const Mask *>(rhs);
614 for (int i = 0; i < __kmp_num_proc_groups; ++i)
615 mask[i] |= convert->mask[i];
616 }
617 void bitwise_not() override {
618 for (int i = 0; i < __kmp_num_proc_groups; ++i)
619 mask[i] = ~(mask[i]);
620 }
621 bool is_equal(const KMPAffinity::Mask *rhs) const override {
622 const Mask *convert = static_cast<const Mask *>(rhs);
623 for (size_t i = 0; i < __kmp_num_proc_groups; ++i)
624 if (mask[i] != convert->mask[i])
625 return false;
626 return true;
627 }
628 int begin() const override {
629 int retval = 0;
630 while (retval < end() && !is_set(retval))
631 ++retval;
632 return retval;
633 }
634 int end() const override { return __kmp_num_proc_groups * BITS_PER_MASK_T; }
635 int next(int previous) const override {
636 int retval = previous + 1;
637 while (retval < end() && !is_set(retval))
638 ++retval;
639 return retval;
640 }
641 int set_process_affinity(bool abort_on_error) const override {
642 if (__kmp_num_proc_groups <= 1) {
643 if (!SetProcessAffinityMask(GetCurrentProcess(), *mask)) {
644 DWORD error = GetLastError();
645 if (abort_on_error) {
646 __kmp_fatal(KMP_MSG(CantSetThreadAffMask), KMP_ERR(error),
648 }
649 return error;
650 }
651 }
652 return 0;
653 }
654 int set_system_affinity(bool abort_on_error) const override {
655 if (__kmp_num_proc_groups > 1) {
656 // Check for a valid mask.
657 GROUP_AFFINITY ga;
658 int group = get_proc_group();
659 if (group < 0) {
660 if (abort_on_error) {
661 KMP_FATAL(AffinityInvalidMask, "kmp_set_affinity");
662 }
663 return -1;
664 }
665 // Transform the bit vector into a GROUP_AFFINITY struct
666 // and make the system call to set affinity.
667 ga.Group = group;
668 ga.Mask = mask[group];
669 ga.Reserved[0] = ga.Reserved[1] = ga.Reserved[2] = 0;
670
671 KMP_DEBUG_ASSERT(__kmp_SetThreadGroupAffinity != NULL);
672 if (__kmp_SetThreadGroupAffinity(GetCurrentThread(), &ga, NULL) == 0) {
673 DWORD error = GetLastError();
674 if (abort_on_error) {
675 __kmp_fatal(KMP_MSG(CantSetThreadAffMask), KMP_ERR(error),
677 }
678 return error;
679 }
680 } else {
681 if (!SetThreadAffinityMask(GetCurrentThread(), *mask)) {
682 DWORD error = GetLastError();
683 if (abort_on_error) {
684 __kmp_fatal(KMP_MSG(CantSetThreadAffMask), KMP_ERR(error),
686 }
687 return error;
688 }
689 }
690 return 0;
691 }
692 int get_system_affinity(bool abort_on_error) override {
693 if (__kmp_num_proc_groups > 1) {
694 this->zero();
695 GROUP_AFFINITY ga;
696 KMP_DEBUG_ASSERT(__kmp_GetThreadGroupAffinity != NULL);
697 if (__kmp_GetThreadGroupAffinity(GetCurrentThread(), &ga) == 0) {
698 DWORD error = GetLastError();
699 if (abort_on_error) {
700 __kmp_fatal(KMP_MSG(FunctionError, "GetThreadGroupAffinity()"),
701 KMP_ERR(error), __kmp_msg_null);
702 }
703 return error;
704 }
705 if ((ga.Group < 0) || (ga.Group > __kmp_num_proc_groups) ||
706 (ga.Mask == 0)) {
707 return -1;
708 }
709 mask[ga.Group] = ga.Mask;
710 } else {
711 mask_t newMask, sysMask, retval;
712 if (!GetProcessAffinityMask(GetCurrentProcess(), &newMask, &sysMask)) {
713 DWORD error = GetLastError();
714 if (abort_on_error) {
715 __kmp_fatal(KMP_MSG(FunctionError, "GetProcessAffinityMask()"),
716 KMP_ERR(error), __kmp_msg_null);
717 }
718 return error;
719 }
720 retval = SetThreadAffinityMask(GetCurrentThread(), newMask);
721 if (!retval) {
722 DWORD error = GetLastError();
723 if (abort_on_error) {
724 __kmp_fatal(KMP_MSG(FunctionError, "SetThreadAffinityMask()"),
725 KMP_ERR(error), __kmp_msg_null);
726 }
727 return error;
728 }
729 newMask = SetThreadAffinityMask(GetCurrentThread(), retval);
730 if (!newMask) {
731 DWORD error = GetLastError();
732 if (abort_on_error) {
733 __kmp_fatal(KMP_MSG(FunctionError, "SetThreadAffinityMask()"),
734 KMP_ERR(error), __kmp_msg_null);
735 }
736 }
737 *mask = retval;
738 }
739 return 0;
740 }
741 int get_proc_group() const override {
742 int group = -1;
743 if (__kmp_num_proc_groups == 1) {
744 return 1;
745 }
746 for (int i = 0; i < __kmp_num_proc_groups; i++) {
747 if (mask[i] == 0)
748 continue;
749 if (group >= 0)
750 return -1;
751 group = i;
752 }
753 return group;
754 }
755 };
756 void determine_capable(const char *env_var) override {
758 }
759 void bind_thread(int which) override { __kmp_affinity_bind_thread(which); }
760 KMPAffinity::Mask *allocate_mask() override { return new Mask(); }
761 void deallocate_mask(KMPAffinity::Mask *m) override { delete m; }
762 KMPAffinity::Mask *allocate_mask_array(int num) override {
763 return new Mask[num];
764 }
765 void deallocate_mask_array(KMPAffinity::Mask *array) override {
766 Mask *windows_array = static_cast<Mask *>(array);
767 delete[] windows_array;
768 }
769 KMPAffinity::Mask *index_mask_array(KMPAffinity::Mask *array,
770 int index) override {
771 Mask *windows_array = static_cast<Mask *>(array);
772 return &(windows_array[index]);
773 }
774 api_type get_api_type() const override { return NATIVE_OS; }
775};
776#endif /* KMP_OS_WINDOWS */
777#endif /* KMP_AFFINITY_SUPPORTED */
778
779// Describe an attribute for a level in the machine topology
781 int core_type : 8;
782 int core_eff : 8;
783 unsigned valid : 1;
784 unsigned reserved : 15;
785
786 static const int UNKNOWN_CORE_EFF = -1;
787
795 void set_core_eff(int eff) {
796 valid = 1;
797 core_eff = eff;
798 }
802 int get_core_eff() const { return core_eff; }
803 bool is_core_type_valid() const {
805 }
806 bool is_core_eff_valid() const { return core_eff != UNKNOWN_CORE_EFF; }
807 operator bool() const { return valid; }
813 bool contains(const kmp_hw_attr_t &other) const {
814 if (!valid && !other.valid)
815 return true;
816 if (valid && other.valid) {
817 if (other.is_core_type_valid()) {
818 if (!is_core_type_valid() || (get_core_type() != other.get_core_type()))
819 return false;
820 }
821 if (other.is_core_eff_valid()) {
822 if (!is_core_eff_valid() || (get_core_eff() != other.get_core_eff()))
823 return false;
824 }
825 return true;
826 }
827 return false;
828 }
829#if KMP_AFFINITY_SUPPORTED
830 bool contains(const kmp_affinity_attrs_t &attr) const {
831 if (!valid && !attr.valid)
832 return true;
833 if (valid && attr.valid) {
834 if (attr.core_type != KMP_HW_CORE_TYPE_UNKNOWN)
835 return (is_core_type_valid() &&
836 (get_core_type() == (kmp_hw_core_type_t)attr.core_type));
837 if (attr.core_eff != UNKNOWN_CORE_EFF)
838 return (is_core_eff_valid() && (get_core_eff() == attr.core_eff));
839 return true;
840 }
841 return false;
842 }
843#endif // KMP_AFFINITY_SUPPORTED
844 bool operator==(const kmp_hw_attr_t &rhs) const {
845 return (rhs.valid == valid && rhs.core_eff == core_eff &&
846 rhs.core_type == core_type);
847 }
848 bool operator!=(const kmp_hw_attr_t &rhs) const { return !operator==(rhs); }
849};
850
851#if KMP_AFFINITY_SUPPORTED
852KMP_BUILD_ASSERT(sizeof(kmp_hw_attr_t) == sizeof(kmp_affinity_attrs_t));
853#endif
854
856public:
857 static const int UNKNOWN_ID = -1;
858 static const int MULTIPLE_ID = -2;
859 static int compare_ids(const void *a, const void *b);
860 static int compare_compact(const void *a, const void *b);
863 bool leader;
864 int os_id;
867
868 void print() const;
869 void clear() {
870 for (int i = 0; i < (int)KMP_HW_LAST; ++i)
871 ids[i] = UNKNOWN_ID;
872 leader = false;
873 attrs.clear();
874 }
875};
876
878
879 struct flags_t {
880 int uniform : 1;
881 int reserved : 31;
882 };
883
884 int depth;
885
886 // The following arrays are all 'depth' long and have been
887 // allocated to hold up to KMP_HW_LAST number of objects if
888 // needed so layers can be added without reallocation of any array
889
890 // Orderd array of the types in the topology
891 kmp_hw_t *types;
892
893 // Keep quick topology ratios, for non-uniform topologies,
894 // this ratio holds the max number of itemAs per itemB
895 // e.g., [ 4 packages | 6 cores / package | 2 threads / core ]
896 int *ratio;
897
898 // Storage containing the absolute number of each topology layer
899 int *count;
900
901 // The number of core efficiencies. This is only useful for hybrid
902 // topologies. Core efficiencies will range from 0 to num efficiencies - 1
903 int num_core_efficiencies;
904 int num_core_types;
906
907 // The hardware threads array
908 // hw_threads is num_hw_threads long
909 // Each hw_thread's ids and sub_ids are depth deep
910 int num_hw_threads;
911 kmp_hw_thread_t *hw_threads;
912
913 // Equivalence hash where the key is the hardware topology item
914 // and the value is the equivalent hardware topology type in the
915 // types[] array, if the value is KMP_HW_UNKNOWN, then there is no
916 // known equivalence for the topology type
917 kmp_hw_t equivalent[KMP_HW_LAST];
918
919 // Flags describing the topology
920 flags_t flags;
921
922 // Compact value used during sort_compact()
923 int compact;
924
925#if KMP_GROUP_AFFINITY
926 // Insert topology information about Windows Processor groups
927 void _insert_windows_proc_groups();
928#endif
929
930 // Count each item & get the num x's per y
931 // e.g., get the number of cores and the number of threads per core
932 // for each (x, y) in (KMP_HW_* , KMP_HW_*)
933 void _gather_enumeration_information();
934
935 // Remove layers that don't add information to the topology.
936 // This is done by having the layer take on the id = UNKNOWN_ID (-1)
937 void _remove_radix1_layers();
938
939 // Find out if the topology is uniform
940 void _discover_uniformity();
941
942 // Set all the sub_ids for each hardware thread
943 void _set_sub_ids();
944
945 // Set global affinity variables describing the number of threads per
946 // core, the number of packages, the number of cores per package, and
947 // the number of cores.
948 void _set_globals();
949
950 // Set the last level cache equivalent type
951 void _set_last_level_cache();
952
953 // Return the number of cores with a particular attribute, 'attr'.
954 // If 'find_all' is true, then find all cores on the machine, otherwise find
955 // all cores per the layer 'above'
956 int _get_ncores_with_attr(const kmp_hw_attr_t &attr, int above,
957 bool find_all = false) const;
958
959public:
960 // Force use of allocate()/deallocate()
961 kmp_topology_t() = delete;
962 kmp_topology_t(const kmp_topology_t &t) = delete;
966
967 static kmp_topology_t *allocate(int nproc, int ndepth, const kmp_hw_t *types);
968 static void deallocate(kmp_topology_t *);
969
970 // Functions used in create_map() routines
971 kmp_hw_thread_t &at(int index) {
972 KMP_DEBUG_ASSERT(index >= 0 && index < num_hw_threads);
973 return hw_threads[index];
974 }
975 const kmp_hw_thread_t &at(int index) const {
976 KMP_DEBUG_ASSERT(index >= 0 && index < num_hw_threads);
977 return hw_threads[index];
978 }
979 int get_num_hw_threads() const { return num_hw_threads; }
980 void sort_ids() {
981 qsort(hw_threads, num_hw_threads, sizeof(kmp_hw_thread_t),
983 }
984
985 // Insert a new topology layer after allocation
986 void insert_layer(kmp_hw_t type, const int *ids);
987
988 // Check if the hardware ids are unique, if they are
989 // return true, otherwise return false
990 bool check_ids() const;
991
992 // Function to call after the create_map() routine
993 void canonicalize();
994 void canonicalize(int pkgs, int cores_per_pkg, int thr_per_core, int cores);
995
996// Functions used after canonicalize() called
997
998#if KMP_AFFINITY_SUPPORTED
999 // Set the granularity for affinity settings
1000 void set_granularity(kmp_affinity_t &stgs) const;
1001 bool is_close(int hwt1, int hwt2, const kmp_affinity_t &stgs) const;
1002 bool restrict_to_mask(const kmp_affin_mask_t *mask);
1003 bool filter_hw_subset();
1004#endif
1005 bool is_uniform() const { return flags.uniform; }
1006 // Tell whether a type is a valid type in the topology
1007 // returns KMP_HW_UNKNOWN when there is no equivalent type
1009 if (type == KMP_HW_UNKNOWN)
1010 return KMP_HW_UNKNOWN;
1011 return equivalent[type];
1012 }
1013 // Set type1 = type2
1017 kmp_hw_t real_type2 = equivalent[type2];
1018 if (real_type2 == KMP_HW_UNKNOWN)
1019 real_type2 = type2;
1020 equivalent[type1] = real_type2;
1021 // This loop is required since any of the types may have been set to
1022 // be equivalent to type1. They all must be checked and reset to type2.
1024 if (equivalent[type] == type1) {
1025 equivalent[type] = real_type2;
1026 }
1027 }
1028 }
1029 // Calculate number of types corresponding to level1
1030 // per types corresponding to level2 (e.g., number of threads per core)
1031 int calculate_ratio(int level1, int level2) const {
1032 KMP_DEBUG_ASSERT(level1 >= 0 && level1 < depth);
1033 KMP_DEBUG_ASSERT(level2 >= 0 && level2 < depth);
1034 int r = 1;
1035 for (int level = level1; level > level2; --level)
1036 r *= ratio[level];
1037 return r;
1038 }
1039 int get_ratio(int level) const {
1040 KMP_DEBUG_ASSERT(level >= 0 && level < depth);
1041 return ratio[level];
1042 }
1043 int get_depth() const { return depth; };
1045 KMP_DEBUG_ASSERT(level >= 0 && level < depth);
1046 return types[level];
1047 }
1050 int eq_type = equivalent[type];
1051 if (eq_type == KMP_HW_UNKNOWN)
1052 return -1;
1053 for (int i = 0; i < depth; ++i)
1054 if (types[i] == eq_type)
1055 return i;
1056 return -1;
1057 }
1058 int get_count(int level) const {
1059 KMP_DEBUG_ASSERT(level >= 0 && level < depth);
1060 return count[level];
1061 }
1062 // Return the total number of cores with attribute 'attr'
1063 int get_ncores_with_attr(const kmp_hw_attr_t &attr) const {
1064 return _get_ncores_with_attr(attr, -1, true);
1065 }
1066 // Return the number of cores with attribute
1067 // 'attr' per topology level 'above'
1068 int get_ncores_with_attr_per(const kmp_hw_attr_t &attr, int above) const {
1069 return _get_ncores_with_attr(attr, above, false);
1070 }
1071
1072#if KMP_AFFINITY_SUPPORTED
1073 friend int kmp_hw_thread_t::compare_compact(const void *a, const void *b);
1074 void sort_compact(kmp_affinity_t &affinity) {
1075 compact = affinity.compact;
1076 qsort(hw_threads, num_hw_threads, sizeof(kmp_hw_thread_t),
1078 }
1079#endif
1080 void print(const char *env_var = "KMP_AFFINITY") const;
1081 void dump() const;
1082};
1084
1086 const static size_t MAX_ATTRS = KMP_HW_MAX_NUM_CORE_EFFS;
1087
1088public:
1089 // Describe a machine topology item in KMP_HW_SUBSET
1090 struct item_t {
1093 int num[MAX_ATTRS];
1094 int offset[MAX_ATTRS];
1096 };
1097 // Put parenthesis around max to avoid accidental use of Windows max macro.
1098 const static int USE_ALL = (std::numeric_limits<int>::max)();
1099
1100private:
1101 int depth;
1102 int capacity;
1103 item_t *items;
1104 kmp_uint64 set;
1105 bool absolute;
1106 // The set must be able to handle up to KMP_HW_LAST number of layers
1107 KMP_BUILD_ASSERT(sizeof(set) * 8 >= KMP_HW_LAST);
1108 // Sorting the KMP_HW_SUBSET items to follow topology order
1109 // All unknown topology types will be at the beginning of the subset
1110 static int hw_subset_compare(const void *i1, const void *i2) {
1111 kmp_hw_t type1 = ((const item_t *)i1)->type;
1112 kmp_hw_t type2 = ((const item_t *)i2)->type;
1113 int level1 = __kmp_topology->get_level(type1);
1114 int level2 = __kmp_topology->get_level(type2);
1115 return level1 - level2;
1116 }
1117
1118public:
1119 // Force use of allocate()/deallocate()
1125
1127 int initial_capacity = 5;
1128 kmp_hw_subset_t *retval =
1130 retval->depth = 0;
1131 retval->capacity = initial_capacity;
1132 retval->set = 0ull;
1133 retval->absolute = false;
1134 retval->items = (item_t *)__kmp_allocate(sizeof(item_t) * initial_capacity);
1135 return retval;
1136 }
1137 static void deallocate(kmp_hw_subset_t *subset) {
1138 __kmp_free(subset->items);
1139 __kmp_free(subset);
1140 }
1141 void set_absolute() { absolute = true; }
1142 bool is_absolute() const { return absolute; }
1143 void push_back(int num, kmp_hw_t type, int offset, kmp_hw_attr_t attr) {
1144 for (int i = 0; i < depth; ++i) {
1145 // Found an existing item for this layer type
1146 // Add the num, offset, and attr to this item
1147 if (items[i].type == type) {
1148 int idx = items[i].num_attrs++;
1149 if ((size_t)idx >= MAX_ATTRS)
1150 return;
1151 items[i].num[idx] = num;
1152 items[i].offset[idx] = offset;
1153 items[i].attr[idx] = attr;
1154 return;
1155 }
1156 }
1157 if (depth == capacity - 1) {
1158 capacity *= 2;
1159 item_t *new_items = (item_t *)__kmp_allocate(sizeof(item_t) * capacity);
1160 for (int i = 0; i < depth; ++i)
1161 new_items[i] = items[i];
1162 __kmp_free(items);
1163 items = new_items;
1164 }
1165 items[depth].num_attrs = 1;
1166 items[depth].type = type;
1167 items[depth].num[0] = num;
1168 items[depth].offset[0] = offset;
1169 items[depth].attr[0] = attr;
1170 depth++;
1171 set |= (1ull << type);
1172 }
1173 int get_depth() const { return depth; }
1174 const item_t &at(int index) const {
1175 KMP_DEBUG_ASSERT(index >= 0 && index < depth);
1176 return items[index];
1177 }
1178 item_t &at(int index) {
1179 KMP_DEBUG_ASSERT(index >= 0 && index < depth);
1180 return items[index];
1181 }
1182 void remove(int index) {
1183 KMP_DEBUG_ASSERT(index >= 0 && index < depth);
1184 set &= ~(1ull << items[index].type);
1185 for (int j = index + 1; j < depth; ++j) {
1186 items[j - 1] = items[j];
1187 }
1188 depth--;
1189 }
1190 void sort() {
1192 qsort(items, depth, sizeof(item_t), hw_subset_compare);
1193 }
1194 bool specified(kmp_hw_t type) const { return ((set & (1ull << type)) > 0); }
1195
1196 // Canonicalize the KMP_HW_SUBSET value if it is not an absolute subset.
1197 // This means putting each of {sockets, cores, threads} in the topology if
1198 // they are not specified:
1199 // e.g., 1s,2c => 1s,2c,*t | 2c,1t => *s,2c,1t | 1t => *s,*c,1t | etc.
1200 // e.g., 3module => *s,3module,*c,*t
1201 // By doing this, the runtime assumes users who fiddle with KMP_HW_SUBSET
1202 // are expecting the traditional sockets/cores/threads topology. For newer
1203 // hardware, there can be intervening layers like dies/tiles/modules
1204 // (usually corresponding to a cache level). So when a user asks for
1205 // 1s,6c,2t and the topology is really 1s,2modules,4cores,2threads, the user
1206 // should get 12 hardware threads across 6 cores and effectively ignore the
1207 // module layer.
1208 void canonicalize(const kmp_topology_t *top) {
1209 // Layers to target for KMP_HW_SUBSET canonicalization
1211
1212 // Do not target-layer-canonicalize absolute KMP_HW_SUBSETS
1213 if (is_absolute())
1214 return;
1215
1216 // Do not target-layer-canonicalize KMP_HW_SUBSETS when the
1217 // topology doesn't have these layers
1218 for (kmp_hw_t type : targeted)
1219 if (top->get_level(type) == KMP_HW_UNKNOWN)
1220 return;
1221
1222 // Put targeted layers in topology if they do not exist
1223 for (kmp_hw_t type : targeted) {
1224 bool found = false;
1225 for (int i = 0; i < get_depth(); ++i) {
1226 if (top->get_equivalent_type(items[i].type) == type) {
1227 found = true;
1228 break;
1229 }
1230 }
1231 if (!found) {
1233 }
1234 }
1235 sort();
1236 // Set as an absolute topology that only targets the targeted layers
1237 set_absolute();
1238 }
1239 void dump() const {
1240 printf("**********************\n");
1241 printf("*** kmp_hw_subset: ***\n");
1242 printf("* depth: %d\n", depth);
1243 printf("* items:\n");
1244 for (int i = 0; i < depth; ++i) {
1245 printf(" type: %s\n", __kmp_hw_get_keyword(items[i].type));
1246 for (int j = 0; j < items[i].num_attrs; ++j) {
1247 printf(" num: %d, offset: %d, attr: ", items[i].num[j],
1248 items[i].offset[j]);
1249 if (!items[i].attr[j]) {
1250 printf(" (none)\n");
1251 } else {
1252 printf(
1253 " core_type = %s, core_eff = %d\n",
1254 __kmp_hw_get_core_type_string(items[i].attr[j].get_core_type()),
1255 items[i].attr[j].get_core_eff());
1256 }
1257 }
1258 }
1259 printf("* set: 0x%llx\n", set);
1260 printf("* absolute: %d\n", absolute);
1261 printf("**********************\n");
1262 }
1263};
1265
1266/* A structure for holding machine-specific hierarchy info to be computed once
1267 at init. This structure represents a mapping of threads to the actual machine
1268 hierarchy, or to our best guess at what the hierarchy might be, for the
1269 purpose of performing an efficient barrier. In the worst case, when there is
1270 no machine hierarchy information, it produces a tree suitable for a barrier,
1271 similar to the tree used in the hyper barrier. */
1273public:
1274 /* Good default values for number of leaves and branching factor, given no
1275 affinity information. Behaves a bit like hyper barrier. */
1276 static const kmp_uint32 maxLeaves = 4;
1277 static const kmp_uint32 minBranch = 4;
1278 /** Number of levels in the hierarchy. Typical levels are threads/core,
1279 cores/package or socket, packages/node, nodes/machine, etc. We don't want
1280 to get specific with nomenclature. When the machine is oversubscribed we
1281 add levels to duplicate the hierarchy, doubling the thread capacity of the
1282 hierarchy each time we add a level. */
1284
1285 /** This is specifically the depth of the machine configuration hierarchy, in
1286 terms of the number of levels along the longest path from root to any
1287 leaf. It corresponds to the number of entries in numPerLevel if we exclude
1288 all but one trailing 1. */
1292 volatile kmp_int8 uninitialized; // 0=initialized, 1=not initialized,
1293 // 2=initialization in progress
1294 volatile kmp_int8 resizing; // 0=not resizing, 1=resizing
1295
1296 /** Level 0 corresponds to leaves. numPerLevel[i] is the number of children
1297 the parent of a node at level i has. For example, if we have a machine
1298 with 4 packages, 4 cores/package and 2 HT per core, then numPerLevel =
1299 {2, 4, 4, 1, 1}. All empty levels are set to 1. */
1302
1304 int hier_depth = __kmp_topology->get_depth();
1305 for (int i = hier_depth - 1, level = 0; i >= 0; --i, ++level) {
1306 numPerLevel[level] = __kmp_topology->get_ratio(i);
1307 }
1308 }
1309
1312
1313 void fini() {
1314 if (!uninitialized && numPerLevel) {
1316 numPerLevel = NULL;
1318 }
1319 }
1320
1321 void init(int num_addrs) {
1324 if (bool_result == 0) { // Wait for initialization
1325 while (TCR_1(uninitialized) != initialized)
1326 KMP_CPU_PAUSE();
1327 return;
1328 }
1329 KMP_DEBUG_ASSERT(bool_result == 1);
1330
1331 /* Added explicit initialization of the data fields here to prevent usage of
1332 dirty value observed when static library is re-initialized multiple times
1333 (e.g. when non-OpenMP thread repeatedly launches/joins thread that uses
1334 OpenMP). */
1335 depth = 1;
1336 resizing = 0;
1337 maxLevels = 7;
1338 numPerLevel =
1341 for (kmp_uint32 i = 0; i < maxLevels;
1342 ++i) { // init numPerLevel[*] to 1 item per level
1343 numPerLevel[i] = 1;
1344 skipPerLevel[i] = 1;
1345 }
1346
1347 // Sort table by physical ID
1348 if (__kmp_topology && __kmp_topology->get_depth() > 0) {
1349 deriveLevels();
1350 } else {
1352 numPerLevel[1] = num_addrs / maxLeaves;
1353 if (num_addrs % maxLeaves)
1354 numPerLevel[1]++;
1355 }
1356
1357 base_num_threads = num_addrs;
1358 for (int i = maxLevels - 1; i >= 0;
1359 --i) // count non-empty levels to get depth
1360 if (numPerLevel[i] != 1 || depth > 1) // only count one top-level '1'
1361 depth++;
1362
1363 kmp_uint32 branch = minBranch;
1364 if (numPerLevel[0] == 1)
1365 branch = num_addrs / maxLeaves;
1366 if (branch < minBranch)
1367 branch = minBranch;
1368 for (kmp_uint32 d = 0; d < depth - 1; ++d) { // optimize hierarchy width
1369 while (numPerLevel[d] > branch ||
1370 (d == 0 && numPerLevel[d] > maxLeaves)) { // max 4 on level 0!
1371 if (numPerLevel[d] & 1)
1372 numPerLevel[d]++;
1373 numPerLevel[d] = numPerLevel[d] >> 1;
1374 if (numPerLevel[d + 1] == 1)
1375 depth++;
1376 numPerLevel[d + 1] = numPerLevel[d + 1] << 1;
1377 }
1378 if (numPerLevel[0] == 1) {
1379 branch = branch >> 1;
1380 if (branch < 4)
1381 branch = minBranch;
1382 }
1383 }
1384
1385 for (kmp_uint32 i = 1; i < depth; ++i)
1386 skipPerLevel[i] = numPerLevel[i - 1] * skipPerLevel[i - 1];
1387 // Fill in hierarchy in the case of oversubscription
1388 for (kmp_uint32 i = depth; i < maxLevels; ++i)
1389 skipPerLevel[i] = 2 * skipPerLevel[i - 1];
1390
1391 uninitialized = initialized; // One writer
1392 }
1393
1394 // Resize the hierarchy if nproc changes to something larger than before
1395 void resize(kmp_uint32 nproc) {
1396 kmp_int8 bool_result = KMP_COMPARE_AND_STORE_ACQ8(&resizing, 0, 1);
1397 while (bool_result == 0) { // someone else is trying to resize
1398 KMP_CPU_PAUSE();
1399 if (nproc <= base_num_threads) // happy with other thread's resize
1400 return;
1401 else // try to resize
1402 bool_result = KMP_COMPARE_AND_STORE_ACQ8(&resizing, 0, 1);
1403 }
1404 KMP_DEBUG_ASSERT(bool_result != 0);
1405 if (nproc <= base_num_threads)
1406 return; // happy with other thread's resize
1407
1408 // Calculate new maxLevels
1409 kmp_uint32 old_sz = skipPerLevel[depth - 1];
1410 kmp_uint32 incs = 0, old_maxLevels = maxLevels;
1411 // First see if old maxLevels is enough to contain new size
1412 for (kmp_uint32 i = depth; i < maxLevels && nproc > old_sz; ++i) {
1413 skipPerLevel[i] = 2 * skipPerLevel[i - 1];
1414 numPerLevel[i - 1] *= 2;
1415 old_sz *= 2;
1416 depth++;
1417 }
1418 if (nproc > old_sz) { // Not enough space, need to expand hierarchy
1419 while (nproc > old_sz) {
1420 old_sz *= 2;
1421 incs++;
1422 depth++;
1423 }
1424 maxLevels += incs;
1425
1426 // Resize arrays
1427 kmp_uint32 *old_numPerLevel = numPerLevel;
1428 kmp_uint32 *old_skipPerLevel = skipPerLevel;
1429 numPerLevel = skipPerLevel = NULL;
1430 numPerLevel =
1433
1434 // Copy old elements from old arrays
1435 for (kmp_uint32 i = 0; i < old_maxLevels; ++i) {
1436 // init numPerLevel[*] to 1 item per level
1437 numPerLevel[i] = old_numPerLevel[i];
1438 skipPerLevel[i] = old_skipPerLevel[i];
1439 }
1440
1441 // Init new elements in arrays to 1
1442 for (kmp_uint32 i = old_maxLevels; i < maxLevels; ++i) {
1443 // init numPerLevel[*] to 1 item per level
1444 numPerLevel[i] = 1;
1445 skipPerLevel[i] = 1;
1446 }
1447
1448 // Free old arrays
1449 __kmp_free(old_numPerLevel);
1450 }
1451
1452 // Fill in oversubscription levels of hierarchy
1453 for (kmp_uint32 i = old_maxLevels; i < maxLevels; ++i)
1454 skipPerLevel[i] = 2 * skipPerLevel[i - 1];
1455
1456 base_num_threads = nproc;
1457 resizing = 0; // One writer
1458 }
1459};
1460#endif // KMP_AFFINITY_H
char bool
kmp_uint32 * numPerLevel
Level 0 corresponds to leaves.
static const kmp_uint32 maxLeaves
kmp_uint32 * skipPerLevel
void resize(kmp_uint32 nproc)
kmp_uint32 base_num_threads
volatile kmp_int8 uninitialized
kmp_uint32 maxLevels
Number of levels in the hierarchy.
kmp_uint32 depth
This is specifically the depth of the machine configuration hierarchy, in terms of the number of leve...
void init(int num_addrs)
volatile kmp_int8 resizing
static const kmp_uint32 minBranch
bool specified(kmp_hw_t type) const
void push_back(int num, kmp_hw_t type, int offset, kmp_hw_attr_t attr)
bool is_absolute() const
kmp_hw_subset_t()=delete
kmp_hw_subset_t(kmp_hw_subset_t &&t)=delete
void canonicalize(const kmp_topology_t *top)
static kmp_hw_subset_t * allocate()
static void deallocate(kmp_hw_subset_t *subset)
void remove(int index)
kmp_hw_subset_t(const kmp_hw_subset_t &t)=delete
int get_depth() const
static const int USE_ALL
const item_t & at(int index) const
item_t & at(int index)
kmp_hw_subset_t & operator=(kmp_hw_subset_t &&t)=delete
kmp_hw_subset_t & operator=(const kmp_hw_subset_t &t)=delete
void dump() const
kmp_hw_attr_t attrs
static const int UNKNOWN_ID
int sub_ids[KMP_HW_LAST]
static int compare_compact(const void *a, const void *b)
void print() const
static int compare_ids(const void *a, const void *b)
static const int MULTIPLE_ID
int ids[KMP_HW_LAST]
kmp_hw_thread_t & at(int index)
void dump() const
int get_level(kmp_hw_t type) const
int get_count(int level) const
int get_ratio(int level) const
static void deallocate(kmp_topology_t *)
kmp_hw_t get_equivalent_type(kmp_hw_t type) const
void set_equivalent_type(kmp_hw_t type1, kmp_hw_t type2)
int get_num_hw_threads() const
int get_ncores_with_attr_per(const kmp_hw_attr_t &attr, int above) const
const kmp_hw_thread_t & at(int index) const
int get_depth() const
void insert_layer(kmp_hw_t type, const int *ids)
int calculate_ratio(int level1, int level2) const
bool is_uniform() const
static kmp_topology_t * allocate(int nproc, int ndepth, const kmp_hw_t *types)
void print(const char *env_var="KMP_AFFINITY") const
kmp_topology_t & operator=(kmp_topology_t &&t)=delete
kmp_topology_t(const kmp_topology_t &t)=delete
kmp_hw_t get_type(int level) const
kmp_topology_t()=delete
kmp_topology_t(kmp_topology_t &&t)=delete
kmp_topology_t & operator=(const kmp_topology_t &t)=delete
bool check_ids() const
int get_ncores_with_attr(const kmp_hw_attr_t &attr) const
void
Definition ittnotify.h:3324
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int mask
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t ITT_FORMAT d no args const wchar_t const wchar_t ITT_FORMAT s __itt_heap_function void size_t int ITT_FORMAT d __itt_heap_function void ITT_FORMAT p __itt_heap_function void void size_t int ITT_FORMAT d no args no args unsigned int ITT_FORMAT u const __itt_domain __itt_id ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain __itt_id ITT_FORMAT p const __itt_domain __itt_id __itt_timestamp __itt_timestamp end
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t ITT_FORMAT d no args const wchar_t const wchar_t ITT_FORMAT s __itt_heap_function void size_t int ITT_FORMAT d __itt_heap_function void ITT_FORMAT p __itt_heap_function void void size_t int ITT_FORMAT d no args no args unsigned int ITT_FORMAT u const __itt_domain __itt_id ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain __itt_id ITT_FORMAT p const __itt_domain __itt_id __itt_timestamp begin
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t ITT_FORMAT d no args const wchar_t const wchar_t ITT_FORMAT s __itt_heap_function void size_t int ITT_FORMAT d __itt_heap_function void ITT_FORMAT p __itt_heap_function void void size_t int ITT_FORMAT d no args no args unsigned int ITT_FORMAT u const __itt_domain __itt_id ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain __itt_id ITT_FORMAT p const __itt_domain __itt_id __itt_timestamp __itt_timestamp ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain ITT_FORMAT p const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_id __itt_string_handle __itt_metadata_type size_t void ITT_FORMAT p const __itt_domain __itt_id __itt_string_handle const wchar_t size_t ITT_FORMAT lu const __itt_domain __itt_id __itt_relation __itt_id ITT_FORMAT p const wchar_t int ITT_FORMAT __itt_group_mark d int
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t ITT_FORMAT d no args const wchar_t const wchar_t ITT_FORMAT s __itt_heap_function void size_t int ITT_FORMAT d __itt_heap_function void ITT_FORMAT p __itt_heap_function void void size_t int ITT_FORMAT d no args no args unsigned int ITT_FORMAT u const __itt_domain __itt_id ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain __itt_id ITT_FORMAT p const __itt_domain __itt_id __itt_timestamp __itt_timestamp ITT_FORMAT lu const __itt_domain __itt_id __itt_id __itt_string_handle ITT_FORMAT p const __itt_domain ITT_FORMAT p const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_string_handle unsigned long long ITT_FORMAT lu const __itt_domain __itt_id __itt_string_handle __itt_metadata_type type
#define __kmp_free(ptr)
Definition kmp.h:3749
#define KMP_HW_MAX_NUM_CORE_EFFS
Definition kmp.h:619
#define KMP_CPU_PAUSE()
Definition kmp.h:1585
int __kmp_xproc
#define KMP_FOREACH_HW_TYPE(type)
Definition kmp.h:626
#define __kmp_entry_gtid()
Definition kmp.h:3594
const char * __kmp_hw_get_keyword(kmp_hw_t type, bool plural=false)
#define __kmp_allocate(size)
Definition kmp.h:3747
#define TRUE
Definition kmp.h:1341
const char * __kmp_hw_get_core_type_string(kmp_hw_core_type_t type)
kmp_hw_t
Definition kmp.h:591
@ KMP_HW_UNKNOWN
Definition kmp.h:592
@ KMP_HW_SOCKET
Definition kmp.h:593
@ KMP_HW_CORE
Definition kmp.h:603
@ KMP_HW_THREAD
Definition kmp.h:604
@ KMP_HW_LAST
Definition kmp.h:605
kmp_hw_core_type_t
Definition kmp.h:608
@ KMP_HW_MAX_NUM_CORE_TYPES
Definition kmp.h:615
@ KMP_HW_CORE_TYPE_UNKNOWN
Definition kmp.h:609
static void __kmp_type_convert(T1 src, T2 *dest)
Definition kmp.h:4879
#define KMP_DEBUG_ASSERT_VALID_HW_TYPE(type)
Definition kmp.h:621
kmp_hw_subset_t * __kmp_hw_subset
kmp_topology_t * __kmp_topology
kmp_topology_t * __kmp_topology
KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 kmp_int8
KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86 KMP_ARCH_X86<<, 2i, 1, KMP_ARCH_X86) ATOMIC_CMPXCHG(fixed2, shr, kmp_int16, 16, > KMP_ARCH_X86 KMP_ARCH_X86 kmp_uint32
#define KA_TRACE(d, x)
Definition kmp_debug.h:157
#define KMP_BUILD_ASSERT(expr)
Definition kmp_debug.h:26
#define KMP_DEBUG_ASSERT(cond)
Definition kmp_debug.h:61
#define KMP_ASSERT2(cond, msg)
Definition kmp_debug.h:60
unsigned long long kmp_uint64
kmp_msg_t __kmp_msg_null
Definition kmp_i18n.cpp:36
void __kmp_fatal(kmp_msg_t message,...)
Definition kmp_i18n.cpp:873
#define KMP_WARNING(...)
Definition kmp_i18n.h:144
#define KMP_MSG(...)
Definition kmp_i18n.h:121
#define KMP_FATAL(...)
Definition kmp_i18n.h:146
#define KMP_ERR
Definition kmp_i18n.h:125
#define TCR_1(a)
Definition kmp_os.h:1139
#define KMP_COMPARE_AND_STORE_ACQ8(p, cv, sv)
Definition kmp_os.h:806
#define i
Definition kmp_stub.cpp:88
int a
bool contains(const kmp_hw_attr_t &other) const
int get_core_eff() const
unsigned reserved
bool is_core_type_valid() const
kmp_hw_core_type_t get_core_type() const
void set_core_type(kmp_hw_core_type_t type)
static const int UNKNOWN_CORE_EFF
void set_core_eff(int eff)
bool operator==(const kmp_hw_attr_t &rhs) const
bool is_core_eff_valid() const
bool operator!=(const kmp_hw_attr_t &rhs) const
kmp_hw_attr_t attr[MAX_ATTRS]
void __kmp_affinity_determine_capable(const char *env_var)
void __kmp_affinity_bind_thread(int proc)