LLVM OpenMP
parallel-wsloop-intfor.c
Go to the documentation of this file.
1// RUN: %libomp-compile -fopenmp-version=61 && %libomp-run \
2// RUN: | FileCheck %s --match-full-lines
3
4// Flatten changes which iterations a worksharing loop assigns to which
5// thread. With schedule(static,1) and two threads:
6// * without flatten, the outer loop (2 iterations) is distributed
7// * with flatten, the product space (4 iterations) is distributed
8// Results are printed per thread after the parallel region so FileCheck
9// is not racy.
10
11#ifndef HEADER
12#define HEADER
13
14#include <omp.h>
15#include <stdio.h>
16#include <stdlib.h>
17
18enum { NThreads = 2, MaxIters = 4 };
19
20static void dump(const char *tag, int count[NThreads],
21 int pairs[NThreads][MaxIters][2]) {
22 printf("%s\n", tag);
23 for (int t = 0; t < NThreads; ++t) {
24 printf("tid=%d count=%d\n", t, count[t]);
25 for (int k = 0; k < count[t]; ++k)
26 printf("tid=%d i=%d j=%d\n", t, pairs[t][k][0], pairs[t][k][1]);
27 }
28}
29
30int main() {
31 int count[NThreads];
32 int pairs[NThreads][MaxIters][2];
33
34 count[0] = count[1] = 0;
35#pragma omp parallel for schedule(static, 1) num_threads(2)
36#pragma omp flatten
37 for (int i = 0; i < 2; ++i)
38 for (int j = 0; j < 2; ++j) {
39 int t = omp_get_thread_num();
40 int c = count[t]++;
41 pairs[t][c][0] = i;
42 pairs[t][c][1] = j;
43 }
44 dump("with-flatten", count, pairs);
45
46 count[0] = count[1] = 0;
47#pragma omp parallel for schedule(static, 1) num_threads(2)
48 for (int i = 0; i < 2; ++i)
49 for (int j = 0; j < 2; ++j) {
50 int t = omp_get_thread_num();
51 int c = count[t]++;
52 pairs[t][c][0] = i;
53 pairs[t][c][1] = j;
54 }
55 dump("without-flatten", count, pairs);
56
57 return EXIT_SUCCESS;
58}
59
60#endif /* HEADER */
61
62// Flattened product space has 4 iterations; static,1 with 2 threads gives
63// each thread 2 chunks: (0,0)+(1,0) and (0,1)+(1,1).
64// CHECK: with-flatten
65// CHECK-NEXT: tid=0 count=2
66// CHECK-NEXT: tid=0 i=0 j=0
67// CHECK-NEXT: tid=0 i=1 j=0
68// CHECK-NEXT: tid=1 count=2
69// CHECK-NEXT: tid=1 i=0 j=1
70// CHECK-NEXT: tid=1 i=1 j=1
71
72// Without flatten the outer loop has 2 iterations, so each thread gets one
73// value of i and both j.
74// CHECK: without-flatten
75// CHECK-NEXT: tid=0 count=2
76// CHECK-NEXT: tid=0 i=0 j=0
77// CHECK-NEXT: tid=0 i=0 j=1
78// CHECK-NEXT: tid=1 count=2
79// CHECK-NEXT: tid=1 i=1 j=0
80// CHECK-NEXT: tid=1 i=1 j=1
void const char const char int ITT_FORMAT __itt_group_sync x void const char ITT_FORMAT __itt_group_sync s void ITT_FORMAT __itt_group_sync p void ITT_FORMAT p void ITT_FORMAT p no args __itt_suppress_mode_t unsigned int void size_t ITT_FORMAT d void ITT_FORMAT p void ITT_FORMAT p __itt_model_site __itt_model_site_instance ITT_FORMAT p __itt_model_task __itt_model_task_instance ITT_FORMAT p void ITT_FORMAT p void ITT_FORMAT p void size_t ITT_FORMAT d void ITT_FORMAT p const wchar_t ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s const char ITT_FORMAT s no args void ITT_FORMAT p size_t count
#define i
Definition kmp_stub.cpp:88
static void dump(const char *tag, int count[NThreads], int pairs[NThreads][MaxIters][2])
int main()