// compile for host-based OpenMP // icc -mkl -Ofast -no-offload -openmp -Wno-unknown-pragmas -std=c99 -vec-report3 matrix.c -o matrix.omp // compile for offload mode //icc -mkl -Ofast -offload-build -Wno-unknown-pragmas -std=c99 -vec-report3 matrix.c -o matrix.off // compile to run natively on the Xeon Phi //icc -mkl -Ofast -mmic -openmp -L /opt/intel/lib/mic -Wno-unknown-pragmas -std=c99 -vec-report3 matrix.c -o matrix.mic -liomp5 // export PHI_OMP_NUM_THREADS=240 // export PHI_KMP_AFFINITY="granularity=thread,balanced" // export LD_LIBRARY_PATH=/tmp float (*restrict A)[size] = malloc(sizeof(float)*size*size); void doMult(int size, float (* restrict A)[size], float (* restrict B)[size], float (* restrict C)[size]){ #pragma offload target(mic:0) in(A:length(size*size)) in( B:length(size*size)) out(C:length(size*size)) { #pragma omp parallel for default(none) shared(A, B, C, size) for (int i = 0; i < size; ++i) for (int k = 0; k < size; ++k) for (int j = 0; j < size; ++j) C[i][j] += A[i][k] * B[k][j]; } } // nowait_example #pragma omp parallel { #pragma omp for schedule(static) nowait for (i=0; ileft) #pragma omp task // p is firstprivate by default postorder_traverse(p->left); if (p->right) #pragma omp task // p is firstprivate by default postorder_traverse(p->right); #pragma omp taskwait process(p); } // task 2 void process(node * p) { /* do work here */ } void increment_list_items(node * head) { #pragma omp parallel { #pragma omp single { node * p = head; while (p) { #pragma omp task // p is firstprivate by default process(p); p = p->next; } } } // task 3 int fib(int n) { int i, j; if (n < 2) return n; else { #pragma omp task shared(i) i = fib(n-1); #pragma omp task shared(j) j = fib(n-2); #pragma omp taskwait return i+j; } } // Reduction #pragma omp parallel for private(i) shared(x, y, n) reduction(+:a) reduction(^:b) reduction(min:c) reduction(max:d) for (i=0; i y[i]) c = y[i]; d = fmaxf(d,x[i]); }