| Function: k_means(int, point_t*, point_t*, int*, point_t*, int, int) | Module: kmeans-clang-O3-all | Source: main.cpp:55-96 [...] | Coverage (incl. loops): 5.91% | (excl. loops): 0.00% |
|---|
| Function: k_means(int, point_t*, point_t*, int*, point_t*, int, int) | Module: kmeans-clang-O3-all | Source: main.cpp:55-96 [...] | Coverage (incl. loops): 5.91% | (excl. loops): 0.00% |
|---|
/home/fmusial/KMEANS_Benchmarks/kmeans/main.cpp: 55 - 96 |
-------------------------------------------------------------------------------- |
55: void k_means(int niters, point_t *points, point_t *centroids, int *assignment, point_t* memory, int n, int k) { |
56: for (int iter = 0; iter < niters; ++iter) { |
57: // determine nearest centroids |
58: #pragma omp parallel for |
[...] |
73: int count[k]; |
74: double sum_x[k]; |
75: double sum_y[k]; |
76: for (int j = 0; j < k; ++j) { |
77: count[j] = 0; |
78: sum_x[j] = 0.; |
79: sum_y[j] = 0.; |
80: } |
81: for (int i = 0; i < n; ++i) { |
82: count[assignment[i]]++; |
83: sum_x[assignment[i]] += points[i].x; |
84: sum_y[assignment[i]] += points[i].y; |
85: } |
86: for (int j = 0; j < k; ++j) { |
87: if (count[j] != 0) { |
88: centroids[j].x = sum_x[j] / count[j]; |
89: centroids[j].y = sum_y[j] / count[j]; |
90: } |
91: // save centroids to memory |
92: memory[(iter + 1) * k + j].x = centroids[j].x; |
93: memory[(iter + 1) * k + j].y = centroids[j].y; |
94: } |
95: } |
96: } |
0x2790 PUSH %RBP |
0x2791 MOV %RSP,%RBP |
0x2794 PUSH %R15 |
0x2796 PUSH %R14 |
0x2798 PUSH %R13 |
0x279a PUSH %R12 |
0x279c PUSH %RBX |
0x279d SUB $0x58,%RSP |
0x27a1 MOV %R8,-0x58(%RBP) |
0x27a5 MOV %FS:0x28,%RAX |
0x27ae MOV %RAX,-0x30(%RBP) |
0x27b2 MOV %RSI,-0x38(%RBP) |
0x27b6 MOV %RDX,-0x40(%RBP) |
0x27ba MOV %RCX,-0x48(%RBP) |
0x27be MOV %R9D,-0x4c(%RBP) |
0x27c2 MOV %EDI,-0x5c(%RBP) |
0x27c5 TEST %EDI,%EDI |
0x27c7 JLE 2c80 |
0x27cd MOV -0x58(%RBP),%RAX |
0x27d1 ADD $0x10,%RAX |
0x27d5 MOV %RAX,-0x70(%RBP) |
0x27d9 MOV $0x1,%EBX |
0x27de XOR %R10D,%R10D |
0x27e1 JMP 2802 |
0x27e3 NOPW %CS:(%RAX,%RAX,1) |
(4) 0x27f0 INC %R10D |
(4) 0x27f3 MOV %R11,%RSP |
(4) 0x27f6 INC %EBX |
(4) 0x27f8 CMP -0x5c(%RBP),%R10D |
(4) 0x27fc JE 2c80 |
(4) 0x2802 MOV %R10D,-0x64(%RBP) |
(4) 0x2806 LEA 0x356b(%RIP),%RDI |
(4) 0x280d MOV $0x5,%ESI |
(4) 0x2812 LEA 0x497(%RIP),%RDX |
(4) 0x2819 LEA -0x4c(%RBP),%RCX |
(4) 0x281d LEA 0x10(%RBP),%R8 |
(4) 0x2821 LEA -0x38(%RBP),%R9 |
(4) 0x2825 XOR %EAX,%EAX |
(4) 0x2827 LEA -0x48(%RBP),%R10 |
(4) 0x282b PUSH %R10 |
(4) 0x282d LEA -0x40(%RBP),%R10 |
(4) 0x2831 PUSH %R10 |
(4) 0x2833 VZEROUPPER |
(4) 0x2836 CALL 21d0 <__kmpc_fork_call@plt> |
(4) 0x283b ADD $0x10,%RSP |
(4) 0x283f MOV %RSP,%R11 |
(4) 0x2842 MOV 0x10(%RBP),%EAX |
(4) 0x2845 MOV %RSP,%R13 |
(4) 0x2848 LEA 0xf(,%RAX,4),%RAX |
(4) 0x2850 AND $-0x10,%RAX |
(4) 0x2854 SUB %RAX,%R13 |
(4) 0x2857 MOV %R13,%RSP |
(4) 0x285a MOV 0x10(%RBP),%R14D |
(4) 0x285e MOV %RSP,%R15 |
(4) 0x2861 LEA 0xf(,%R14,8),%RAX |
(4) 0x2869 AND $-0x10,%RAX |
(4) 0x286d SUB %RAX,%R15 |
(4) 0x2870 MOV %R15,%RSP |
(4) 0x2873 MOV %RSP,%R12 |
(4) 0x2876 SUB %RAX,%R12 |
(4) 0x2879 MOV %R12,%RSP |
(4) 0x287c TEST %R14D,%R14D |
(4) 0x287f JLE 28c3 |
(4) 0x2881 LEA (,%R14,4),%RDX |
(4) 0x2889 MOV %R13,%RDI |
(4) 0x288c XOR %ESI,%ESI |
(4) 0x288e MOV %R11,-0x78(%RBP) |
(4) 0x2892 CALL 20a0 <memset@plt> |
(4) 0x2897 MOV %EBX,-0x60(%RBP) |
(4) 0x289a LEA (,%R14,8),%RBX |
(4) 0x28a2 MOV %R15,%RDI |
(4) 0x28a5 XOR %ESI,%ESI |
(4) 0x28a7 MOV %RBX,%RDX |
(4) 0x28aa CALL 20a0 <memset@plt> |
(4) 0x28af MOV %R12,%RDI |
(4) 0x28b2 XOR %ESI,%ESI |
(4) 0x28b4 MOV %RBX,%RDX |
(4) 0x28b7 MOV -0x60(%RBP),%EBX |
(4) 0x28ba CALL 20a0 <memset@plt> |
(4) 0x28bf MOV -0x78(%RBP),%R11 |
(4) 0x28c3 MOVSXD -0x4c(%RBP),%RCX |
(4) 0x28c7 TEST %RCX,%RCX |
(4) 0x28ca VMOVSD 0x173e(%RIP),%XMM7 |
(4) 0x28d2 VBROADCASTSD 0x1735(%RIP),%YMM8 |
(4) 0x28db VPMOVSXBQ 0x17bc(%RIP),%YMM9 |
(4) 0x28e4 VPMOVSXBQ 0x17b7(%RIP),%YMM10 |
(4) 0x28ed VPBROADCASTQ 0x1722(%RIP),%YMM11 |
(4) 0x28f6 MOV -0x64(%RBP),%R10D |
(4) 0x28fa JLE 29cc |
(4) 0x2900 MOV -0x48(%RBP),%RDX |
(4) 0x2904 MOV -0x38(%RBP),%RAX |
(4) 0x2908 CMP $0x1,%ECX |
(4) 0x290b JNE 2920 |
(4) 0x290d XOR %ESI,%ESI |
(4) 0x290f JMP 2997 |
0x2914 NOPW %CS:(%RAX,%RAX,1) |
(4) 0x2920 MOV %ECX,%EDI |
(4) 0x2922 AND $0x7ffffffe,%EDI |
(4) 0x2928 LEA 0x18(%RAX),%R8 |
(4) 0x292c XOR %ESI,%ESI |
(4) 0x292e XCHG %AX,%AX |
(7) 0x2930 MOVSXD (%RDX,%RSI,4),%R9 |
(7) 0x2934 INCL (%R13,%R9,4) |
(7) 0x2939 VMOVSD (%R15,%R9,8),%XMM0 |
(7) 0x293f VADDSD -0x18(%R8),%XMM0,%XMM0 |
(7) 0x2945 VMOVSD %XMM0,(%R15,%R9,8) |
(7) 0x294b VMOVSD (%R12,%R9,8),%XMM0 |
(7) 0x2951 VADDSD -0x10(%R8),%XMM0,%XMM0 |
(7) 0x2957 VMOVSD %XMM0,(%R12,%R9,8) |
(7) 0x295d MOVSXD 0x4(%RDX,%RSI,4),%R9 |
(7) 0x2962 INCL (%R13,%R9,4) |
(7) 0x2967 VMOVSD (%R15,%R9,8),%XMM0 |
(7) 0x296d VADDSD -0x8(%R8),%XMM0,%XMM0 |
(7) 0x2973 VMOVSD %XMM0,(%R15,%R9,8) |
(7) 0x2979 VMOVSD (%R12,%R9,8),%XMM0 |
(7) 0x297f VADDSD (%R8),%XMM0,%XMM0 |
(7) 0x2984 VMOVSD %XMM0,(%R12,%R9,8) |
(7) 0x298a ADD $0x2,%RSI |
(7) 0x298e ADD $0x20,%R8 |
(7) 0x2992 CMP %RSI,%RDI |
(7) 0x2995 JNE 2930 |
(4) 0x2997 TEST $0x1,%CL |
(4) 0x299a JE 29cc |
(4) 0x299c MOVSXD (%RDX,%RSI,4),%RCX |
(4) 0x29a0 INCL (%R13,%RCX,4) |
(4) 0x29a5 SAL $0x4,%RSI |
(4) 0x29a9 VMOVSD (%R15,%RCX,8),%XMM0 |
(4) 0x29af VADDSD (%RAX,%RSI,1),%XMM0,%XMM0 |
(4) 0x29b4 VMOVSD %XMM0,(%R15,%RCX,8) |
(4) 0x29ba VMOVSD (%R12,%RCX,8),%XMM0 |
(4) 0x29c0 VADDSD 0x8(%RAX,%RSI,1),%XMM0,%XMM0 |
(4) 0x29c6 VMOVSD %XMM0,(%R12,%RCX,8) |
(4) 0x29cc TEST %R14D,%R14D |
(4) 0x29cf JLE 27f0 |
(4) 0x29d5 MOV -0x40(%RBP),%RAX |
(4) 0x29d9 INC %R10D |
(4) 0x29dc MOV %R14D,%ECX |
(4) 0x29df IMUL %R10D,%ECX |
(4) 0x29e3 CMP $0xc,%R14D |
(4) 0x29e7 JB 2a0c |
(4) 0x29e9 MOV %R14,%RDX |
(4) 0x29ec SAL $0x4,%RDX |
(4) 0x29f0 MOV %RCX,%RSI |
(4) 0x29f3 SAL $0x4,%RSI |
(4) 0x29f7 ADD -0x58(%RBP),%RSI |
(4) 0x29fb LEA (%RSI,%RDX,1),%RDI |
(4) 0x29ff CMP %RDI,%RAX |
(4) 0x2a02 JAE 2a55 |
(4) 0x2a04 ADD %RAX,%RDX |
(4) 0x2a07 CMP %RDX,%RSI |
(4) 0x2a0a JAE 2a55 |
(4) 0x2a0c XOR %EDX,%EDX |
(4) 0x2a0e MOV %RDX,%RDI |
(4) 0x2a11 TEST $0x1,%R14B |
(4) 0x2a15 JE 2b74 |
(4) 0x2a1b MOV (%R13,%RDX,4),%ESI |
(4) 0x2a20 TEST %ESI,%ESI |
(4) 0x2a22 JE 2b4d |
(4) 0x2a28 VCVTSI2SD %ESI,%XMM12,%XMM0 |
(4) 0x2a2c MOV %RDX,%RSI |
(4) 0x2a2f VMOVSD (%R15,%RDX,8),%XMM1 |
(4) 0x2a35 VMOVHPD (%R12,%RDX,8),%XMM1,%XMM1 |
(4) 0x2a3b SAL $0x4,%RSI |
(4) 0x2a3f VDIVSD %XMM0,%XMM7,%XMM0 |
(4) 0x2a43 VMOVDDUP %XMM0,%XMM0 |
(4) 0x2a47 VMULPD %XMM0,%XMM1,%XMM0 |
(4) 0x2a4b VMOVUPD %XMM0,(%RAX,%RSI,1) |
(4) 0x2a50 JMP 2b59 |
(4) 0x2a55 MOV %R14D,%EDX |
(4) 0x2a58 AND $0x7ffffffc,%EDX |
(4) 0x2a5e MOV %R14D,%ESI |
(4) 0x2a61 IMUL %EBX,%ESI |
(4) 0x2a64 SAL $0x4,%RSI |
(4) 0x2a68 ADD -0x58(%RBP),%RSI |
(4) 0x2a6c LEA (%R14,%R14,1),%EDI |
(4) 0x2a70 AND $-0x8,%EDI |
(4) 0x2a73 VPBROADCASTQ %RAX,%YMM0 |
(4) 0x2a79 XOR %R8D,%R8D |
(4) 0x2a7c VPMOVSXBQ 0x1623(%RIP),%YMM1 |
(4) 0x2a85 NOPW %CS:(%RAX,%RAX,1) |
(6) 0x2a90 VMOVDQA (%R13,%R8,2),%XMM2 |
(6) 0x2a97 VPTESTMD %XMM2,%XMM2,%K2 |
(6) 0x2a9d VMOVUPD (%R15,%R8,4),%YMM3{%K2}{z} |
(6) 0x2aa4 VCVTDQ2PD %XMM2,%YMM4 |
(6) 0x2aa8 VDIVPD %YMM4,%YMM8,%YMM4 |
(6) 0x2aac VMULPD %YMM4,%YMM3,%YMM3 |
(6) 0x2ab0 VPSLLQ $0x4,%YMM1,%YMM5 |
(6) 0x2ab5 KMOVQ %K2,%K1 |
(6) 0x2aba VSCATTERQPD %YMM3,(%RAX,%YMM5,1){%K1} |
(6) 0x2ac1 VPTESTNMD %XMM2,%XMM2,%K1 |
(6) 0x2ac7 VPADDQ %YMM5,%YMM0,%YMM2 |
(6) 0x2acb VMOVUPD (%R12,%R8,4),%YMM6{%K2}{z} |
(6) 0x2ad2 VMULPD %YMM4,%YMM6,%YMM4 |
(6) 0x2ad6 VSCATTERQPD %YMM4,0x8(,%YMM2,1){%K2} |
(6) 0x2ae1 VXORPD %XMM6,%XMM6,%XMM6 |
(6) 0x2ae5 KMOVQ %K1,%K2 |
(6) 0x2aea VGATHERQPD (%RAX,%YMM5,1),%YMM6{%K2} |
(6) 0x2af1 VXORPD %XMM5,%XMM5,%XMM5 |
(6) 0x2af5 KMOVQ %K1,%K2 |
(6) 0x2afa VGATHERQPD 0x8(,%YMM2,1),%YMM5{%K2} |
(6) 0x2b05 VMOVAPD %YMM5,%YMM4{%K1} |
(6) 0x2b0b VMOVAPD %YMM6,%YMM3{%K1} |
(6) 0x2b11 VMOVAPD %YMM3,%YMM2 |
(6) 0x2b15 VPERMT2PD %YMM4,%YMM10,%YMM3 |
(6) 0x2b1b VMOVUPD %YMM3,0x20(%RSI,%R8,8) |
(6) 0x2b22 VPERMT2PD %YMM4,%YMM9,%YMM2 |
(6) 0x2b28 VMOVUPD %YMM2,(%RSI,%R8,8) |
(6) 0x2b2e VPADDQ %YMM1,%YMM11,%YMM1 |
(6) 0x2b32 ADD $0x8,%R8 |
(6) 0x2b36 CMP %R8,%RDI |
(6) 0x2b39 JNE 2a90 |
(4) 0x2b3f CMP %R14D,%EDX |
(4) 0x2b42 JE 27f3 |
(4) 0x2b48 JMP 2a0e |
(4) 0x2b4d MOV %RDX,%RSI |
(4) 0x2b50 SAL $0x4,%RSI |
(4) 0x2b54 VMOVUPD (%RAX,%RSI,1),%XMM0 |
(4) 0x2b59 MOV %RDX,%RSI |
(4) 0x2b5c SAL $0x4,%RSI |
(4) 0x2b60 ADD -0x58(%RBP),%RSI |
(4) 0x2b64 SAL $0x4,%RCX |
(4) 0x2b68 VMOVUPD %XMM0,(%RCX,%RSI,1) |
(4) 0x2b6d MOV %RDX,%RDI |
(4) 0x2b70 OR $0x1,%RDI |
(4) 0x2b74 LEA -0x1(%R14),%RCX |
(4) 0x2b78 CMP %RCX,%RDX |
(4) 0x2b7b JE 27f3 |
(4) 0x2b81 MOV %RDI,%RDX |
(4) 0x2b84 SAL $0x4,%RDX |
(4) 0x2b88 MOV %R14D,%ECX |
(4) 0x2b8b IMUL %EBX,%ECX |
(4) 0x2b8e SAL $0x4,%RCX |
(4) 0x2b92 MOV -0x70(%RBP),%RSI |
(4) 0x2b96 ADD %RDX,%RSI |
(4) 0x2b99 ADD %RSI,%RCX |
(4) 0x2b9c SUB %RDI,%R14 |
(4) 0x2b9f ADD %RDX,%RAX |
(4) 0x2ba2 ADD $0x10,%RAX |
(4) 0x2ba6 LEA (%R12,%RDI,8),%RDX |
(4) 0x2baa ADD $0x8,%RDX |
(4) 0x2bae LEA (%R15,%RDI,8),%RSI |
(4) 0x2bb2 ADD $0x8,%RSI |
(4) 0x2bb6 LEA 0x4(,%RDI,4),%RDI |
(4) 0x2bbe ADD %R13,%RDI |
(4) 0x2bc1 XOR %R8D,%R8D |
(4) 0x2bc4 JMP 2c3a |
0x2bc6 NOPW %CS:(%RAX,%RAX,1) |
(5) 0x2bd0 VMOVUPD -0x10(%RAX,%R8,4),%XMM0 |
(5) 0x2bd7 VMOVUPD %XMM0,-0x10(%RCX,%R8,4) |
(5) 0x2bde MOV (%RDI,%R8,1),%R9D |
(5) 0x2be2 TEST %R9D,%R9D |
(5) 0x2be5 JE 2c20 |
(5) 0x2be7 VCVTSI2SD %R9D,%XMM12,%XMM0 |
(5) 0x2bec VMOVSD (%RSI,%R8,2),%XMM1 |
(5) 0x2bf2 VMOVHPD (%RDX,%R8,2),%XMM1,%XMM1 |
(5) 0x2bf8 VDIVSD %XMM0,%XMM7,%XMM0 |
(5) 0x2bfc VMOVDDUP %XMM0,%XMM0 |
(5) 0x2c00 VMULPD %XMM0,%XMM1,%XMM0 |
(5) 0x2c04 VMOVUPD %XMM0,(%RAX,%R8,4) |
(5) 0x2c0a VMOVUPD %XMM0,(%RCX,%R8,4) |
(5) 0x2c10 ADD $0x8,%R8 |
(5) 0x2c14 ADD $-0x2,%R14 |
(5) 0x2c18 JNE 2c3a |
(4) 0x2c1a JMP 27f3 |
0x2c1f NOP |
(5) 0x2c20 VMOVUPD (%RAX,%R8,4),%XMM0 |
(5) 0x2c26 VMOVUPD %XMM0,(%RCX,%R8,4) |
(5) 0x2c2c ADD $0x8,%R8 |
(5) 0x2c30 ADD $-0x2,%R14 |
(5) 0x2c34 JE 27f3 |
(5) 0x2c3a MOV -0x4(%RDI,%R8,1),%R9D |
(5) 0x2c3f TEST %R9D,%R9D |
(5) 0x2c42 JE 2bd0 |
(5) 0x2c44 VCVTSI2SD %R9D,%XMM12,%XMM0 |
(5) 0x2c49 VMOVSD -0x8(%RSI,%R8,2),%XMM1 |
(5) 0x2c50 VMOVHPD -0x8(%RDX,%R8,2),%XMM1,%XMM1 |
(5) 0x2c57 VDIVSD %XMM0,%XMM7,%XMM0 |
(5) 0x2c5b VMOVDDUP %XMM0,%XMM0 |
(5) 0x2c5f VMULPD %XMM0,%XMM1,%XMM0 |
(5) 0x2c63 VMOVUPD %XMM0,-0x10(%RAX,%R8,4) |
(5) 0x2c6a VMOVUPD %XMM0,-0x10(%RCX,%R8,4) |
(5) 0x2c71 MOV (%RDI,%R8,1),%R9D |
(5) 0x2c75 TEST %R9D,%R9D |
(5) 0x2c78 JNE 2be7 |
(5) 0x2c7e JMP 2c20 |
0x2c80 MOV %FS:0x28,%RAX |
0x2c89 CMP -0x30(%RBP),%RAX |
0x2c8d JNE 2ca1 |
0x2c8f LEA -0x28(%RBP),%RSP |
0x2c93 POP %RBX |
0x2c94 POP %R12 |
0x2c96 POP %R13 |
0x2c98 POP %R14 |
0x2c9a POP %R15 |
0x2c9c POP %RBP |
0x2c9d VZEROUPPER |
0x2ca0 RET |
0x2ca1 VZEROUPPER |
0x2ca4 CALL 2110 <__stack_chk_fail@plt> |
0x2ca9 NOPL (%RAX) |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:125 | kmeans-clang-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-clang-O3-all |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:125 | kmeans-clang-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-clang-O3-all |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:125 | kmeans-clang-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-clang-O3-all |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:125 | kmeans-clang-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-clang-O3-all |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:125 | kmeans-clang-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-clang-O3-all |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:125 | kmeans-clang-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-clang-O3-all |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| Path / |
The code analyzed by CQA in that panel excludes loops and represents 0.00% of application time for run run_1_thread
| Source file and lines | main.cpp:55-96 |
| Module | kmeans-clang-O3-all |
| nb instructions | 43 |
| nb uops | 50 |
| loop length | 167 |
| used x86 registers | 15 |
| used mmx registers | 0 |
| used xmm registers | 0 |
| used ymm registers | 0 |
| used zmm registers | 0 |
| nb stack references | 9 |
| micro-operation queue | 12.50 cycles |
| front end | 12.50 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | |
|---|---|---|---|---|---|---|---|---|
| uops | 2.75 | 2.75 | 8.67 | 8.67 | 15.00 | 2.50 | 3.00 | 8.67 |
| cycles | 2.75 | 2.75 | 8.67 | 8.67 | 15.00 | 2.50 | 3.00 | 8.67 |
| Cycles executing div or sqrt instructions | NA |
| FE+BE cycles | 14.12 |
| Stall cycles | 2.95 |
| SB full (events) | 4.54 |
| Front-end | 12.50 |
| Dispatch | 15.00 |
| Overall L1 | 15.00 |
| all | 14% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 40% |
| all | 12% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | 10% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 13% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| PUSH %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| PUSH %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| SUB $0x58,%RSP | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (12.5%) |
| MOV %R8,-0x58(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %FS:0x28,%RAX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| MOV %RAX,-0x30(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RSI,-0x38(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RDX,-0x40(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RCX,-0x48(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R9D,-0x4c(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (6.3%) |
| MOV %EDI,-0x5c(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (6.3%) |
| TEST %EDI,%EDI | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (6.3%) |
| JLE 2c80 <_Z7k_meansiP7point_tS0_PiS0_ii+0x4f0> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50-1 | N/A |
| MOV -0x58(%RBP),%RAX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| ADD $0x10,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0x70(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV $0x1,%EBX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (6.3%) |
| XOR %R10D,%R10D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (6.3%) |
| JMP 2802 <_Z7k_meansiP7point_tS0_PiS0_ii+0x72> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 1-2 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| MOV %FS:0x28,%RAX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| CMP -0x30(%RBP),%RAX | 1 | 0.25 | 0.25 | 0.50 | 0.50 | 0 | 0.25 | 0.25 | 0 | 1 | 0.50 | N/A |
| JNE 2ca1 <_Z7k_meansiP7point_tS0_PiS0_ii+0x511> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50-1 | N/A |
| LEA -0x28(%RBP),%RSP | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| POP %RBX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R12 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R13 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R14 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R15 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %RBP | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| VZEROUPPER | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | vect (25.0%) |
| RET | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 1 | 0.33 | 0 | 1 | N/A |
| VZEROUPPER | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | vect (25.0%) |
| CALL 2110 <__stack_chk_fail@plt> | 2 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 1 | 0.33 | 0 | 1 | N/A |
| NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
The code analyzed by CQA in that panel excludes loops and represents 0.00% of application time for run run_1_thread
| Source file and lines | main.cpp:55-96 |
| Module | kmeans-clang-O3-all |
| nb instructions | 43 |
| nb uops | 50 |
| loop length | 167 |
| used x86 registers | 15 |
| used mmx registers | 0 |
| used xmm registers | 0 |
| used ymm registers | 0 |
| used zmm registers | 0 |
| nb stack references | 9 |
| micro-operation queue | 12.50 cycles |
| front end | 12.50 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | |
|---|---|---|---|---|---|---|---|---|
| uops | 2.75 | 2.75 | 8.67 | 8.67 | 15.00 | 2.50 | 3.00 | 8.67 |
| cycles | 2.75 | 2.75 | 8.67 | 8.67 | 15.00 | 2.50 | 3.00 | 8.67 |
| Cycles executing div or sqrt instructions | NA |
| FE+BE cycles | 14.12 |
| Stall cycles | 2.95 |
| SB full (events) | 4.54 |
| Front-end | 12.50 |
| Dispatch | 15.00 |
| Overall L1 | 15.00 |
| all | 14% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 40% |
| all | 12% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | 10% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 13% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| PUSH %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| PUSH %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| SUB $0x58,%RSP | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (12.5%) |
| MOV %R8,-0x58(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %FS:0x28,%RAX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| MOV %RAX,-0x30(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RSI,-0x38(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RDX,-0x40(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RCX,-0x48(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R9D,-0x4c(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (6.3%) |
| MOV %EDI,-0x5c(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (6.3%) |
| TEST %EDI,%EDI | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (6.3%) |
| JLE 2c80 <_Z7k_meansiP7point_tS0_PiS0_ii+0x4f0> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50-1 | N/A |
| MOV -0x58(%RBP),%RAX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| ADD $0x10,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0x70(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV $0x1,%EBX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (6.3%) |
| XOR %R10D,%R10D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (6.3%) |
| JMP 2802 <_Z7k_meansiP7point_tS0_PiS0_ii+0x72> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 1-2 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| MOV %FS:0x28,%RAX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| CMP -0x30(%RBP),%RAX | 1 | 0.25 | 0.25 | 0.50 | 0.50 | 0 | 0.25 | 0.25 | 0 | 1 | 0.50 | N/A |
| JNE 2ca1 <_Z7k_meansiP7point_tS0_PiS0_ii+0x511> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50-1 | N/A |
| LEA -0x28(%RBP),%RSP | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| POP %RBX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R12 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R13 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R14 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R15 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %RBP | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| VZEROUPPER | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | vect (25.0%) |
| RET | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 1 | 0.33 | 0 | 1 | N/A |
| VZEROUPPER | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | vect (25.0%) |
| CALL 2110 <__stack_chk_fail@plt> | 2 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 1 | 0.33 | 0 | 1 | N/A |
| NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| Run run_1_thread | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: closeOMP_NUM_THREADS: 1 |
|---|---|
| Run run_2_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: closeOMP_NUM_THREADS: 2 |
| Run run_4_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: closeOMP_NUM_THREADS: 4 |
| Run run_8_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: closeOMP_NUM_THREADS: 8 |
| Run run_16_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: closeOMP_NUM_THREADS: 16 |
| Run run_26_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: closeOMP_NUM_THREADS: 26 |
| (run_1_thread) Efficiency | (run_1_thread) Potential Speed-Up (%) | (run_2_threads) Efficiency | (run_2_threads) Potential Speed-Up (%) | (run_4_threads) Efficiency | (run_4_threads) Potential Speed-Up (%) | (run_8_threads) Efficiency | (run_8_threads) Potential Speed-Up (%) | (run_16_threads) Efficiency | (run_16_threads) Potential Speed-Up (%) | (run_26_threads) Efficiency | (run_26_threads) Potential Speed-Up (%) |
|---|---|---|---|---|---|---|---|---|---|---|---|
| 1 | 0 | 0.97 | 0.17 | 0.92 | 0.46 | 0.84 | 0.82 | 0.73 | 1.14 | 0.66 | 1.22 |
| Run | Number of threads | Efficiency (ideal is 1) | Speedup | Ideal Speedup | Time (s) | Coverage (%) |
|---|---|---|---|---|---|---|
| run_1_thread | 1 | 1 | 1 | 1 | 11.185000419617 | 5.9106397628784 |
| run_2_threads | 1 | 0.97 | 1.94 | 2 | 11.159998893738 | 5.7528734207153 |
| run_4_threads | 1 | 0.92 | 3.66 | 4 | 11.159998893738 | 5.4840292930603 |
| run_8_threads | 1 | 0.84 | 6.68 | 8 | 11.154999732971 | 5.0074067115784 |
| run_16_threads | 1 | 0.73 | 11.73 | 16 | 11.159997940063 | 4.2754521369934 |
| run_26_threads | 1 | 0.66 | 17.2 | 26 | 11.154999732971 | 3.6116101741791 |
| Name | Coverage (%) | Time (s) |
|---|---|---|
| ▼k_means(int, point_t*, point_t*, int*, point_t*, int, int)– | 5.91 | 11.19 |
| ▼Loop 4 - main.cpp:56-93 - kmeans-clang-O3-all– | 0.00 | 0.00 |
| ○Loop 7 - main.cpp:81-84 - kmeans-clang-O3-all | 5.91 | 11.19 |
| ○Loop 5 - main.cpp:86-92 - kmeans-clang-O3-all | 0.00 | 0.00 |
| ○Loop 6 - main.cpp:86-93 - kmeans-clang-O3-all | 0.00 | 0.00 |
