| Function: k_means(int, point_t*, point_t*, int*, point_t*, int, int) | Module: kmeans-gcc-Ofast | Source: main.cpp:67-108 [...] | Coverage (incl. loops): 4.63% | (excl. loops): 0.00% |
|---|
| Function: k_means(int, point_t*, point_t*, int*, point_t*, int, int) | Module: kmeans-gcc-Ofast | Source: main.cpp:67-108 [...] | Coverage (incl. loops): 4.63% | (excl. loops): 0.00% |
|---|
/home/fmusial/KMEANS_Benchmarks/kmeans/main.cpp: 67 - 108 |
-------------------------------------------------------------------------------- |
67: void k_means(int niters, point_t *points, point_t *centroids, int *assignment, point_t* memory, int n, int k) { |
68: for (int iter = 0; iter < niters; ++iter) { |
69: // determine nearest centroids |
70: #pragma omp parallel for |
[...] |
85: int count[k]; |
86: double sum_x[k]; |
87: double sum_y[k]; |
88: for (int j = 0; j < k; ++j) { |
89: count[j] = 0; |
90: sum_x[j] = 0.; |
91: sum_y[j] = 0.; |
92: } |
93: for (int i = 0; i < n; ++i) { |
94: count[assignment[i]]++; |
95: sum_x[assignment[i]] += points[i].x; |
96: sum_y[assignment[i]] += points[i].y; |
97: } |
98: for (int j = 0; j < k; ++j) { |
99: if (count[j] != 0) { |
100: centroids[j].x = sum_x[j] / count[j]; |
101: centroids[j].y = sum_y[j] / count[j]; |
102: } |
103: // save centroids to memory |
104: memory[(iter + 1) * k + j].x = centroids[j].x; |
105: memory[(iter + 1) * k + j].y = centroids[j].y; |
106: } |
107: } |
108: } |
0x2b80 PUSH %RBP |
0x2b81 MOV %RSP,%RBP |
0x2b84 PUSH %R15 |
0x2b86 PUSH %R14 |
0x2b88 PUSH %R13 |
0x2b8a MOV %R9D,%R13D |
0x2b8d PUSH %R12 |
0x2b8f PUSH %RBX |
0x2b90 SUB $0xb8,%RSP |
0x2b97 MOV %RSI,-0xc0(%RBP) |
0x2b9e MOV %FS:0x28,%R9 |
0x2ba7 MOV %R9,-0x38(%RBP) |
0x2bab MOV 0x10(%RBP),%R9D |
0x2baf TEST %EDI,%EDI |
0x2bb1 JLE 2dde |
0x2bb7 LEA -0x1(%R9),%EAX |
0x2bbb MOV %EDI,%R11D |
0x2bbe MOVSXD %R9D,%R14 |
0x2bc1 MOV %RDX,%R10 |
0x2bc4 INC %RAX |
0x2bc7 VMOVQ %RSI,%XMM4 |
0x2bcc VMOVD %R13D,%XMM6 |
0x2bd1 MOV %R9D,-0x68(%RBP) |
0x2bd5 LEA (,%RAX,4),%RDI |
0x2bdd SAL $0x3,%RAX |
0x2be1 VPINSRQ $0x1,%RDX,%XMM4,%XMM3 |
0x2be7 MOVL $0,-0x64(%RBP) |
0x2bee MOV %RAX,-0x78(%RBP) |
0x2bf2 LEA -0x60(%RBP),%RAX |
0x2bf6 LEA 0xf(,%R14,4),%RDX |
0x2bfe VPINSRD $0x1,%R9D,%XMM6,%XMM5 |
0x2c04 MOV %RAX,-0xb0(%RBP) |
0x2c0b LEA 0xf(,%R14,8),%RAX |
0x2c13 SHR $0x4,%RDX |
0x2c17 MOV %RCX,%R12 |
0x2c1a SHR $0x4,%RAX |
0x2c1e SAL $0x4,%RDX |
0x2c22 MOV %RDI,-0xc8(%RBP) |
0x2c29 MOV %R13D,%R15D |
0x2c2c SAL $0x4,%RAX |
0x2c30 MOV %RDX,-0xa0(%RBP) |
0x2c37 MOV %RAX,-0xa8(%RBP) |
0x2c3e MOV %R11D,-0xb4(%RBP) |
0x2c45 MOV %R10,-0xd0(%RBP) |
0x2c4c MOV %R8,-0xd8(%RBP) |
0x2c53 VMOVDQA %XMM3,-0x90(%RBP) |
0x2c5b VMOVQ %XMM5,-0x98(%RBP) |
0x2c63 NOPL (%RAX,%RAX,1) |
(5) 0x2c68 MOV -0x98(%RBP),%RAX |
(5) 0x2c6f VMOVDQA -0x90(%RBP),%XMM3 |
(5) 0x2c77 XOR %EDX,%EDX |
(5) 0x2c79 XOR %ECX,%ECX |
(5) 0x2c7b MOV -0xb0(%RBP),%RSI |
(5) 0x2c82 LEA -0x3a9(%RIP),%RDI |
(5) 0x2c89 MOV %RSP,-0x70(%RBP) |
(5) 0x2c8d MOV %RAX,-0x48(%RBP) |
(5) 0x2c91 MOV %R12,-0x50(%RBP) |
(5) 0x2c95 VMOVDQA %XMM3,-0x60(%RBP) |
(5) 0x2c9a CALL 21e0 <GOMP_parallel@plt> |
(5) 0x2c9f MOV -0xa8(%RBP),%RAX |
(5) 0x2ca6 SUB -0xa0(%RBP),%RSP |
(5) 0x2cad MOV %RSP,%RBX |
(5) 0x2cb0 MOV 0x10(%RBP),%EDX |
(5) 0x2cb3 SUB %RAX,%RSP |
(5) 0x2cb6 MOV %RSP,%R13 |
(5) 0x2cb9 SUB %RAX,%RSP |
(5) 0x2cbc INCL -0x64(%RBP) |
(5) 0x2cbf MOV %RSP,%R11 |
(5) 0x2cc2 TEST %EDX,%EDX |
(5) 0x2cc4 JLE 2e00 |
(5) 0x2cca MOV -0xc8(%RBP),%RDX |
(5) 0x2cd1 XOR %ESI,%ESI |
(5) 0x2cd3 MOV %RBX,%RDI |
(5) 0x2cd6 MOV %RSP,-0x80(%RBP) |
(5) 0x2cda CALL 20c0 <memset@plt> |
(5) 0x2cdf MOV -0x78(%RBP),%RDX |
(5) 0x2ce3 XOR %ESI,%ESI |
(5) 0x2ce5 MOV %R13,%RDI |
(5) 0x2ce8 CALL 20c0 <memset@plt> |
(5) 0x2ced MOV -0x78(%RBP),%RDX |
(5) 0x2cf1 MOV -0x80(%RBP),%RDI |
(5) 0x2cf5 XOR %ESI,%ESI |
(5) 0x2cf7 CALL 20c0 <memset@plt> |
(5) 0x2cfc MOV %RAX,%R11 |
(5) 0x2cff TEST %R15D,%R15D |
(5) 0x2d02 JLE 2d4e |
(5) 0x2d04 MOV -0xc0(%RBP),%RSI |
(5) 0x2d0b XOR %EDX,%EDX |
(5) 0x2d0d NOPL (%RAX) |
(4) 0x2d10 MOVSXD (%R12,%RDX,4),%RAX |
(4) 0x2d14 INC %RDX |
(4) 0x2d17 ADD $0x10,%RSI |
(4) 0x2d1b VMOVSD (%R13,%RAX,8),%XMM0 |
(4) 0x2d22 INCL (%RBX,%RAX,4) |
(4) 0x2d25 VADDSD -0x10(%RSI),%XMM0,%XMM0 |
(4) 0x2d2a VMOVSD %XMM0,(%R13,%RAX,8) |
(4) 0x2d31 VMOVSD (%R11,%RAX,8),%XMM0 |
(4) 0x2d37 VADDSD -0x8(%RSI),%XMM0,%XMM0 |
(4) 0x2d3c VMOVSD %XMM0,(%R11,%RAX,8) |
(4) 0x2d42 CMP %EDX,%R15D |
(4) 0x2d45 JG 2d10 |
(5) 0x2d47 MOV 0x10(%RBP),%EAX |
(5) 0x2d4a TEST %EAX,%EAX |
(5) 0x2d4c JLE 2dc5 |
(5) 0x2d4e MOVSXD -0x68(%RBP),%RSI |
(5) 0x2d52 MOV -0xd0(%RBP),%RDX |
(5) 0x2d59 XOR %EAX,%EAX |
(5) 0x2d5b SAL $0x4,%RSI |
(5) 0x2d5f ADD -0xd8(%RBP),%RSI |
(5) 0x2d66 JMP 2d88 |
0x2d68 NOPL (%RAX,%RAX,1) |
(6) 0x2d70 VMOVUPD (%RDX),%XMM0 |
(6) 0x2d74 INC %RAX |
(6) 0x2d77 VMOVUPD %XMM0,(%RSI) |
(6) 0x2d7b CMP %RAX,%R14 |
(6) 0x2d7e JE 2dc5 |
(6) 0x2d80 ADD $0x10,%RSI |
(6) 0x2d84 ADD $0x10,%RDX |
(6) 0x2d88 MOV (%RBX,%RAX,4),%ECX |
(6) 0x2d8b TEST %ECX,%ECX |
(6) 0x2d8d JE 2d70 |
(6) 0x2d8f VMOVSD (%R13,%RAX,8),%XMM0 |
(6) 0x2d96 VXORPD %XMM7,%XMM7,%XMM7 |
(6) 0x2d9a VCVTSI2SD %ECX,%XMM7,%XMM1 |
(6) 0x2d9e VDIVSD %XMM1,%XMM0,%XMM2 |
(6) 0x2da2 VMOVDDUP %XMM1,%XMM1 |
(6) 0x2da6 VMOVSD %XMM2,(%RDX) |
(6) 0x2daa VMOVHPD (%R11,%RAX,8),%XMM0,%XMM0 |
(6) 0x2db0 VDIVPD %XMM1,%XMM0,%XMM0 |
(6) 0x2db4 INC %RAX |
(6) 0x2db7 VMOVHPD %XMM0,0x8(%RDX) |
(6) 0x2dbc VMOVUPD %XMM0,(%RSI) |
(6) 0x2dc0 CMP %RAX,%R14 |
(6) 0x2dc3 JNE 2d80 |
(5) 0x2dc5 MOV 0x10(%RBP),%EDI |
(5) 0x2dc8 MOV -0x70(%RBP),%RSP |
(5) 0x2dcc ADD %EDI,-0x68(%RBP) |
(5) 0x2dcf MOV -0x64(%RBP),%EDI |
(5) 0x2dd2 CMP %EDI,-0xb4(%RBP) |
(5) 0x2dd8 JNE 2c68 |
0x2dde MOV -0x38(%RBP),%RAX |
0x2de2 SUB %FS:0x28,%RAX |
0x2deb JNE 2e0b |
0x2ded LEA -0x28(%RBP),%RSP |
0x2df1 POP %RBX |
0x2df2 POP %R12 |
0x2df4 POP %R13 |
0x2df6 POP %R14 |
0x2df8 POP %R15 |
0x2dfa POP %RBP |
0x2dfb RET |
0x2dfc NOPL (%RAX) |
(5) 0x2e00 TEST %R15D,%R15D |
(5) 0x2e03 JG 2d04 |
(5) 0x2e09 JMP 2dc5 |
0x2e0b CALL 2130 <__stack_chk_fail@plt> |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:139 | kmeans-gcc-Ofast |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | basic_string.h:896 | kmeans-gcc-Ofast |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:139 | kmeans-gcc-Ofast |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | basic_string.h:896 | kmeans-gcc-Ofast |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:139 | kmeans-gcc-Ofast |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | basic_string.h:896 | kmeans-gcc-Ofast |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:139 | kmeans-gcc-Ofast |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | basic_string.h:896 | kmeans-gcc-Ofast |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:139 | kmeans-gcc-Ofast |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | basic_string.h:896 | kmeans-gcc-Ofast |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| Path / |
The code analyzed by CQA in that panel excludes loops and represents 0.00% of application time for run run_1_thread
| Source file and lines | main.cpp:67-108 |
| Module | kmeans-gcc-Ofast |
| nb instructions | 62 |
| nb uops | 65 |
| loop length | 279 |
| used x86 registers | 16 |
| used mmx registers | 0 |
| used xmm registers | 4 |
| used ymm registers | 0 |
| used zmm registers | 0 |
| nb stack references | 17 |
| micro-operation queue | 16.25 cycles |
| front end | 16.25 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | |
|---|---|---|---|---|---|---|---|---|
| uops | 6.50 | 6.50 | 10.67 | 10.67 | 21.00 | 6.50 | 6.50 | 10.67 |
| cycles | 6.50 | 6.50 | 10.67 | 10.67 | 21.00 | 6.50 | 6.50 | 10.67 |
| Cycles executing div or sqrt instructions | NA |
| FE+BE cycles | 20.10 |
| Stall cycles | 3.67 |
| SB full (events) | 5.50 |
| Front-end | 16.25 |
| Dispatch | 21.00 |
| Overall L1 | 21.00 |
| all | 3% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | 7% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 0% |
| all | 10% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 9% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| PUSH %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| PUSH %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| MOV %R9D,%R13D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (6.3%) |
| PUSH %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| SUB $0xb8,%RSP | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (12.5%) |
| MOV %RSI,-0xc0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %FS:0x28,%R9 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| MOV %R9,-0x38(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV 0x10(%RBP),%R9D | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| TEST %EDI,%EDI | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (6.3%) |
| JLE 2dde <_Z7k_meansiP7point_tS0_PiS0_ii+0x25e> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50-1 | N/A |
| LEA -0x1(%R9),%EAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %EDI,%R11D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (6.3%) |
| MOVSXD %R9D,%R14 | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RDX,%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (12.5%) |
| INC %RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| VMOVQ %RSI,%XMM4 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 1 | 1 | scal (12.5%) |
| VMOVD %R13D,%XMM6 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 2 | 1 | scal (6.3%) |
| MOV %R9D,-0x68(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (6.3%) |
| LEA (,%RAX,4),%RDI | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| SAL $0x3,%RAX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | N/A |
| VPINSRQ $0x1,%RDX,%XMM4,%XMM3 | 2 | 0 | 0 | 0 | 0 | 0 | 2 | 0 | 0 | 3 | 2 | scal (12.5%) |
| MOVL $0,-0x64(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 2 | 1 | scal (6.3%) |
| MOV %RAX,-0x78(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA -0x60(%RBP),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| LEA 0xf(,%R14,4),%RDX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| VPINSRD $0x1,%R9D,%XMM6,%XMM5 | 2 | 0 | 0 | 0 | 0 | 0 | 2 | 0 | 0 | 3 | 2 | scal (6.3%) |
| MOV %RAX,-0xb0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA 0xf(,%R14,8),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| SHR $0x4,%RDX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | scal (12.5%) |
| MOV %RCX,%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (12.5%) |
| SHR $0x4,%RAX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | N/A |
| SAL $0x4,%RDX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | scal (12.5%) |
| MOV %RDI,-0xc8(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R13D,%R15D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (6.3%) |
| SAL $0x4,%RAX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | N/A |
| MOV %RDX,-0xa0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RAX,-0xa8(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R11D,-0xb4(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (6.3%) |
| MOV %R10,-0xd0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R8,-0xd8(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| VMOVDQA %XMM3,-0x90(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 4 | 1 | vect (25.0%) |
| VMOVQ %XMM5,-0x98(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| MOV -0x38(%RBP),%RAX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| SUB %FS:0x28,%RAX | 1 | 0.25 | 0.25 | 0.50 | 0.50 | 0 | 0.25 | 0.25 | 0 | 1 | 0.50 | N/A |
| JNE 2e0b <_Z7k_meansiP7point_tS0_PiS0_ii+0x28b> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50-1 | N/A |
| LEA -0x28(%RBP),%RSP | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| POP %RBX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R12 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R13 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R14 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R15 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %RBP | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| RET | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 1 | 0.33 | 0 | 1 | N/A |
| NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| CALL 2130 <__stack_chk_fail@plt> | 2 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 1 | 0.33 | 0 | 1 | N/A |
The code analyzed by CQA in that panel excludes loops and represents 0.00% of application time for run run_1_thread
| Source file and lines | main.cpp:67-108 |
| Module | kmeans-gcc-Ofast |
| nb instructions | 62 |
| nb uops | 65 |
| loop length | 279 |
| used x86 registers | 16 |
| used mmx registers | 0 |
| used xmm registers | 4 |
| used ymm registers | 0 |
| used zmm registers | 0 |
| nb stack references | 17 |
| micro-operation queue | 16.25 cycles |
| front end | 16.25 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | |
|---|---|---|---|---|---|---|---|---|
| uops | 6.50 | 6.50 | 10.67 | 10.67 | 21.00 | 6.50 | 6.50 | 10.67 |
| cycles | 6.50 | 6.50 | 10.67 | 10.67 | 21.00 | 6.50 | 6.50 | 10.67 |
| Cycles executing div or sqrt instructions | NA |
| FE+BE cycles | 20.10 |
| Stall cycles | 3.67 |
| SB full (events) | 5.50 |
| Front-end | 16.25 |
| Dispatch | 21.00 |
| Overall L1 | 21.00 |
| all | 3% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | 7% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 0% |
| all | 10% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 9% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| PUSH %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| PUSH %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| MOV %R9D,%R13D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (6.3%) |
| PUSH %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| SUB $0xb8,%RSP | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (12.5%) |
| MOV %RSI,-0xc0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %FS:0x28,%R9 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| MOV %R9,-0x38(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV 0x10(%RBP),%R9D | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| TEST %EDI,%EDI | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (6.3%) |
| JLE 2dde <_Z7k_meansiP7point_tS0_PiS0_ii+0x25e> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50-1 | N/A |
| LEA -0x1(%R9),%EAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %EDI,%R11D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (6.3%) |
| MOVSXD %R9D,%R14 | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RDX,%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (12.5%) |
| INC %RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| VMOVQ %RSI,%XMM4 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 1 | 1 | scal (12.5%) |
| VMOVD %R13D,%XMM6 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 2 | 1 | scal (6.3%) |
| MOV %R9D,-0x68(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (6.3%) |
| LEA (,%RAX,4),%RDI | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| SAL $0x3,%RAX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | N/A |
| VPINSRQ $0x1,%RDX,%XMM4,%XMM3 | 2 | 0 | 0 | 0 | 0 | 0 | 2 | 0 | 0 | 3 | 2 | scal (12.5%) |
| MOVL $0,-0x64(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 2 | 1 | scal (6.3%) |
| MOV %RAX,-0x78(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA -0x60(%RBP),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| LEA 0xf(,%R14,4),%RDX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| VPINSRD $0x1,%R9D,%XMM6,%XMM5 | 2 | 0 | 0 | 0 | 0 | 0 | 2 | 0 | 0 | 3 | 2 | scal (6.3%) |
| MOV %RAX,-0xb0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA 0xf(,%R14,8),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| SHR $0x4,%RDX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | scal (12.5%) |
| MOV %RCX,%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (12.5%) |
| SHR $0x4,%RAX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | N/A |
| SAL $0x4,%RDX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | scal (12.5%) |
| MOV %RDI,-0xc8(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R13D,%R15D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (6.3%) |
| SAL $0x4,%RAX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | N/A |
| MOV %RDX,-0xa0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RAX,-0xa8(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R11D,-0xb4(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (6.3%) |
| MOV %R10,-0xd0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R8,-0xd8(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| VMOVDQA %XMM3,-0x90(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 4 | 1 | vect (25.0%) |
| VMOVQ %XMM5,-0x98(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| MOV -0x38(%RBP),%RAX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| SUB %FS:0x28,%RAX | 1 | 0.25 | 0.25 | 0.50 | 0.50 | 0 | 0.25 | 0.25 | 0 | 1 | 0.50 | N/A |
| JNE 2e0b <_Z7k_meansiP7point_tS0_PiS0_ii+0x28b> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50-1 | N/A |
| LEA -0x28(%RBP),%RSP | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| POP %RBX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R12 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R13 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R14 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R15 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %RBP | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| RET | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 1 | 0.33 | 0 | 1 | N/A |
| NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| CALL 2130 <__stack_chk_fail@plt> | 2 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 1 | 0.33 | 0 | 1 | N/A |
| Run run_1_thread | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: trueOMP_NUM_THREADS: 1 |
|---|---|
| Run run_2_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: trueOMP_NUM_THREADS: 2 |
| Run run_4_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: trueOMP_NUM_THREADS: 4 |
| Run run_8_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: trueOMP_NUM_THREADS: 8 |
| Run run_10_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: trueOMP_NUM_THREADS: 10 |
| (run_1_thread) Efficiency | (run_1_thread) Potential Speed-Up (%) | (run_2_threads) Efficiency | (run_2_threads) Potential Speed-Up (%) | (run_4_threads) Efficiency | (run_4_threads) Potential Speed-Up (%) | (run_8_threads) Efficiency | (run_8_threads) Potential Speed-Up (%) | (run_10_threads) Efficiency | (run_10_threads) Potential Speed-Up (%) |
|---|---|---|---|---|---|---|---|---|---|
| 1 | 0 | 0.95 | 0.22 | 0.89 | 0.47 | 0.78 | 0.93 | 0.73 | 1.12 |
| Run | Number of threads | Efficiency (ideal is 1) | Speedup | Ideal Speedup | Time (s) | Coverage (%) |
|---|---|---|---|---|---|---|
| run_1_thread | 1 | 1 | 1 | 1 | 7.1700005531311 | 4.6280465126038 |
| run_2_threads | 1 | 0.95 | 1.9 | 2 | 7.2049989700317 | 4.4856028556824 |
| run_4_threads | 1 | 0.89 | 3.56 | 4 | 7.1449999809265 | 4.2442607879639 |
| run_8_threads | 1 | 0.78 | 6.22 | 8 | 7.1749997138977 | 4.1801390647888 |
| run_10_threads | 1 | 0.73 | 7.32 | 10 | 7.1550006866455 | 4.1607303619385 |
| Name | Coverage (%) | Time (s) |
|---|---|---|
| ▼k_means(int, point_t*, point_t*, int*, point_t*, int, int)– | 4.63 | 7.17 |
| ▼Loop 5 - main.cpp:68-107 - kmeans-gcc-Ofast– | 0.00 | 0.00 |
| ○Loop 4 - main.cpp:93-96 - kmeans-gcc-Ofast | 4.63 | 7.17 |
| ○Loop 6 - main.cpp:98-104 - kmeans-gcc-Ofast | 0.00 | 0.00 |
