| Function: k_means(int, point_t*, point_t*, int*, point_t*, int, int) | Module: kmeans-icpx-O3-all | Source: main.cpp:67-108 [...] | Coverage (incl. loops): 7.22% | (excl. loops): 0.00% |
|---|
| Function: k_means(int, point_t*, point_t*, int*, point_t*, int, int) | Module: kmeans-icpx-O3-all | Source: main.cpp:67-108 [...] | Coverage (incl. loops): 7.22% | (excl. loops): 0.00% |
|---|
/home/fmusial/KMEANS_Benchmarks/kmeans/main.cpp: 67 - 108 |
-------------------------------------------------------------------------------- |
67: void k_means(int niters, point_t *points, point_t *centroids, int *assignment, point_t* memory, int n, int k) { |
68: for (int iter = 0; iter < niters; ++iter) { |
69: // determine nearest centroids |
70: #pragma omp parallel for |
71: for (int i = 0; i < n; ++i) { |
[...] |
85: int count[k]; |
86: double sum_x[k]; |
87: double sum_y[k]; |
88: for (int j = 0; j < k; ++j) { |
89: count[j] = 0; |
90: sum_x[j] = 0.; |
91: sum_y[j] = 0.; |
92: } |
93: for (int i = 0; i < n; ++i) { |
94: count[assignment[i]]++; |
95: sum_x[assignment[i]] += points[i].x; |
96: sum_y[assignment[i]] += points[i].y; |
97: } |
98: for (int j = 0; j < k; ++j) { |
99: if (count[j] != 0) { |
100: centroids[j].x = sum_x[j] / count[j]; |
101: centroids[j].y = sum_y[j] / count[j]; |
102: } |
103: // save centroids to memory |
104: memory[(iter + 1) * k + j].x = centroids[j].x; |
105: memory[(iter + 1) * k + j].y = centroids[j].y; |
106: } |
107: } |
108: } |
0x402c10 PUSH %RBP |
0x402c11 MOV %RSP,%RBP |
0x402c14 PUSH %R15 |
0x402c16 PUSH %R14 |
0x402c18 PUSH %R13 |
0x402c1a PUSH %R12 |
0x402c1c PUSH %RBX |
0x402c1d SUB $0xa8,%RSP |
0x402c24 MOV %R8,-0xa0(%RBP) |
0x402c2b MOV %RSI,-0x38(%RBP) |
0x402c2f TEST %EDI,%EDI |
0x402c31 JLE 403187 |
0x402c37 MOV %RCX,%R15 |
0x402c3a MOV %RDX,%R12 |
0x402c3d MOV 0x10(%RBP),%ECX |
0x402c40 LEA -0x1(%R9),%EAX |
0x402c44 MOV %RAX,-0x98(%RBP) |
0x402c4b MOV %ECX,%EDX |
0x402c4d LEA (,%RDX,4),%RAX |
0x402c55 MOV %RAX,-0x90(%RBP) |
0x402c5c LEA (,%RDX,8),%RAX |
0x402c64 MOV %RAX,-0x88(%RBP) |
0x402c6b MOV %EDI,%EAX |
0x402c6d MOV %R9D,%ESI |
0x402c70 DEC %RAX |
0x402c73 MOV %RAX,-0xc0(%RBP) |
0x402c7a LEA -0x1(%RDX),%RAX |
0x402c7e MOV %RAX,-0x68(%RBP) |
0x402c82 SAL $0x4,%RAX |
0x402c86 ADD %R12,%RAX |
0x402c89 ADD $0x8,%RAX |
0x402c8d MOV %RAX,-0x70(%RBP) |
0x402c91 MOV %RSI,-0x78(%RBP) |
0x402c95 AND $0x7ffffffe,%ESI |
0x402c9b MOV %RSI,-0xb8(%RBP) |
0x402ca2 MOV %EDX,%EAX |
0x402ca4 AND $0x7ffffff8,%EAX |
0x402ca9 MOV %RAX,-0x60(%RBP) |
0x402cad LEA 0xf(,%RDX,4),%RAX |
0x402cb5 AND $-0x10,%RAX |
0x402cb9 MOV %RAX,-0xb0(%RBP) |
0x402cc0 LEA 0xf(,%RDX,8),%RAX |
0x402cc8 VMOVSD 0x11340(%RIP),%XMM6 |
0x402cd0 VBROADCASTSD 0x11337(%RIP),%YMM7 |
0x402cd9 VMOVUPD 0x1133f(%RIP),%YMM8 |
0x402ce1 AND $-0x10,%RAX |
0x402ce5 MOV %RAX,-0xa8(%RBP) |
0x402cec VMOVUPD 0x1134c(%RIP),%YMM9 |
0x402cf4 MOV -0x38(%RBP),%RAX |
0x402cf8 ADD $0x18,%RAX |
0x402cfc MOV %RAX,-0x58(%RBP) |
0x402d00 MOV %RDX,-0x40(%RBP) |
0x402d04 LEA (%RDX,%RDX,1),%RAX |
0x402d08 MOV %RAX,-0xd0(%RBP) |
0x402d0f MOV %ECX,-0x2c(%RBP) |
0x402d12 XOR %EAX,%EAX |
0x402d14 MOV %RAX,-0x48(%RBP) |
0x402d18 MOV %R9,-0x50(%RBP) |
0x402d1c MOV %R15,-0x80(%RBP) |
0x402d20 JMP 402d59 |
0x402d22 NOPW %CS:(%RAX,%RAX,1) |
(5) 0x402d30 MOV -0xc8(%RBP),%RSP |
(5) 0x402d37 MOV -0x48(%RBP),%RDX |
(5) 0x402d3b LEA 0x1(%RDX),%RAX |
(5) 0x402d3f MOV -0x2c(%RBP),%ECX |
(5) 0x402d42 ADD 0x10(%RBP),%ECX |
(5) 0x402d45 MOV %ECX,-0x2c(%RBP) |
(5) 0x402d48 CMP -0xc0(%RBP),%RDX |
(5) 0x402d4f MOV %RAX,-0x48(%RBP) |
(5) 0x402d53 JE 403187 |
(5) 0x402d59 TEST %R9D,%R9D |
(5) 0x402d5c JLE 402db9 |
(5) 0x402d5e SUB $0x8,%RSP |
(5) 0x402d62 MOV $0x4181a0,%EDI |
(5) 0x402d67 MOV $0x4037a0,%EDX |
(5) 0x402d6c MOV $0x6,%ESI |
(5) 0x402d71 MOV -0x38(%RBP),%RCX |
(5) 0x402d75 MOV %R12,%R8 |
(5) 0x402d78 MOV %R15,%R9 |
(5) 0x402d7b XOR %EAX,%EAX |
(5) 0x402d7d PUSHQ -0x98(%RBP) |
(5) 0x402d83 PUSH $0 |
(5) 0x402d85 PUSHQ -0x40(%RBP) |
(5) 0x402d88 VZEROUPPER |
(5) 0x402d8b CALL 402290 <__kmpc_fork_call@plt> |
(5) 0x402d90 VMOVUPD 0x112a8(%RIP),%YMM9 |
(5) 0x402d98 VMOVUPD 0x11280(%RIP),%YMM8 |
(5) 0x402da0 VBROADCASTSD 0x11267(%RIP),%YMM7 |
(5) 0x402da9 VMOVSD 0x1125f(%RIP),%XMM6 |
(5) 0x402db1 MOV -0x50(%RBP),%R9 |
(5) 0x402db5 ADD $0x20,%RSP |
(5) 0x402db9 MOV %RSP,-0xc8(%RBP) |
(5) 0x402dc0 MOV %RSP,%R13 |
(5) 0x402dc3 SUB -0xb0(%RBP),%R13 |
(5) 0x402dca MOV %R13,%RSP |
(5) 0x402dcd MOV %RSP,%R14 |
(5) 0x402dd0 MOV -0xa8(%RBP),%RAX |
(5) 0x402dd7 SUB %RAX,%R14 |
(5) 0x402dda MOV %R14,%RSP |
(5) 0x402ddd MOV %RSP,%RBX |
(5) 0x402de0 SUB %RAX,%RBX |
(5) 0x402de3 MOV %RBX,%RSP |
(5) 0x402de6 CMPL $0,0x10(%RBP) |
(5) 0x402dea JLE 402e4a |
(5) 0x402dec MOV %R13,%RDI |
(5) 0x402def XOR %ESI,%ESI |
(5) 0x402df1 MOV -0x90(%RBP),%RDX |
(5) 0x402df8 VZEROUPPER |
(5) 0x402dfb CALL 403b80 <_intel_fast_memset> |
(5) 0x402e00 MOV %R14,%RDI |
(5) 0x402e03 XOR %ESI,%ESI |
(5) 0x402e05 MOV -0x88(%RBP),%R15 |
(5) 0x402e0c MOV %R15,%RDX |
(5) 0x402e0f CALL 403b80 <_intel_fast_memset> |
(5) 0x402e14 MOV %RBX,%RDI |
(5) 0x402e17 XOR %ESI,%ESI |
(5) 0x402e19 MOV %R15,%RDX |
(5) 0x402e1c MOV -0x80(%RBP),%R15 |
(5) 0x402e20 CALL 403b80 <_intel_fast_memset> |
(5) 0x402e25 VMOVUPD 0x11213(%RIP),%YMM9 |
(5) 0x402e2d VMOVUPD 0x111eb(%RIP),%YMM8 |
(5) 0x402e35 VBROADCASTSD 0x111d2(%RIP),%YMM7 |
(5) 0x402e3e VMOVSD 0x111ca(%RIP),%XMM6 |
(5) 0x402e46 MOV -0x50(%RBP),%R9 |
(5) 0x402e4a TEST %R9D,%R9D |
(5) 0x402e4d MOV -0xb8(%RBP),%RSI |
(5) 0x402e54 JLE 402f17 |
(5) 0x402e5a CMP $0x1,%R9D |
(5) 0x402e5e JNE 402e70 |
(5) 0x402e60 XOR %EAX,%EAX |
(5) 0x402e62 JMP 402edf |
0x402e64 NOPW %CS:(%RAX,%RAX,1) |
(5) 0x402e70 MOV -0x58(%RBP),%RCX |
(5) 0x402e74 XOR %EAX,%EAX |
(5) 0x402e76 NOPW %CS:(%RAX,%RAX,1) |
(9) 0x402e80 MOVSXD (%R15,%RAX,4),%RDX |
(9) 0x402e84 INCL (%R13,%RDX,4) |
(9) 0x402e89 VMOVSD (%R14,%RDX,8),%XMM0 |
(9) 0x402e8f VADDSD -0x18(%RCX),%XMM0,%XMM0 |
(9) 0x402e94 VMOVSD %XMM0,(%R14,%RDX,8) |
(9) 0x402e9a VMOVSD (%RBX,%RDX,8),%XMM0 |
(9) 0x402e9f VADDSD -0x10(%RCX),%XMM0,%XMM0 |
(9) 0x402ea4 VMOVSD %XMM0,(%RBX,%RDX,8) |
(9) 0x402ea9 MOVSXD 0x4(%R15,%RAX,4),%RDX |
(9) 0x402eae INCL (%R13,%RDX,4) |
(9) 0x402eb3 VMOVSD (%R14,%RDX,8),%XMM0 |
(9) 0x402eb9 VADDSD -0x8(%RCX),%XMM0,%XMM0 |
(9) 0x402ebe VMOVSD %XMM0,(%R14,%RDX,8) |
(9) 0x402ec4 VMOVSD (%RBX,%RDX,8),%XMM0 |
(9) 0x402ec9 VADDSD (%RCX),%XMM0,%XMM0 |
(9) 0x402ecd VMOVSD %XMM0,(%RBX,%RDX,8) |
(9) 0x402ed2 ADD $0x2,%RAX |
(9) 0x402ed6 ADD $0x20,%RCX |
(9) 0x402eda CMP %RAX,%RSI |
(9) 0x402edd JNE 402e80 |
(5) 0x402edf TESTB $0x1,-0x78(%RBP) |
(5) 0x402ee3 JE 402f17 |
(5) 0x402ee5 MOVSXD (%R15,%RAX,4),%RCX |
(5) 0x402ee9 INCL (%R13,%RCX,4) |
(5) 0x402eee VMOVSD (%R14,%RCX,8),%XMM0 |
(5) 0x402ef4 SAL $0x4,%RAX |
(5) 0x402ef8 MOV -0x38(%RBP),%RDX |
(5) 0x402efc VADDSD (%RDX,%RAX,1),%XMM0,%XMM0 |
(5) 0x402f01 VMOVSD %XMM0,(%R14,%RCX,8) |
(5) 0x402f07 VMOVSD (%RBX,%RCX,8),%XMM0 |
(5) 0x402f0c VADDSD 0x8(%RDX,%RAX,1),%XMM0,%XMM0 |
(5) 0x402f12 VMOVSD %XMM0,(%RBX,%RCX,8) |
(5) 0x402f17 CMPL $0,0x10(%RBP) |
(5) 0x402f1b JLE 402d30 |
(5) 0x402f21 MOV -0x2c(%RBP),%EAX |
(5) 0x402f24 SAL $0x4,%RAX |
(5) 0x402f28 MOV -0xa0(%RBP),%RSI |
(5) 0x402f2f ADD %RSI,%RAX |
(5) 0x402f32 MOV -0x48(%RBP),%RCX |
(5) 0x402f36 INC %ECX |
(5) 0x402f38 IMUL 0x10(%RBP),%ECX |
(5) 0x402f3c MOV %RCX,%RDX |
(5) 0x402f3f SAL $0x4,%RDX |
(5) 0x402f43 ADD %RSI,%RDX |
(5) 0x402f46 CMP %RDX,-0x70(%RBP) |
(5) 0x402f4a JB 402fd0 |
(5) 0x402f50 ADD -0x68(%RBP),%RCX |
(5) 0x402f54 SAL $0x4,%RCX |
(5) 0x402f58 ADD %RSI,%RCX |
(5) 0x402f5b ADD $0x8,%RCX |
(5) 0x402f5f CMP %R12,%RCX |
(5) 0x402f62 JB 402fd0 |
(5) 0x402f64 XOR %ECX,%ECX |
(5) 0x402f66 JMP 402fa7 |
0x402f68 NOPL (%RAX,%RAX,1) |
(8) 0x402f70 VCVTSI2SD %EDX,%XMM10,%XMM0 |
(8) 0x402f74 VMOVSD (%R14,%RCX,4),%XMM1 |
(8) 0x402f7a VMOVHPD (%RBX,%RCX,4),%XMM1,%XMM1 |
(8) 0x402f7f VDIVSD %XMM0,%XMM6,%XMM0 |
(8) 0x402f83 VMOVDDUP %XMM0,%XMM0 |
(8) 0x402f87 VMULPD %XMM0,%XMM1,%XMM0 |
(8) 0x402f8b VMOVUPD %XMM0,(%R12,%RCX,8) |
(8) 0x402f91 VMOVUPD %XMM0,(%RAX,%RCX,8) |
(8) 0x402f96 ADD $0x2,%RCX |
(8) 0x402f9a CMP %RCX,-0xd0(%RBP) |
(8) 0x402fa1 JE 402d30 |
(8) 0x402fa7 MOV (%R13,%RCX,2),%EDX |
(8) 0x402fac TEST %EDX,%EDX |
(8) 0x402fae JNE 402f70 |
(8) 0x402fb0 VMOVUPD (%R12,%RCX,8),%XMM0 |
(8) 0x402fb6 VMOVUPD %XMM0,(%RAX,%RCX,8) |
(8) 0x402fbb ADD $0x2,%RCX |
(8) 0x402fbf CMP %RCX,-0xd0(%RBP) |
(8) 0x402fc6 JNE 402fa7 |
(5) 0x402fc8 JMP 402d30 |
0x402fcd NOPL (%RAX) |
(5) 0x402fd0 MOV -0x60(%RBP),%RSI |
(5) 0x402fd4 TEST %RSI,%RSI |
(5) 0x402fd7 JE 403109 |
(5) 0x402fdd XOR %ECX,%ECX |
(5) 0x402fdf XOR %EDX,%EDX |
(5) 0x402fe1 NOPW %CS:(%RAX,%RAX,1) |
(7) 0x402ff0 VMOVDQU (%R13,%RDX,4),%XMM0 |
(7) 0x402ff7 VMOVDQU 0x10(%R13,%RDX,4),%XMM1 |
(7) 0x402ffe VMOVDQU (%R13,%RDX,4),%YMM2 |
(7) 0x403005 VCVTDQ2PD %XMM1,%YMM3 |
(7) 0x403009 VCVTDQ2PD %XMM0,%YMM4 |
(7) 0x40300d VPTESTMD %XMM0,%XMM0,%K1 |
(7) 0x403013 VPTESTMD %XMM1,%XMM1,%K2 |
(7) 0x403019 VMOVUPD 0x20(%R14,%RDX,8),%YMM0{%K2}{z} |
(7) 0x403021 VMOVUPD (%R14,%RDX,8),%YMM1{%K1}{z} |
(7) 0x403028 VPTESTMD %YMM2,%YMM2,%K0 |
(7) 0x40302e VDIVPD %YMM4,%YMM7,%YMM2 |
(7) 0x403032 VMULPD %YMM2,%YMM1,%YMM1 |
(7) 0x403036 VDIVPD %YMM3,%YMM7,%YMM3 |
(7) 0x40303a VMOVUPD 0x20(%RBX,%RDX,8),%YMM4{%K2}{z} |
(7) 0x403042 VMULPD %YMM3,%YMM0,%YMM0 |
(7) 0x403046 VMOVUPD (%RBX,%RDX,8),%YMM5{%K1}{z} |
(7) 0x40304d VMULPD %YMM2,%YMM5,%YMM2 |
(7) 0x403051 VMULPD %YMM3,%YMM4,%YMM3 |
(7) 0x403055 VMOVAPD %YMM0,%YMM4 |
(7) 0x403059 VMOVAPD %YMM1,%YMM5 |
(7) 0x40305d VPERMT2PD %YMM2,%YMM8,%YMM5 |
(7) 0x403063 VPERMT2PD %YMM3,%YMM9,%YMM0 |
(7) 0x403069 VPERMT2PD %YMM2,%YMM9,%YMM1 |
(7) 0x40306f VPMOVM2W %K0,%YMM2 |
(7) 0x403075 VPMOVSXWD %XMM2,%YMM2 |
(7) 0x40307a VPMOVW2M %YMM2,%K1 |
(7) 0x403080 VMOVUPD %YMM1,(%R12,%RCX,1){%K1} |
(7) 0x403087 KSHIFTRW $0x8,%K1,%K2 |
(7) 0x40308d VMOVUPD %YMM0,0x40(%R12,%RCX,1){%K2} |
(7) 0x403095 KSHIFTRB $0x4,%K1,%K1 |
(7) 0x40309b VMOVUPD %YMM5,0x20(%R12,%RCX,1){%K1} |
(7) 0x4030a3 VPERMT2PD %YMM3,%YMM8,%YMM4 |
(7) 0x4030a9 KSHIFTRB $0x4,%K2,%K1 |
(7) 0x4030af VMOVUPD %YMM4,0x60(%R12,%RCX,1){%K1} |
(7) 0x4030b7 VMOVUPD (%R12,%RCX,1),%YMM0 |
(7) 0x4030bd VMOVUPD 0x20(%R12,%RCX,1),%YMM1 |
(7) 0x4030c4 VMOVUPS 0x40(%R12,%RCX,1),%YMM2 |
(7) 0x4030cb VMOVUPS 0x60(%R12,%RCX,1),%YMM3 |
(7) 0x4030d2 VMOVUPS %YMM2,0x40(%RAX,%RCX,1) |
(7) 0x4030d8 VMOVUPS %YMM3,0x60(%RAX,%RCX,1) |
(7) 0x4030de VMOVUPD %YMM0,(%RAX,%RCX,1) |
(7) 0x4030e3 VMOVUPD %YMM1,0x20(%RAX,%RCX,1) |
(7) 0x4030e9 ADD $0x8,%RDX |
(7) 0x4030ed SUB $-0x80,%RCX |
(7) 0x4030f1 CMP %RSI,%RDX |
(7) 0x4030f4 JB 402ff0 |
(5) 0x4030fa MOV %RSI,%R8 |
(5) 0x4030fd CMP -0x40(%RBP),%RSI |
(5) 0x403101 JE 402d30 |
(5) 0x403107 JMP 40310c |
(5) 0x403109 XOR %R8D,%R8D |
(5) 0x40310c MOV %R8,%RCX |
(5) 0x40310f SAL $0x4,%RCX |
(5) 0x403113 ADD %RCX,%RAX |
(5) 0x403116 ADD %R12,%RCX |
(5) 0x403119 MOV -0x40(%RBP),%RDX |
(5) 0x40311d SUB %R8,%RDX |
(5) 0x403120 LEA (%RBX,%R8,8),%RSI |
(5) 0x403124 LEA (%R14,%R8,8),%RDI |
(5) 0x403128 LEA (,%R8,4),%R8 |
(5) 0x403130 ADD %R13,%R8 |
(5) 0x403133 XOR %R11D,%R11D |
(5) 0x403136 JMP 403176 |
0x403138 NOPL (%RAX,%RAX,1) |
(6) 0x403140 VCVTSI2SD %R10D,%XMM10,%XMM0 |
(6) 0x403145 VMOVSD (%RDI,%R11,2),%XMM1 |
(6) 0x40314b VMOVHPD (%RSI,%R11,2),%XMM1,%XMM1 |
(6) 0x403151 VDIVSD %XMM0,%XMM6,%XMM0 |
(6) 0x403155 VMOVDDUP %XMM0,%XMM0 |
(6) 0x403159 VMULPD %XMM0,%XMM1,%XMM0 |
(6) 0x40315d VMOVUPD %XMM0,(%RCX,%R11,4) |
(6) 0x403163 VMOVUPD %XMM0,(%RAX,%R11,4) |
(6) 0x403169 ADD $0x4,%R11 |
(6) 0x40316d DEC %RDX |
(6) 0x403170 JE 402d30 |
(6) 0x403176 MOV (%R8,%R11,1),%R10D |
(6) 0x40317a TEST %R10D,%R10D |
(6) 0x40317d JNE 403140 |
(6) 0x40317f VMOVUPD (%RCX,%R11,4),%XMM0 |
(6) 0x403185 JMP 403163 |
0x403187 LEA -0x28(%RBP),%RSP |
0x40318b POP %RBX |
0x40318c POP %R12 |
0x40318e POP %R13 |
0x403190 POP %R14 |
0x403192 POP %R15 |
0x403194 POP %RBP |
0x403195 VZEROUPPER |
0x403198 RET |
0x403199 NOPL (%RAX) |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:27 | kmeans-icpx-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-icpx-O3-all |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:27 | kmeans-icpx-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-icpx-O3-all |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:27 | kmeans-icpx-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-icpx-O3-all |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:27 | kmeans-icpx-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-icpx-O3-all |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | main | main.cpp:27 | kmeans-icpx-O3-all |
| ○ | __libc_init_first | libc.so.6 | |
| ○ | __libc_start_main | libc.so.6 | |
| ○ | _start | kmeans-icpx-O3-all |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| Path / |
The code analyzed by CQA in that panel excludes loops and represents 0.00% of application time for run run_1_thread
| Source file and lines | main.cpp:67-108 |
| Module | kmeans-icpx-O3-all |
| nb instructions | 75 |
| nb uops | 78 |
| loop length | 344 |
| used x86 registers | 14 |
| used mmx registers | 0 |
| used xmm registers | 1 |
| used ymm registers | 3 |
| used zmm registers | 0 |
| nb stack references | 22 |
| micro-operation queue | 19.50 cycles |
| front end | 19.50 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | |
|---|---|---|---|---|---|---|---|---|
| uops | 5.50 | 5.50 | 13.00 | 13.00 | 26.00 | 5.50 | 5.50 | 13.00 |
| cycles | 5.50 | 6.50 | 13.00 | 13.00 | 26.00 | 5.50 | 5.50 | 13.00 |
| Cycles executing div or sqrt instructions | NA |
| FE+BE cycles | 26.11 |
| Stall cycles | 6.97 |
| SB full (events) | 12.50 |
| Front-end | 19.50 |
| Dispatch | 26.00 |
| Overall L1 | 26.00 |
| all | 3% |
| load | 0% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 16% |
| all | 50% |
| load | 50% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 0% |
| all | 9% |
| load | 40% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 14% |
| all | 11% |
| load | 6% |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 11% |
| all | 31% |
| load | 31% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 12% |
| all | 14% |
| load | 26% |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 11% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| PUSH %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| PUSH %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| SUB $0xa8,%RSP | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (12.5%) |
| MOV %R8,-0xa0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RSI,-0x38(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| TEST %EDI,%EDI | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (6.3%) |
| JLE 403187 <_Z7k_meansiP7point_tS0_PiS0_ii+0x577> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50-1 | N/A |
| MOV %RCX,%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (12.5%) |
| MOV %RDX,%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (12.5%) |
| MOV 0x10(%RBP),%ECX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | scal (6.3%) |
| LEA -0x1(%R9),%EAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RAX,-0x98(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %ECX,%EDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| LEA (,%RDX,4),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RAX,-0x90(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA (,%RDX,8),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RAX,-0x88(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %EDI,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| MOV %R9D,%ESI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (6.3%) |
| DEC %RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0xc0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA -0x1(%RDX),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RAX,-0x68(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| SAL $0x4,%RAX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | N/A |
| ADD %R12,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| ADD $0x8,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0x70(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RSI,-0x78(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| AND $0x7ffffffe,%ESI | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (6.3%) |
| MOV %RSI,-0xb8(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %EDX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| AND $0x7ffffff8,%EAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0x60(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA 0xf(,%RDX,4),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| AND $-0x10,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0xb0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA 0xf(,%RDX,8),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| VMOVSD 0x11340(%RIP),%XMM6 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | scal (12.5%) |
| VBROADCASTSD 0x11337(%RIP),%YMM7 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 5 | 0.50 | scal (12.5%) |
| VMOVUPD 0x1133f(%RIP),%YMM8 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 5-6 | 0.50 | vect (50.0%) |
| AND $-0x10,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0xa8(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| VMOVUPD 0x1134c(%RIP),%YMM9 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 5-6 | 0.50 | vect (50.0%) |
| MOV -0x38(%RBP),%RAX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| ADD $0x18,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0x58(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RDX,-0x40(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA (%RDX,%RDX,1),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RAX,-0xd0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %ECX,-0x2c(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (6.3%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| MOV %RAX,-0x48(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R9,-0x50(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R15,-0x80(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| JMP 402d59 <_Z7k_meansiP7point_tS0_PiS0_ii+0x149> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 1-2 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| LEA -0x28(%RBP),%RSP | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| POP %RBX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R12 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R13 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R14 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R15 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %RBP | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| VZEROUPPER | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | vect (25.0%) |
| RET | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 1 | 0.33 | 0 | 1 | N/A |
| NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
The code analyzed by CQA in that panel excludes loops and represents 0.00% of application time for run run_1_thread
| Source file and lines | main.cpp:67-108 |
| Module | kmeans-icpx-O3-all |
| nb instructions | 75 |
| nb uops | 78 |
| loop length | 344 |
| used x86 registers | 14 |
| used mmx registers | 0 |
| used xmm registers | 1 |
| used ymm registers | 3 |
| used zmm registers | 0 |
| nb stack references | 22 |
| micro-operation queue | 19.50 cycles |
| front end | 19.50 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | |
|---|---|---|---|---|---|---|---|---|
| uops | 5.50 | 5.50 | 13.00 | 13.00 | 26.00 | 5.50 | 5.50 | 13.00 |
| cycles | 5.50 | 6.50 | 13.00 | 13.00 | 26.00 | 5.50 | 5.50 | 13.00 |
| Cycles executing div or sqrt instructions | NA |
| FE+BE cycles | 26.11 |
| Stall cycles | 6.97 |
| SB full (events) | 12.50 |
| Front-end | 19.50 |
| Dispatch | 26.00 |
| Overall L1 | 26.00 |
| all | 3% |
| load | 0% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 16% |
| all | 50% |
| load | 50% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 0% |
| all | 9% |
| load | 40% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 14% |
| all | 11% |
| load | 6% |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 11% |
| all | 31% |
| load | 31% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 12% |
| all | 14% |
| load | 26% |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 11% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
| PUSH %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| PUSH %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| PUSH %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | N/A |
| SUB $0xa8,%RSP | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (12.5%) |
| MOV %R8,-0xa0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RSI,-0x38(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| TEST %EDI,%EDI | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (6.3%) |
| JLE 403187 <_Z7k_meansiP7point_tS0_PiS0_ii+0x577> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50-1 | N/A |
| MOV %RCX,%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (12.5%) |
| MOV %RDX,%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (12.5%) |
| MOV 0x10(%RBP),%ECX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | scal (6.3%) |
| LEA -0x1(%R9),%EAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RAX,-0x98(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %ECX,%EDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| LEA (,%RDX,4),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RAX,-0x90(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA (,%RDX,8),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RAX,-0x88(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %EDI,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| MOV %R9D,%ESI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | scal (6.3%) |
| DEC %RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0xc0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA -0x1(%RDX),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RAX,-0x68(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| SAL $0x4,%RAX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 1 | 0.50 | N/A |
| ADD %R12,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| ADD $0x8,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0x70(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RSI,-0x78(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| AND $0x7ffffffe,%ESI | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | scal (6.3%) |
| MOV %RSI,-0xb8(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %EDX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| AND $0x7ffffff8,%EAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0x60(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA 0xf(,%RDX,4),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| AND $-0x10,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0xb0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA 0xf(,%RDX,8),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| VMOVSD 0x11340(%RIP),%XMM6 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | scal (12.5%) |
| VBROADCASTSD 0x11337(%RIP),%YMM7 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 5 | 0.50 | scal (12.5%) |
| VMOVUPD 0x1133f(%RIP),%YMM8 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 5-6 | 0.50 | vect (50.0%) |
| AND $-0x10,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0xa8(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| VMOVUPD 0x1134c(%RIP),%YMM9 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 5-6 | 0.50 | vect (50.0%) |
| MOV -0x38(%RBP),%RAX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4-5 | 0.50 | N/A |
| ADD $0x18,%RAX | 1 | 0.25 | 0.25 | 0 | 0 | 0 | 0.25 | 0.25 | 0 | 1 | 0.25 | N/A |
| MOV %RAX,-0x58(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %RDX,-0x40(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| LEA (%RDX,%RDX,1),%RAX | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RAX,-0xd0(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %ECX,-0x2c(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (6.3%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| MOV %RAX,-0x48(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R9,-0x50(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| MOV %R15,-0x80(%RBP) | 1 | 0 | 0 | 0.33 | 0.33 | 1 | 0 | 0 | 0.33 | 3 | 1 | scal (12.5%) |
| JMP 402d59 <_Z7k_meansiP7point_tS0_PiS0_ii+0x149> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 1-2 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| LEA -0x28(%RBP),%RSP | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 1 | 0.50 | N/A |
| POP %RBX | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R12 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R13 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R14 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %R15 | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| POP %RBP | 1 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 2 | 0.50 | N/A |
| VZEROUPPER | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | vect (25.0%) |
| RET | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 1 | 0.33 | 0 | 1 | N/A |
| NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
| Run run_1_thread | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: trueOMP_NUM_THREADS: 1 |
|---|---|
| Run run_2_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: trueOMP_NUM_THREADS: 2 |
| Run run_4_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: trueOMP_NUM_THREADS: 4 |
| Run run_8_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: trueOMP_NUM_THREADS: 8 |
| Run run_10_threads | Number processes: 1Number nodes: 1Run Command: <executable> input/100000000.in 1000 100000000 50 25MPI Command: Dataset: Run Directory: /home/fmusial/KMEANS_BenchmarksOMP_PROC_BIND: trueOMP_NUM_THREADS: 10 |
| (run_1_thread) Efficiency | (run_1_thread) Potential Speed-Up (%) | (run_2_threads) Efficiency | (run_2_threads) Potential Speed-Up (%) | (run_4_threads) Efficiency | (run_4_threads) Potential Speed-Up (%) | (run_8_threads) Efficiency | (run_8_threads) Potential Speed-Up (%) | (run_10_threads) Efficiency | (run_10_threads) Potential Speed-Up (%) |
|---|---|---|---|---|---|---|---|---|---|
| 1 | 0 | 0.96 | 0.28 | 0.89 | 0.67 | 0.82 | 0.97 | 0.76 | 1.2 |
| Run | Number of threads | Efficiency (ideal is 1) | Speedup | Ideal Speedup | Time (s) | Coverage (%) |
|---|---|---|---|---|---|---|
| run_1_thread | 1 | 1 | 1 | 1 | 7.3499999046326 | 7.2172055244446 |
| run_2_threads | 1 | 0.96 | 1.92 | 2 | 7.4749989509583 | 6.8552808761597 |
| run_4_threads | 1 | 0.89 | 3.57 | 4 | 7.6949996948242 | 6.2089004516602 |
| run_8_threads | 1 | 0.82 | 6.53 | 8 | 7.7649993896484 | 5.2895102500916 |
| run_10_threads | 1 | 0.76 | 7.6 | 10 | 7.9350004196167 | 5.0064668655396 |
| Name | Coverage (%) | Time (s) |
|---|---|---|
| ▼k_means(int, point_t*, point_t*, int*, point_t*, int, int)– | 7.22 | 7.35 |
| ▼Loop 5 - main.cpp:68-105 - kmeans-icpx-O3-all– | 0.00 | 0.00 |
| ○Loop 9 - main.cpp:93-96 - kmeans-icpx-O3-all | 7.22 | 7.35 |
| ○Loop 6 - main.cpp:98-104 - kmeans-icpx-O3-all | 0.00 | 0.00 |
| ○Loop 8 - main.cpp:98-104 - kmeans-icpx-O3-all | 0.00 | 0.00 |
| ○Loop 7 - main.cpp:98-105 - kmeans-icpx-O3-all | 0.00 | 0.00 |
