| Loop Id: 66 | Module: exec | Source: advec_cell_kernel.f90:107-155 [...] | Coverage: 0.05% |
|---|
| Loop Id: 66 | Module: exec | Source: advec_cell_kernel.f90:107-155 [...] | Coverage: 0.05% |
|---|
(65) 0x4199ec CMP W9, W8 |
(65) 0x4199f0 ADD W9, W9, #1 |
(65) 0x4199f4 ADD W2, W2, #1 |
(65) 0x4199f8 B.EQ 419bbc |
(65) 0x4199fc TBNZ X17, #63, 4199ec |
(67) 0x419a00 LDP X18, X12, [SP, #104] |
(67) 0x419a04 LDR X7, [SP, #128] |
(67) 0x419a08 ADD X27, X17, #2 |
(67) 0x419a0c ORR X3, XZR, XZR |
(67) 0x419a10 ORR X28, XZR, X13 |
(67) 0x419a14 ADD X16, X12, W2,SXTW |
(67) 0x419a18 SUB X12, X13, X24 |
(67) 0x419a1c MADD X16, X7, X16, X12 |
(67) 0x419a20 LDUR W12, [X29, #368] |
(67) 0x419a24 ADD X4, X18, X16,LSL #3 |
(67) 0x419a28 LDR X18, [SP, #120] |
(67) 0x419a2c ADD X5, X25, X16,LSL #3 |
(67) 0x419a30 ADD X6, X19, X16,LSL #3 |
(67) 0x419a34 ADD W16, W12, W9 |
(67) 0x419a38 SBFM X16, X16, #0, #31 |
(67) 0x419a3c SUB X16, X16, X18 |
(67) 0x419a40 MUL X18, X16, X7 |
(67) 0x419a44 ADD X7, X21, X18,LSL #3 |
(67) 0x419a48 LDR X18, [SP, #136] |
(67) 0x419a4c MUL X26, X16, X18 |
(67) 0x419a50 B 419a7c |
(67) 0x419a60 FMUL D3, D16, D3 |
(67) 0x419a64 SUB X27, X27, #1 |
(67) 0x419a68 ADD X28, X28, #1 |
(67) 0x419a6c CMP X27, #1 |
(67) 0x419a70 STR D3, [X4, X3,LSL #3] |
(67) 0x419a74 ADD X3, X3, #1 |
(67) 0x419a78 B.LE 4199ec |
(67) 0x419a7c LDR D3, [X6, X3,LSL #3] |
(67) 0x419a80 FCMP D3, #0 |
(67) 0x419a84 B.LE 419aa0 |
(67) 0x419a88 ADD W30, W1, W3 |
(67) 0x419a8c SUB W18, W28, #1 |
(67) 0x419a90 ADD W23, W0, W3 |
(67) 0x419a94 ORR X16, XZR, X28 |
(67) 0x419a98 B 419abc |
0x419aa0 ADD W16, W10, W3 |
0x419aa4 ADD X18, X10, X3 |
0x419aa8 ADD W23, W16, #1 |
0x419aac CMP W23, W11 |
0x419ab0 CSINC W30, W11, W16, #10 |
0x419ab4 ADD X16, X15, X3 |
0x419ab8 ORR W23, WZR, W30 |
(67) 0x419abc SBFM X18, X18, #0, #31 |
(67) 0x419ac0 FABS D5, D3 |
(67) 0x419ac4 LDR X12, [SP, #152] |
(67) 0x419ac8 SBFM X23, X23, #0, #31 |
(67) 0x419acc SUB X18, X18, X24 |
(67) 0x419ad0 SUB X23, X23, X24 |
(67) 0x419ad4 LDR D4, [X7, X18,LSL #3] |
(67) 0x419ad8 LDR D6, [X12, X23,LSL #3] |
(67) 0x419adc ADD X23, X18, X26 |
(67) 0x419ae0 SUB X18, X26, X24 |
(67) 0x419ae4 ADD X30, X18, W30,SXTW |
(67) 0x419ae8 ADD X16, X18, W16,SXTW |
(67) 0x419aec LDR D7, [X20, X23,LSL #3] |
(67) 0x419af0 LDR D17, [X20, X30,LSL #3] |
(67) 0x419af4 LDR D18, [X20, X16,LSL #3] |
(67) 0x419af8 FDIV D16, D5, D4 |
(67) 0x419afc LDR D5, [X14, X3,LSL #3] |
(67) 0x419b00 FSUB D17, D7, S17 |
(67) 0x419b04 FSUB D18, D18, S7 |
(67) 0x419b08 FMUL D19, D18, D17 |
(67) 0x419b0c FCMP D19, #0 |
(67) 0x419b10 FMOV D19, D7 |
(67) 0x419b14 FMADD D5, D16, D5, D5 |
(67) 0x419b18 FDIV D5, D5, D6 |
(67) 0x419b1c FSUB D6, D0, S16 |
(67) 0x419b20 B.LE 419b54 |
(67) 0x419b24 FSUB D16, D1, S16 |
(67) 0x419b28 FCMP D18, #0 |
(67) 0x419b2c FABS D17, D17 |
(67) 0x419b30 FABS D18, D18 |
(67) 0x419b34 FNEG D19, D16 |
(67) 0x419b38 FCSEL D16, D16, D19, #12 |
(67) 0x419b3c FMUL D19, D17, D5 |
(67) 0x419b40 FMINNM D17, D17, D18 |
(67) 0x419b44 FMADD D19, D18, D6, D19 |
(67) 0x419b48 FMUL D19, D19, D2 |
(67) 0x419b4c FMINNM D17, D17, D19 |
(67) 0x419b50 FMADD D19, D17, D16, D7 |
(67) 0x419b54 LDR D16, [X22, X23,LSL #3] |
(67) 0x419b58 LDR D17, [X22, X30,LSL #3] |
(67) 0x419b5c LDR D18, [X22, X16,LSL #3] |
(67) 0x419b60 FMUL D3, D19, D3 |
(67) 0x419b64 FSUB D17, D16, S17 |
(67) 0x419b68 FSUB D18, D18, S16 |
(67) 0x419b6c STR D3, [X5, X3,LSL #3] |
(67) 0x419b70 FMUL D19, D18, D17 |
(67) 0x419b74 FCMP D19, #0 |
(67) 0x419b78 B.LE 419a60 |
(67) 0x419b7c FMUL D4, D7, D4 |
(67) 0x419b80 FABS D19, D3 |
(67) 0x419b84 FCMP D18, #0 |
(67) 0x419b88 FDIV D4, D19, D4 |
(67) 0x419b8c FSUB D4, D1, S4 |
(67) 0x419b90 FNEG D7, D4 |
(67) 0x419b94 FCSEL D4, D4, D7, #12 |
(67) 0x419b98 FABS D7, D17 |
(67) 0x419b9c FABS D17, D18 |
(67) 0x419ba0 FMUL D5, D7, D5 |
(67) 0x419ba4 FMADD D5, D17, D6, D5 |
(67) 0x419ba8 FMINNM D6, D7, D17 |
(67) 0x419bac FMUL D5, D5, D2 |
(67) 0x419bb0 FMINNM D5, D6, D5 |
(67) 0x419bb4 FMADD D16, D5, D4, D16 |
(67) 0x419bb8 B 419a60 |
/home/eoseret/qaas/qaas_runs/178-231-1255/intel/CloverLeaf1.3-FC/build/CloverLeaf1.3-FC/CloverLeaf_ref/kernels/advec_cell_kernel.f90: 107 - 155 |
-------------------------------------------------------------------------------- |
107: !$OMP DO PRIVATE(upwind,donor,downwind,dif,sigmat,sigma3,sigma4,sigmav,sigma,sigmam, & |
108: !$OMP diffuw,diffdw,limiter,wind) |
109: DO k=y_min,y_max |
110: DO j=x_min,x_max+2 |
111: |
112: IF(vol_flux_x(j,k).GT.0.0)THEN |
113: upwind =j-2 |
114: donor =j-1 |
115: downwind =j |
116: dif =donor |
117: ELSE |
118: upwind =MIN(j+1,x_max+2) |
[...] |
124: sigmat=ABS(vol_flux_x(j,k))/pre_vol(donor,k) |
125: sigma3=(1.0_8+sigmat)*(vertexdx(j)/vertexdx(dif)) |
126: sigma4=2.0_8-sigmat |
127: |
128: sigma=sigmat |
129: sigmav=sigmat |
130: |
131: diffuw=density1(donor,k)-density1(upwind,k) |
132: diffdw=density1(downwind,k)-density1(donor,k) |
133: wind=1.0_8 |
134: IF(diffdw.LE.0.0) wind=-1.0_8 |
135: IF(diffuw*diffdw.GT.0.0)THEN |
136: limiter=(1.0_8-sigmav)*wind*MIN(ABS(diffuw),ABS(diffdw)& |
137: ,one_by_six*(sigma3*ABS(diffuw)+sigma4*ABS(diffdw))) |
138: ELSE |
139: limiter=0.0 |
140: ENDIF |
141: mass_flux_x(j,k)=vol_flux_x(j,k)*(density1(donor,k)+limiter) |
142: |
143: sigmam=ABS(mass_flux_x(j,k))/(density1(donor,k)*pre_vol(donor,k)) |
144: diffuw=energy1(donor,k)-energy1(upwind,k) |
145: diffdw=energy1(downwind,k)-energy1(donor,k) |
146: wind=1.0_8 |
147: IF(diffdw.LE.0.0) wind=-1.0_8 |
148: IF(diffuw*diffdw.GT.0.0)THEN |
149: limiter=(1.0_8-sigmam)*wind*MIN(ABS(diffuw),ABS(diffdw)& |
[...] |
155: ener_flux(j,k)=mass_flux_x(j,k)*(energy1(donor,k)+limiter) |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►98.61+ | __kmp_invoke_microtask | libomp.so | |
| ○ | __kmp_invoke_task_func | libomp.so | |
| ○ | __kmp_launch_thread | libomp.so | |
| ○ | __kmp_launch_worker(void*) | libomp.so | |
| ○ | start_thread | libc.so.6 | |
| ○ | thread_start | libc.so.6 |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| Path / |
| Metric | Value |
|---|---|
| CQA speedup if no scalar integer | NA |
| CQA speedup if FP arith vectorized | NA |
| CQA speedup if fully vectorized | NA |
| CQA speedup if no inter-iteration dependency | NA |
| CQA speedup if next bottleneck killed | NA |
| Bottlenecks | NA |
| Function | _QMadvec_cell_kernel_modulePadvec_cell_kernel..omp_par |
| Source | advec_cell_kernel.f90:112-112,advec_cell_kernel.f90:118-118 |
| Source loop unroll info | NA |
| Source loop unroll confidence level | NA |
| Unroll/vectorization loop type | NA |
| Unroll factor | NA |
| CQA cycles | NA |
| CQA cycles if no scalar integer | NA |
| CQA cycles if FP arith vectorized | NA |
| CQA cycles if fully vectorized | NA |
| Front-end cycles | NA |
| P0 cycles | NA |
| P1 cycles | NA |
| P2 cycles | NA |
| P3 cycles | NA |
| P4 cycles | NA |
| P5 cycles | NA |
| P6 cycles | NA |
| P7 cycles | NA |
| P8 cycles | NA |
| P9 cycles | NA |
| P10 cycles | NA |
| P11 cycles | NA |
| P12 cycles | NA |
| P13 cycles | NA |
| P14 cycles | NA |
| DIV/SQRT cycles | NA |
| Inter-iter dependencies cycles | NA |
| FE+BE cycles (UFS) | NA |
| Stall cycles (UFS) | NA |
| Nb insns | NA |
| Nb uops | NA |
| Nb loads | NA |
| Nb stores | NA |
| Nb stack references | NA |
| FLOP/cycle | NA |
| Nb FLOP add-sub | NA |
| Nb FLOP mul | NA |
| Nb FLOP fma | NA |
| Nb FLOP div | NA |
| Nb FLOP rcp | NA |
| Nb FLOP sqrt | NA |
| Nb FLOP rsqrt | NA |
| Bytes/cycle | NA |
| Bytes prefetched | NA |
| Bytes loaded | NA |
| Bytes stored | NA |
| Stride 0 | NA |
| Stride 1 | NA |
| Stride n | NA |
| Stride unknown | NA |
| Stride indirect | NA |
| Vectorization ratio all | NA |
| Vectorization ratio load | NA |
| Vectorization ratio store | NA |
| Vectorization ratio mul | NA |
| Vectorization ratio add_sub | NA |
| Vectorization ratio fma | NA |
| Vectorization ratio div_sqrt | NA |
| Vectorization ratio other | NA |
| Vector-efficiency ratio all | NA |
| Vector-efficiency ratio load | NA |
| Vector-efficiency ratio store | NA |
| Vector-efficiency ratio mul | NA |
| Vector-efficiency ratio add_sub | NA |
| Vector-efficiency ratio fma | NA |
| Vector-efficiency ratio div_sqrt | NA |
| Vector-efficiency ratio other | NA |
| Path / |
| Function | _QMadvec_cell_kernel_modulePadvec_cell_kernel..omp_par |
| Source file and lines | advec_cell_kernel.f90:112-112,advec_cell_kernel.f90:118-118 |
| Module | exec |
