| Loop Id: 360 | Module: exec | Source: BsplineFunctor.h:233-260 [...] | Coverage: 0.01% |
|---|
| Loop Id: 360 | Module: exec | Source: BsplineFunctor.h:233-260 [...] | Coverage: 0.01% |
|---|
(358) 0x419c20 MOVI D0, #0 |
(358) 0x419c24 FSUB D0, D8, S0 |
(358) 0x419c28 BL 404990 |
(358) 0x419c2c LDP X8, X9, [X19] |
(358) 0x419c30 STR D0, [X8, X22,LSL #3] |
(358) 0x419c34 ADD X22, X22, #1 |
(358) 0x419c38 SUB X8, X9, X8 |
(358) 0x419c3c CMP X22, X8,ASR #3 |
(358) 0x419c40 B.CS 419e64 |
(358) 0x419c44 LDRB W8, [X20, #664] |
(358) 0x419c48 TBZ W8, #0, 419e80 |
(358) 0x419c4c LDRSW X24, [X20, #672] |
(358) 0x419c50 LDR X8, [X21, #216] |
(358) 0x419c54 LDR W1, [X21, #584] |
(358) 0x419c58 ORR X0, XZR, X20 |
(358) 0x419c5c LDR X25, [X20, #656] |
(358) 0x419c60 LDR D8, [X8, X24,LSL #3] |
(358) 0x419c64 BL 4476b0 |
(358) 0x419c68 LDR X8, [X21, #160] |
(358) 0x419c6c CBZ X8, 419c20 |
(359) 0x419c70 LDR X11, [X25, #24] |
(359) 0x419c74 LDR X12, [X21, #512] |
(359) 0x419c78 LDR X10, [X0, #72] |
(359) 0x419c7c MOVI D0, #0 |
(359) 0x419c80 ORR X9, XZR, XZR |
(359) 0x419c84 MADD X10, X22, X23, X10 |
(359) 0x419c88 LDR W11, [X11, X24,LSL #2] |
(359) 0x419c8c LDR X13, [X21, #464] |
(359) 0x419c90 LDR X10, [X10, #24] |
(359) 0x419c94 MADD W11, W11, W8, WZR |
(359) 0x419c98 ADD X11, X12, W11,SXTW #3 |
(359) 0x419c9c LDR X12, [X25, #616] |
(359) 0x419ca0 ADD X14, X10, #8 |
(359) 0x419ca4 LDR X12, [X12, #24] |
(359) 0x419ca8 B 419cbc |
(359) 0x419cac FADD D0, D18, D0 |
(359) 0x419cb0 ADD X9, X9, #1 |
(359) 0x419cb4 CMP X9, X8 |
(359) 0x419cb8 B.EQ 419c24 |
(359) 0x419cbc ADD X15, X12, X9,LSL #2 |
(359) 0x419cc0 MOVI D18, #0 |
(359) 0x419cc4 LDP W17, W15, [X15] |
(359) 0x419cc8 SBFM X17, X17, #0, #31 |
(359) 0x419ccc SUB W18, W15, W17 |
(359) 0x419cd0 CMP W18, #1 |
(359) 0x419cd4 B.LT 419cac |
(362) 0x419cd8 LDR X15, [X11, X9,LSL #3] |
(362) 0x419cdc LDR D1, [X15, #8] |
(362) 0x419ce0 CMP W18, #1 |
(362) 0x419ce4 B.NE 419de0 |
(362) 0x419ce8 ORR X0, XZR, XZR |
(362) 0x419cec ORR W16, WZR, WZR |
(362) 0x419cf0 TBZ W18, #0, 419d14 |
(362) 0x419cf4 ADD X18, X10, X17,LSL #3 |
(362) 0x419cf8 ADD W17, W17, W0 |
(362) 0x419cfc LDR D2, [X18, X0,LSL #3] |
(362) 0x419d00 FCMP D2, D1 |
(362) 0x419d04 CCMP W17, W24, #4, #11 |
(362) 0x419d08 B.EQ 419d14 |
(362) 0x419d0c STR D2, [X13, X16,SXTW #3] |
(362) 0x419d10 ADD W16, W16, #1 |
(362) 0x419d14 MOVI D18, #0 |
(362) 0x419d18 CMP W16, #1 |
(362) 0x419d1c B.LT 419cac |
(362) 0x419d20 LDP D2, D3, [X15, #24] |
(362) 0x419d24 LDP D4, D5, [X15, #40] |
(362) 0x419d28 LDP D6, D7, [X15, #88] |
(362) 0x419d2c ADD X18, X15, #56 |
(362) 0x419d30 ADD X0, X15, #64 |
(362) 0x419d34 ADD X1, X15, #72 |
(362) 0x419d38 ADD X2, X15, #80 |
(362) 0x419d3c ADD X3, X15, #120 |
(362) 0x419d40 ADD X4, X15, #128 |
(362) 0x419d44 ADD X5, X15, #136 |
(362) 0x419d48 LDP D16, D17, [X15, #104] |
(362) 0x419d4c ADD X6, X15, #144 |
(362) 0x419d50 LDR D1, [X15, #568] |
(362) 0x419d54 LDR X17, [X15, #536] |
(362) 0x419d58 ORR W15, WZR, W16 |
(362) 0x419d5c ORR X16, XZR, X13 |
(362) 0x419d60 LD1 {V17.D[1]}, [X6] |
(362) 0x419d64 LD1 {V5.D[1]}, [X2] |
(362) 0x419d68 LD1 {V7.D[1]}, [X4] |
(362) 0x419d6c LD1 {V3.D[1]}, [X0] |
(362) 0x419d70 LD1 {V16.D[1]}, [X5] |
(362) 0x419d74 LD1 {V4.D[1]}, [X1] |
(362) 0x419d78 LD1 {V6.D[1]}, [X3] |
(362) 0x419d7c LD1 {V2.D[1]}, [X18] |
(363) 0x419d80 LDR D19, [X16], #8 |
(363) 0x419d84 ORR V25.16B, V17.16B, V17.16B |
(363) 0x419d88 ORR V24.16B, V5.16B, V5.16B |
(363) 0x419d8c SUBS X15, X15, #1 |
(363) 0x419d90 FMUL D19, D19, D1 |
(363) 0x419d94 FRINTZ D20, D19 |
(363) 0x419d98 FCVTZS W18, D19 |
(363) 0x419d9c FSUB D19, D19, S20 |
(363) 0x419da0 ADD X18, X17, W18,SXTW #3 |
(363) 0x419da4 FMUL D20, D19, D19 |
(363) 0x419da8 FMLA V25.2D, V16.2D, V19.D[0] |
(363) 0x419dac FMLA V24.2D, V4.2D, V19.D[0] |
(363) 0x419db0 FMUL D21, D20, D19 |
(363) 0x419db4 FMLA V25.2D, V7.2D, V20.D[0] |
(363) 0x419db8 LDP Q22, Q23, [X18] |
(363) 0x419dbc FMLA V24.2D, V3.2D, V20.D[0] |
(363) 0x419dc0 FMLA V25.2D, V6.2D, V21.D[0] |
(363) 0x419dc4 FMLA V24.2D, V2.2D, V21.D[0] |
(363) 0x419dc8 FMUL V19.2D, V25.2D, V23.2D |
(363) 0x419dcc FMLA V19.2D, V24.2D, V22.2D |
(363) 0x419dd0 FADDP D19, V19.2D |
(363) 0x419dd4 FADD D18, D18, D19 |
(363) 0x419dd8 B.NE 419d80 |
(362) 0x419ddc B 419cac |
(361) 0x419de0 AND X1, X18, #6015 |
(361) 0x419de4 SUB W2, W24, W17 |
(361) 0x419de8 ADD X3, X14, X17,LSL #3 |
(361) 0x419dec ORR X0, XZR, XZR |
(361) 0x419df0 ORR W16, WZR, WZR |
(361) 0x419df4 B 419e14 |
(361) 0x419e00 ADD X0, X0, #2 |
(361) 0x419e04 SUB X1, X1, #2 |
(361) 0x419e08 SUB W2, W2, #2 |
(361) 0x419e0c ADD X3, X3, #16 |
(361) 0x419e10 CBZ X1, 419cf0 |
(361) 0x419e14 LDUR D2, [X3, #504] |
(361) 0x419e18 FCMP D2, D1 |
(361) 0x419e1c CCMP W2, #0, #4, #11 |
(361) 0x419e20 B.NE 419e40 |
(361) 0x419e24 LDR D2, [X3] |
(361) 0x419e28 FCMP D2, D1 |
(361) 0x419e2c CCMP W2, #1, #4, #11 |
(361) 0x419e30 B.EQ 419e00 |
(361) 0x419e34 B 419e58 |
0x419e40 STR D2, [X13, X16,SXTW #3] |
0x419e44 ADD W16, W16, #1 |
0x419e48 LDR D2, [X3] |
0x419e4c FCMP D2, D1 |
0x419e50 CCMP W2, #1, #4, #11 |
0x419e54 B.EQ 419e00 |
(361) 0x419e58 STR D2, [X13, X16,SXTW #3] |
(361) 0x419e5c ADD W16, W16, #1 |
(361) 0x419e60 B 419e00 |
/home/hbollore/qaas/qaas-runs/174-135-6342/intel/miniqmc/build/miniqmc/src/Numerics/OhmmsPETE/OhmmsVector.h: 223 - 229 |
-------------------------------------------------------------------------------- |
223: return X[i]; |
[...] |
229: return X[i]; |
/home/hbollore/qaas/qaas-runs/174-135-6342/intel/miniqmc/build/miniqmc/src/QMCWaveFunctions/Jastrow/TwoBodyJastrowRef.h: 107 - 132 |
-------------------------------------------------------------------------------- |
107: for (int k = 0; k < ratios.size(); ++k) |
108: ratios[k] = std::exp(Uat[VP.refPtcl] - computeU(VP.getRefPS(), VP.refPtcl, VP.getDistTableAB(myTableID).getDistRow(k).data())); |
[...] |
126: const int igt = P.GroupID[iat] * NumGroups; |
127: for (int jg = 0; jg < NumGroups; ++jg) |
128: { |
129: const FuncType& f2(*F[igt + jg]); |
130: int iStart = P.first(jg); |
131: int iEnd = P.last(jg); |
132: curUat += f2.evaluateV(iat, iStart, iEnd, dist, DistCompressed.data()); |
/home/hbollore/qaas/qaas-runs/174-135-6342/intel/miniqmc/build/miniqmc/src/Particle/ParticleSet.h: 313 - 313 |
-------------------------------------------------------------------------------- |
313: inline int first(int igroup) const { return (*group_offsets_)[igroup]; } |
/opt/arm/gcc-14.2.0_AmazonLinux-2023/lib/gcc/aarch64-linux-gnu/14.2.0/../../../../include/c++/14.2.0/optional: 469 - 991 |
-------------------------------------------------------------------------------- |
469: { return static_cast<const _Dp*>(this)->_M_payload._M_engaged; } |
[...] |
991: if (this->_M_is_engaged()) |
/opt/arm/gcc-14.2.0_AmazonLinux-2023/lib/gcc/aarch64-linux-gnu/14.2.0/../../../../include/c++/14.2.0/bits/stl_vector.h: 993 - 1150 |
-------------------------------------------------------------------------------- |
993: { return size_type(this->_M_impl._M_finish - this->_M_impl._M_start); } |
[...] |
1131: return *(this->_M_impl._M_start + __n); |
[...] |
1150: return *(this->_M_impl._M_start + __n); |
/opt/arm/gcc-14.2.0_AmazonLinux-2023/lib/gcc/aarch64-linux-gnu/14.2.0/../../../../include/c++/14.2.0/bits/refwrap.h: 351 - 351 |
-------------------------------------------------------------------------------- |
351: { return *_M_data; } |
/home/hbollore/qaas/qaas-runs/174-135-6342/intel/miniqmc/build/miniqmc/src/QMCWaveFunctions/Jastrow/BsplineFunctor.h: 233 - 260 |
-------------------------------------------------------------------------------- |
233: const int iLimit = iEnd - iStart; |
234: |
235: #pragma vector always |
236: for (int jat = 0; jat < iLimit; jat++) |
237: { |
238: real_type r = distArray[jat]; |
239: // pick the distances smaller than the cutoff and avoid the reference atom |
240: if (r < cutoff_radius && iStart + jat != iat) |
241: distArrayCompressed[iCount++] = distArray[jat]; |
242: } |
243: |
244: real_type d = 0.0; |
245: //#pragma omp simd reduction(+:d) |
246: for (int jat = 0; jat < iCount; jat++) |
247: { |
248: real_type r = distArrayCompressed[jat]; |
249: r *= DeltaRInv; |
250: int i = (int)r; |
251: real_type t = r - real_type(i); |
252: real_type tp0 = t * t * t; |
253: real_type tp1 = t * t; |
254: real_type tp2 = t; |
255: |
256: real_type d1 = SplineCoefs[i + 0] * (A[0] * tp0 + A[1] * tp1 + A[2] * tp2 + A[3]); |
257: real_type d2 = SplineCoefs[i + 1] * (A[4] * tp0 + A[5] * tp1 + A[6] * tp2 + A[7]); |
258: real_type d3 = SplineCoefs[i + 2] * (A[8] * tp0 + A[9] * tp1 + A[10] * tp2 + A[11]); |
259: real_type d4 = SplineCoefs[i + 3] * (A[12] * tp0 + A[13] * tp1 + A[14] * tp2 + A[15]); |
260: d += (d1 + d2 + d3 + d4); |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ►100.00+ | qmcplusplus::WaveFunction::eva[...] | WaveFunction.cpp:272 | exec |
| ○ | qmcplusplus::NonLocalPP<double[...] | NonLocalPP.hpp:126 | exec |
| ○ | main.omp_outlined.63 | NewTimer.h:249 | exec |
| ○ | __kmp_invoke_microtask | libomp.so |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| Path / |
| Metric | Value |
|---|---|
| CQA speedup if no scalar integer | 1.20 |
| CQA speedup if FP arith vectorized | 1.00 |
| CQA speedup if fully vectorized | 2.00 |
| CQA speedup if no inter-iteration dependency | NA |
| CQA speedup if next bottleneck killed | 1.20 |
| Bottlenecks | P8, P9, |
| Function | miniqmcreference::TwoBodyJastrowRef |
| Source | BsplineFunctor.h:238-241 |
| Source loop unroll info | NA |
| Source loop unroll confidence level | NA |
| Unroll/vectorization loop type | NA |
| Unroll factor | NA |
| CQA cycles | 1.00 |
| CQA cycles if no scalar integer | 0.83 |
| CQA cycles if FP arith vectorized | 1.00 |
| CQA cycles if fully vectorized | 0.50 |
| Front-end cycles | 0.75 |
| P0 cycles | 0.50 |
| P1 cycles | 0.50 |
| P2 cycles | 0.42 |
| P3 cycles | 0.42 |
| P4 cycles | 0.33 |
| P5 cycles | 0.33 |
| P6 cycles | 0.25 |
| P7 cycles | 0.25 |
| P8 cycles | 1.00 |
| P9 cycles | 1.00 |
| P10 cycles | 0.00 |
| P11 cycles | 0.00 |
| P12 cycles | 0.83 |
| P13 cycles | 0.50 |
| P14 cycles | 0.67 |
| P15 cycles | 0.00 |
| P16 cycles | 0.00 |
| DIV/SQRT cycles | 0.00 |
| Inter-iter dependencies cycles | NA |
| FE+BE cycles (UFS) | NA |
| Stall cycles (UFS) | NA |
| Nb insns | 6.00 |
| Nb uops | 6.00 |
| Nb loads | NA |
| Nb stores | 1.00 |
| Nb stack references | 0.00 |
| FLOP/cycle | 0.00 |
| Nb FLOP add-sub | 0.00 |
| Nb FLOP mul | 0.00 |
| Nb FLOP fma | 0.00 |
| Nb FLOP div | 0.00 |
| Nb FLOP rcp | 0.00 |
| Nb FLOP sqrt | 0.00 |
| Nb FLOP rsqrt | 0.00 |
| Bytes/cycle | 16.00 |
| Bytes prefetched | 0.00 |
| Bytes loaded | 8.00 |
| Bytes stored | 8.00 |
| Stride 0 | NA |
| Stride 1 | NA |
| Stride n | NA |
| Stride unknown | NA |
| Stride indirect | NA |
| Vectorization ratio all | 0.00 |
| Vectorization ratio load | 0.00 |
| Vectorization ratio store | 0.00 |
| Vectorization ratio mul | NA |
| Vectorization ratio add_sub | NA |
| Vectorization ratio fma | NA |
| Vectorization ratio div_sqrt | NA |
| Vectorization ratio other | 0.00 |
| Vector-efficiency ratio all | 50.00 |
| Vector-efficiency ratio load | 50.00 |
| Vector-efficiency ratio store | 50.00 |
| Vector-efficiency ratio mul | NA |
| Vector-efficiency ratio add_sub | NA |
| Vector-efficiency ratio fma | NA |
| Vector-efficiency ratio div_sqrt | NA |
| Vector-efficiency ratio other | 50.00 |
| Metric | Value |
|---|---|
| CQA speedup if no scalar integer | 1.20 |
| CQA speedup if FP arith vectorized | 1.00 |
| CQA speedup if fully vectorized | 2.00 |
| CQA speedup if no inter-iteration dependency | NA |
| CQA speedup if next bottleneck killed | 1.20 |
| Bottlenecks | P8, P9, |
| Function | miniqmcreference::TwoBodyJastrowRef |
| Source | BsplineFunctor.h:238-241 |
| Source loop unroll info | NA |
| Source loop unroll confidence level | NA |
| Unroll/vectorization loop type | NA |
| Unroll factor | NA |
| CQA cycles | 1.00 |
| CQA cycles if no scalar integer | 0.83 |
| CQA cycles if FP arith vectorized | 1.00 |
| CQA cycles if fully vectorized | 0.50 |
| Front-end cycles | 0.75 |
| P0 cycles | 0.50 |
| P1 cycles | 0.50 |
| P2 cycles | 0.42 |
| P3 cycles | 0.42 |
| P4 cycles | 0.33 |
| P5 cycles | 0.33 |
| P6 cycles | 0.25 |
| P7 cycles | 0.25 |
| P8 cycles | 1.00 |
| P9 cycles | 1.00 |
| P10 cycles | 0.00 |
| P11 cycles | 0.00 |
| P12 cycles | 0.83 |
| P13 cycles | 0.50 |
| P14 cycles | 0.67 |
| P15 cycles | 0.00 |
| P16 cycles | 0.00 |
| DIV/SQRT cycles | 0.00 |
| Inter-iter dependencies cycles | NA |
| FE+BE cycles (UFS) | NA |
| Stall cycles (UFS) | NA |
| Nb insns | 6.00 |
| Nb uops | 6.00 |
| Nb loads | NA |
| Nb stores | 1.00 |
| Nb stack references | 0.00 |
| FLOP/cycle | 0.00 |
| Nb FLOP add-sub | 0.00 |
| Nb FLOP mul | 0.00 |
| Nb FLOP fma | 0.00 |
| Nb FLOP div | 0.00 |
| Nb FLOP rcp | 0.00 |
| Nb FLOP sqrt | 0.00 |
| Nb FLOP rsqrt | 0.00 |
| Bytes/cycle | 16.00 |
| Bytes prefetched | 0.00 |
| Bytes loaded | 8.00 |
| Bytes stored | 8.00 |
| Stride 0 | NA |
| Stride 1 | NA |
| Stride n | NA |
| Stride unknown | NA |
| Stride indirect | NA |
| Vectorization ratio all | 0.00 |
| Vectorization ratio load | 0.00 |
| Vectorization ratio store | 0.00 |
| Vectorization ratio mul | NA |
| Vectorization ratio add_sub | NA |
| Vectorization ratio fma | NA |
| Vectorization ratio div_sqrt | NA |
| Vectorization ratio other | 0.00 |
| Vector-efficiency ratio all | 50.00 |
| Vector-efficiency ratio load | 50.00 |
| Vector-efficiency ratio store | 50.00 |
| Vector-efficiency ratio mul | NA |
| Vector-efficiency ratio add_sub | NA |
| Vector-efficiency ratio fma | NA |
| Vector-efficiency ratio div_sqrt | NA |
| Vector-efficiency ratio other | 50.00 |
| Path / |
| Function | miniqmcreference::TwoBodyJastrowRef |
| Source file and lines | BsplineFunctor.h:233-260 |
| Module | exec |
| nb instructions | 6 |
| loop length | 24 |
| nb stack references | 0 |
| front end | 0.75 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | P16 | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| uops | 0.50 | 0.50 | 0.42 | 0.42 | 0.33 | 0.33 | 0.25 | 0.25 | 1.00 | 1.00 | 0.00 | 0.00 | 0.83 | 0.50 | 0.67 | 0.00 | 0.00 |
| cycles | 0.50 | 0.50 | 0.42 | 0.42 | 0.33 | 0.33 | 0.25 | 0.25 | 1.00 | 1.00 | 0.00 | 0.00 | 0.83 | 0.50 | 0.67 | 0.00 | 0.00 |
| Cycles executing div or sqrt instructions | NA |
| Front-end | 0.75 |
| Overall L1 | 1.00 |
| all | 0% |
| load | 0% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 0% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 0% |
| all | 0% |
| load | 0% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 0% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | P16 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| STR D2, [X13, X16,SXTW #3] | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 2 | 0.50 | scal (50.0%) |
| ADD W16, W16, #1 | 1 | 0 | 0 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| LDR D2, [X3] | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 6 | 0.33 | scal (50.0%) |
| FCMP D2, D1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 1 | scal (50.0%) |
| CCMP W2, #1, #4, #11 | 1 | 0 | 0 | 0.25 | 0.25 | 0 | 0 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 1 | N/A |
| B.EQ 419e00 <_ZN16miniqmcreference17TwoBodyJastrowRefIN11qmcplusplus14BsplineFunctorIdEEE14evaluateRatiosERNS1_18VirtualParticleSetERSt6vectorIdSaIdEE+0x220> | 1 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
| Function | miniqmcreference::TwoBodyJastrowRef |
| Source file and lines | BsplineFunctor.h:233-260 |
| Module | exec |
| nb instructions | 6 |
| loop length | 24 |
| nb stack references | 0 |
| front end | 0.75 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | P16 | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| uops | 0.50 | 0.50 | 0.42 | 0.42 | 0.33 | 0.33 | 0.25 | 0.25 | 1.00 | 1.00 | 0.00 | 0.00 | 0.83 | 0.50 | 0.67 | 0.00 | 0.00 |
| cycles | 0.50 | 0.50 | 0.42 | 0.42 | 0.33 | 0.33 | 0.25 | 0.25 | 1.00 | 1.00 | 0.00 | 0.00 | 0.83 | 0.50 | 0.67 | 0.00 | 0.00 |
| Cycles executing div or sqrt instructions | NA |
| Front-end | 0.75 |
| Overall L1 | 1.00 |
| all | 0% |
| load | 0% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 0% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 0% |
| all | 0% |
| load | 0% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 0% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | P16 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| STR D2, [X13, X16,SXTW #3] | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 2 | 0.50 | scal (50.0%) |
| ADD W16, W16, #1 | 1 | 0 | 0 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| LDR D2, [X3] | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 6 | 0.33 | scal (50.0%) |
| FCMP D2, D1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 1 | scal (50.0%) |
| CCMP W2, #1, #4, #11 | 1 | 0 | 0 | 0.25 | 0.25 | 0 | 0 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 1 | N/A |
| B.EQ 419e00 <_ZN16miniqmcreference17TwoBodyJastrowRefIN11qmcplusplus14BsplineFunctorIdEEE14evaluateRatiosERNS1_18VirtualParticleSetERSt6vectorIdSaIdEE+0x220> | 1 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
