Loop Id: 96 | Module: libparcsr_mv.so | Source: par_csr_matop.c:109-231 [...] | Coverage: 1.11% |
---|
Loop Id: 96 | Module: libparcsr_mv.so | Source: par_csr_matop.c:109-231 [...] | Coverage: 1.11% |
---|
0xcf40 MOV -0x30(%RBP),%R11 |
0xcf44 CMP %R11,%R12 |
0xcf47 LEA 0x1(%R12),%R12 |
0xcf4c MOV -0x38(%RBP),%R13 |
0xcf50 JE c980 |
0xcf56 LEA (%R13,%R12,1),%RAX |
0xcf5b MOV -0x78(%RBP),%RCX |
0xcf5f MOV (%RCX,%RAX,8),%RDI |
0xcf63 MOV 0x20(%RBP),%RCX |
0xcf67 MOV (%RCX,%RDI,8),%RAX |
0xcf6b MOV 0x8(%RCX,%RDI,8),%R13 |
0xcf70 MOV %R13,%R9 |
0xcf73 SUB %RAX,%R9 |
0xcf76 JLE d144 |
0xcf7c CMP $0x8,%R9 |
0xcf80 JAE d000 |
0xcf86 MOV %R9,%RCX |
0xcf89 AND $-0x8,%RCX |
0xcf8d CMP %R9,%RCX |
0xcf90 JAE d140 |
0xcf96 ADD %RCX,%RAX |
0xcf99 MOV 0x28(%RBP),%RSI |
0xcf9d MOV -0x30(%RBP),%R11 |
0xcfa1 JMP cfcc |
(99) 0xcfc0 INC %RAX |
(99) 0xcfc3 CMP %RAX,%R13 |
(99) 0xcfc6 JE d144 |
(99) 0xcfcc MOV (%RSI,%RAX,8),%RCX |
(99) 0xcfd0 CMP %R8,(%R14,%RCX,8) |
(99) 0xcfd4 JGE cfc0 |
(99) 0xcfd6 MOV %RBX,(%R14,%RCX,8) |
(99) 0xcfda INC %RBX |
(99) 0xcfdd JMP cfc0 |
0xd000 MOV %R9,%RCX |
0xd003 SHR $0x3,%RCX |
0xd007 MOV -0x70(%RBP),%RSI |
0xd00b LEA (%RSI,%RAX,8),%R11 |
0xd00f JMP d04d |
(100) 0xd040 ADD $0x40,%R11 |
(100) 0xd044 DEC %RCX |
(100) 0xd047 JE cf86 |
(100) 0xd04d MOV -0x38(%R11),%RSI |
(100) 0xd051 CMP %R8,(%R14,%RSI,8) |
(100) 0xd055 JGE d0c0 |
(100) 0xd057 MOV %RBX,(%R14,%RSI,8) |
(100) 0xd05b INC %RBX |
(100) 0xd05e MOV -0x30(%R11),%RSI |
(100) 0xd062 CMP %R8,(%R14,%RSI,8) |
(100) 0xd066 JL d0ca |
(100) 0xd068 MOV -0x28(%R11),%RSI |
(100) 0xd06c CMP %R8,(%R14,%RSI,8) |
(100) 0xd070 JGE d0db |
(100) 0xd072 MOV %RBX,(%R14,%RSI,8) |
(100) 0xd076 INC %RBX |
(100) 0xd079 MOV -0x20(%R11),%RSI |
(100) 0xd07d CMP %R8,(%R14,%RSI,8) |
(100) 0xd081 JL d0e5 |
(100) 0xd083 MOV -0x18(%R11),%RSI |
(100) 0xd087 CMP %R8,(%R14,%RSI,8) |
(100) 0xd08b JGE d0f6 |
(100) 0xd08d MOV %RBX,(%R14,%RSI,8) |
(100) 0xd091 INC %RBX |
(100) 0xd094 MOV -0x10(%R11),%RSI |
(100) 0xd098 CMP %R8,(%R14,%RSI,8) |
(100) 0xd09c JL d100 |
(100) 0xd09e MOV -0x8(%R11),%RSI |
(100) 0xd0a2 CMP %R8,(%R14,%RSI,8) |
(100) 0xd0a6 JGE d111 |
(100) 0xd0a8 MOV %RBX,(%R14,%RSI,8) |
(100) 0xd0ac INC %RBX |
(100) 0xd0af MOV (%R11),%RSI |
(100) 0xd0b2 CMP %R8,(%R14,%RSI,8) |
(100) 0xd0b6 JGE d040 |
(100) 0xd0b8 JMP d11e |
(100) 0xd0c0 MOV -0x30(%R11),%RSI |
(100) 0xd0c4 CMP %R8,(%R14,%RSI,8) |
(100) 0xd0c8 JGE d068 |
(100) 0xd0ca MOV %RBX,(%R14,%RSI,8) |
(100) 0xd0ce INC %RBX |
(100) 0xd0d1 MOV -0x28(%R11),%RSI |
(100) 0xd0d5 CMP %R8,(%R14,%RSI,8) |
(100) 0xd0d9 JL d072 |
(100) 0xd0db MOV -0x20(%R11),%RSI |
(100) 0xd0df CMP %R8,(%R14,%RSI,8) |
(100) 0xd0e3 JGE d083 |
(100) 0xd0e5 MOV %RBX,(%R14,%RSI,8) |
(100) 0xd0e9 INC %RBX |
(100) 0xd0ec MOV -0x18(%R11),%RSI |
(100) 0xd0f0 CMP %R8,(%R14,%RSI,8) |
(100) 0xd0f4 JL d08d |
(100) 0xd0f6 MOV -0x10(%R11),%RSI |
(100) 0xd0fa CMP %R8,(%R14,%RSI,8) |
(100) 0xd0fe JGE d09e |
(100) 0xd100 MOV %RBX,(%R14,%RSI,8) |
(100) 0xd104 INC %RBX |
(100) 0xd107 MOV -0x8(%R11),%RSI |
(100) 0xd10b CMP %R8,(%R14,%RSI,8) |
(100) 0xd10f JL d0a8 |
(100) 0xd111 MOV (%R11),%RSI |
(100) 0xd114 CMP %R8,(%R14,%RSI,8) |
(100) 0xd118 JGE d040 |
(100) 0xd11e MOV %RBX,(%R14,%RSI,8) |
(100) 0xd122 INC %RBX |
(100) 0xd125 JMP d040 |
0xd140 MOV -0x30(%RBP),%R11 |
0xd144 MOV 0x30(%RBP),%RCX |
0xd148 MOV (%RCX,%RDI,8),%RAX |
0xd14c MOV 0x8(%RCX,%RDI,8),%RCX |
0xd151 MOV %RCX,%RDI |
0xd154 SUB %RAX,%RDI |
0xd157 JLE cf44 |
0xd15d CMP $0x4,%RDI |
0xd161 JAE d200 |
0xd167 MOV 0x60(%RBP),%R13 |
0xd16b MOV %RDI,%RSI |
0xd16e AND $-0x4,%RSI |
0xd172 CMP %RDI,%RSI |
0xd175 JAE cf40 |
0xd17b ADD %RSI,%RAX |
0xd17e MOV 0x38(%RBP),%RDI |
0xd182 MOV -0x30(%RBP),%R11 |
0xd186 JMP d1cc |
(97) 0xd1c0 INC %RAX |
(97) 0xd1c3 CMP %RAX,%RCX |
(97) 0xd1c6 JE cf44 |
(97) 0xd1cc MOV (%RDI,%RAX,8),%RSI |
(97) 0xd1d0 MOV (%R13,%RSI,8),%RSI |
(97) 0xd1d5 ADD %R15,%RSI |
(97) 0xd1d8 CMP %R10,(%R14,%RSI,8) |
(97) 0xd1dc JGE d1c0 |
(97) 0xd1de MOV %RDX,(%R14,%RSI,8) |
(97) 0xd1e2 INC %RDX |
(97) 0xd1e5 JMP d1c0 |
0xd200 MOV %RDI,%R9 |
0xd203 SHR $0x2,%R9 |
0xd207 MOV -0xc0(%RBP),%RSI |
0xd20e LEA (%RSI,%RAX,8),%R11 |
0xd212 MOV 0x60(%RBP),%R13 |
0xd216 JMP d24d |
(98) 0xd240 ADD $0x20,%R11 |
(98) 0xd244 DEC %R9 |
(98) 0xd247 JE d16b |
(98) 0xd24d MOV -0x18(%R11),%RSI |
(98) 0xd251 MOV (%R13,%RSI,8),%RSI |
(98) 0xd256 ADD %R15,%RSI |
(98) 0xd259 CMP %R10,(%R14,%RSI,8) |
(98) 0xd25d JGE d266 |
(98) 0xd25f MOV %RDX,(%R14,%RSI,8) |
(98) 0xd263 INC %RDX |
(98) 0xd266 MOV -0x10(%R11),%RSI |
(98) 0xd26a MOV (%R13,%RSI,8),%RSI |
(98) 0xd26f ADD %R15,%RSI |
(98) 0xd272 CMP %R10,(%R14,%RSI,8) |
(98) 0xd276 JGE d27f |
(98) 0xd278 MOV %RDX,(%R14,%RSI,8) |
(98) 0xd27c INC %RDX |
(98) 0xd27f MOV -0x8(%R11),%RSI |
(98) 0xd283 MOV (%R13,%RSI,8),%RSI |
(98) 0xd288 ADD %R15,%RSI |
(98) 0xd28b CMP %R10,(%R14,%RSI,8) |
(98) 0xd28f JGE d298 |
(98) 0xd291 MOV %RDX,(%R14,%RSI,8) |
(98) 0xd295 INC %RDX |
(98) 0xd298 MOV (%R11),%RSI |
(98) 0xd29b MOV (%R13,%RSI,8),%RSI |
(98) 0xd2a0 ADD %R15,%RSI |
(98) 0xd2a3 CMP %R10,(%R14,%RSI,8) |
(98) 0xd2a7 JGE d240 |
(98) 0xd2a9 MOV %RDX,(%R14,%RSI,8) |
(98) 0xd2ad INC %RDX |
(98) 0xd2b0 JMP d240 |
/home/eoseret/qaas_runs_CPU_9468/171-716-5699/intel/AMG/build/AMG/AMG/parcsr_mv/par_csr_matop.c: 109 - 231 |
-------------------------------------------------------------------------------- |
109: if (ii < rest) |
[...] |
187: for (jj2 = A_diag_i[i1]; jj2 < A_diag_i[i1+1]; jj2++) |
188: { |
189: i2 = A_diag_j[jj2]; |
[...] |
195: for (jj3 = B_diag_i[i2]; jj3 < B_diag_i[i2+1]; jj3++) |
196: { |
197: i3 = B_diag_j[jj3]; |
[...] |
205: if (B_marker[i3] < jj_row_begin_diag) |
206: { |
207: B_marker[i3] = jj_count_diag; |
208: jj_count_diag++; |
[...] |
216: if (num_cols_offd_B) |
217: { |
218: for (jj3 = B_offd_i[i2]; jj3 < B_offd_i[i2+1]; jj3++) |
219: { |
220: i3 = num_cols_diag_B+map_B_to_C[B_offd_j[jj3]]; |
[...] |
228: if (B_marker[i3] < jj_row_begin_offd) |
229: { |
230: B_marker[i3] = jj_count_offd; |
231: jj_count_offd++; |
Path / |
Metric | Value |
---|---|
CQA speedup if no scalar integer | 1.00 |
CQA speedup if FP arith vectorized | 1.00 |
CQA speedup if fully vectorized | 8.00 |
CQA speedup if no inter-iteration dependency | NA |
CQA speedup if next bottleneck killed | 1.39 |
Bottlenecks | micro-operation queue, |
Function | hypre_ParMatmul_RowSizes.extracted |
Source | par_csr_matop.c:109-109,par_csr_matop.c:187-189,par_csr_matop.c:195-195,par_csr_matop.c:208-208,par_csr_matop.c:216-218 |
Source loop unroll info | NA |
Source loop unroll confidence level | NA |
Unroll/vectorization loop type | NA |
Unroll factor | NA |
CQA cycles | 8.83 |
CQA cycles if no scalar integer | 8.83 |
CQA cycles if FP arith vectorized | 8.83 |
CQA cycles if fully vectorized | 1.10 |
Front-end cycles | 8.83 |
DIV/SQRT cycles | 4.50 |
P0 cycles | 3.80 |
P1 cycles | 6.33 |
P2 cycles | 6.33 |
P3 cycles | 0.00 |
P4 cycles | 3.60 |
P5 cycles | 4.50 |
P6 cycles | 0.00 |
P7 cycles | 0.00 |
P8 cycles | 0.00 |
P9 cycles | 3.60 |
P10 cycles | 6.33 |
P11 cycles | 0.00 |
Inter-iter dependencies cycles | NA |
FE+BE cycles (UFS) | 15.34 - 16.35 |
Stall cycles (UFS) | 6.08 - 7.09 |
Nb insns | 53.00 |
Nb uops | 53.00 |
Nb loads | 19.00 |
Nb stores | 0.00 |
Nb stack references | 10.00 |
FLOP/cycle | 0.00 |
Nb FLOP add-sub | 0.00 |
Nb FLOP mul | 0.00 |
Nb FLOP fma | 0.00 |
Nb FLOP div | 0.00 |
Nb FLOP rcp | 0.00 |
Nb FLOP sqrt | 0.00 |
Nb FLOP rsqrt | 0.00 |
Bytes/cycle | 17.21 |
Bytes prefetched | 0.00 |
Bytes loaded | 152.00 |
Bytes stored | 0.00 |
Stride 0 | NA |
Stride 1 | NA |
Stride n | NA |
Stride unknown | NA |
Stride indirect | NA |
Vectorization ratio all | 0.00 |
Vectorization ratio load | 0.00 |
Vectorization ratio store | NA |
Vectorization ratio mul | NA |
Vectorization ratio add_sub | 0.00 |
Vectorization ratio fma | NA |
Vectorization ratio div_sqrt | NA |
Vectorization ratio other | 0.00 |
Vector-efficiency ratio all | 12.50 |
Vector-efficiency ratio load | 12.50 |
Vector-efficiency ratio store | NA |
Vector-efficiency ratio mul | NA |
Vector-efficiency ratio add_sub | 12.50 |
Vector-efficiency ratio fma | NA |
Vector-efficiency ratio div_sqrt | NA |
Vector-efficiency ratio other | 12.50 |
Metric | Value |
---|---|
CQA speedup if no scalar integer | 1.00 |
CQA speedup if FP arith vectorized | 1.00 |
CQA speedup if fully vectorized | 8.00 |
CQA speedup if no inter-iteration dependency | NA |
CQA speedup if next bottleneck killed | 1.39 |
Bottlenecks | micro-operation queue, |
Function | hypre_ParMatmul_RowSizes.extracted |
Source | par_csr_matop.c:109-109,par_csr_matop.c:187-189,par_csr_matop.c:195-195,par_csr_matop.c:208-208,par_csr_matop.c:216-218 |
Source loop unroll info | NA |
Source loop unroll confidence level | NA |
Unroll/vectorization loop type | NA |
Unroll factor | NA |
CQA cycles | 8.83 |
CQA cycles if no scalar integer | 8.83 |
CQA cycles if FP arith vectorized | 8.83 |
CQA cycles if fully vectorized | 1.10 |
Front-end cycles | 8.83 |
DIV/SQRT cycles | 4.50 |
P0 cycles | 3.80 |
P1 cycles | 6.33 |
P2 cycles | 6.33 |
P3 cycles | 0.00 |
P4 cycles | 3.60 |
P5 cycles | 4.50 |
P6 cycles | 0.00 |
P7 cycles | 0.00 |
P8 cycles | 0.00 |
P9 cycles | 3.60 |
P10 cycles | 6.33 |
P11 cycles | 0.00 |
Inter-iter dependencies cycles | NA |
FE+BE cycles (UFS) | 15.34 - 16.35 |
Stall cycles (UFS) | 6.08 - 7.09 |
Nb insns | 53.00 |
Nb uops | 53.00 |
Nb loads | 19.00 |
Nb stores | 0.00 |
Nb stack references | 10.00 |
FLOP/cycle | 0.00 |
Nb FLOP add-sub | 0.00 |
Nb FLOP mul | 0.00 |
Nb FLOP fma | 0.00 |
Nb FLOP div | 0.00 |
Nb FLOP rcp | 0.00 |
Nb FLOP sqrt | 0.00 |
Nb FLOP rsqrt | 0.00 |
Bytes/cycle | 17.21 |
Bytes prefetched | 0.00 |
Bytes loaded | 152.00 |
Bytes stored | 0.00 |
Stride 0 | NA |
Stride 1 | NA |
Stride n | NA |
Stride unknown | NA |
Stride indirect | NA |
Vectorization ratio all | 0.00 |
Vectorization ratio load | 0.00 |
Vectorization ratio store | NA |
Vectorization ratio mul | NA |
Vectorization ratio add_sub | 0.00 |
Vectorization ratio fma | NA |
Vectorization ratio div_sqrt | NA |
Vectorization ratio other | 0.00 |
Vector-efficiency ratio all | 12.50 |
Vector-efficiency ratio load | 12.50 |
Vector-efficiency ratio store | NA |
Vector-efficiency ratio mul | NA |
Vector-efficiency ratio add_sub | 12.50 |
Vector-efficiency ratio fma | NA |
Vector-efficiency ratio div_sqrt | NA |
Vector-efficiency ratio other | 12.50 |
Path / |
Function | hypre_ParMatmul_RowSizes.extracted |
Source file and lines | par_csr_matop.c:109-231 |
Module | libparcsr_mv.so |
nb instructions | 53 |
nb uops | 53 |
loop length | 212 |
used x86 registers | 9 |
used mmx registers | 0 |
used xmm registers | 0 |
used ymm registers | 0 |
used zmm registers | 0 |
nb stack references | 10 |
micro-operation queue | 8.83 cycles |
front end | 8.83 cycles |
P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 4.50 | 3.80 | 6.33 | 6.33 | 0.00 | 3.60 | 4.50 | 0.00 | 0.00 | 0.00 | 3.60 | 6.33 |
cycles | 4.50 | 3.80 | 6.33 | 6.33 | 0.00 | 3.60 | 4.50 | 0.00 | 0.00 | 0.00 | 3.60 | 6.33 |
Cycles executing div or sqrt instructions | NA |
FE+BE cycles | 15.34-16.35 |
Stall cycles | 6.08-7.09 |
LM full (events) | 8.52-9.49 |
Front-end | 8.83 |
Dispatch | 6.33 |
Overall L1 | 8.83 |
all | 0% |
load | 0% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 0% |
all | 12% |
load | 12% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 12% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 12% |
Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
MOV -0x30(%RBP),%R11 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
CMP %R11,%R12 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
LEA 0x1(%R12),%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV -0x38(%RBP),%R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JE c980 <hypre_ParMatmul_RowSizes.extracted+0x2c0> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
LEA (%R13,%R12,1),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV -0x78(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RCX,%RAX,8),%RDI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x20(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RCX,%RDI,8),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x8(%RCX,%RDI,8),%R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %R13,%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %RAX,%R9 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE d144 <hypre_ParMatmul_RowSizes.extracted+0xa84> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x8,%R9 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE d000 <hypre_ParMatmul_RowSizes.extracted+0x940> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %R9,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x8,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
CMP %R9,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE d140 <hypre_ParMatmul_RowSizes.extracted+0xa80> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
ADD %RCX,%RAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV 0x28(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x30(%RBP),%R11 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JMP cfcc <hypre_ParMatmul_RowSizes.extracted+0x90c> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
MOV %R9,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SHR $0x3,%RCX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
MOV -0x70(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA (%RSI,%RAX,8),%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP d04d <hypre_ParMatmul_RowSizes.extracted+0x98d> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
MOV -0x30(%RBP),%R11 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x30(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RCX,%RDI,8),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x8(%RCX,%RDI,8),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RCX,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %RAX,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE cf44 <hypre_ParMatmul_RowSizes.extracted+0x884> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x4,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE d200 <hypre_ParMatmul_RowSizes.extracted+0xb40> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV 0x60(%RBP),%R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RDI,%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x4,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
CMP %RDI,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE cf40 <hypre_ParMatmul_RowSizes.extracted+0x880> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
ADD %RSI,%RAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV 0x38(%RBP),%RDI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x30(%RBP),%R11 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JMP d1cc <hypre_ParMatmul_RowSizes.extracted+0xb0c> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
MOV %RDI,%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SHR $0x2,%R9 | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
MOV -0xc0(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA (%RSI,%RAX,8),%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV 0x60(%RBP),%R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JMP d24d <hypre_ParMatmul_RowSizes.extracted+0xb8d> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
Function | hypre_ParMatmul_RowSizes.extracted |
Source file and lines | par_csr_matop.c:109-231 |
Module | libparcsr_mv.so |
nb instructions | 53 |
nb uops | 53 |
loop length | 212 |
used x86 registers | 9 |
used mmx registers | 0 |
used xmm registers | 0 |
used ymm registers | 0 |
used zmm registers | 0 |
nb stack references | 10 |
micro-operation queue | 8.83 cycles |
front end | 8.83 cycles |
P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 4.50 | 3.80 | 6.33 | 6.33 | 0.00 | 3.60 | 4.50 | 0.00 | 0.00 | 0.00 | 3.60 | 6.33 |
cycles | 4.50 | 3.80 | 6.33 | 6.33 | 0.00 | 3.60 | 4.50 | 0.00 | 0.00 | 0.00 | 3.60 | 6.33 |
Cycles executing div or sqrt instructions | NA |
FE+BE cycles | 15.34-16.35 |
Stall cycles | 6.08-7.09 |
LM full (events) | 8.52-9.49 |
Front-end | 8.83 |
Dispatch | 6.33 |
Overall L1 | 8.83 |
all | 0% |
load | 0% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 0% |
all | 12% |
load | 12% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 12% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 12% |
Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
MOV -0x30(%RBP),%R11 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
CMP %R11,%R12 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
LEA 0x1(%R12),%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV -0x38(%RBP),%R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JE c980 <hypre_ParMatmul_RowSizes.extracted+0x2c0> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
LEA (%R13,%R12,1),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV -0x78(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RCX,%RAX,8),%RDI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x20(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RCX,%RDI,8),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x8(%RCX,%RDI,8),%R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %R13,%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %RAX,%R9 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE d144 <hypre_ParMatmul_RowSizes.extracted+0xa84> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x8,%R9 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE d000 <hypre_ParMatmul_RowSizes.extracted+0x940> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %R9,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x8,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
CMP %R9,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE d140 <hypre_ParMatmul_RowSizes.extracted+0xa80> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
ADD %RCX,%RAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV 0x28(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x30(%RBP),%R11 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JMP cfcc <hypre_ParMatmul_RowSizes.extracted+0x90c> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
MOV %R9,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SHR $0x3,%RCX | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
MOV -0x70(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA (%RSI,%RAX,8),%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP d04d <hypre_ParMatmul_RowSizes.extracted+0x98d> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
MOV -0x30(%RBP),%R11 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x30(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RCX,%RDI,8),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x8(%RCX,%RDI,8),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RCX,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %RAX,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE cf44 <hypre_ParMatmul_RowSizes.extracted+0x884> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x4,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE d200 <hypre_ParMatmul_RowSizes.extracted+0xb40> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV 0x60(%RBP),%R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RDI,%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x4,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
CMP %RDI,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE cf40 <hypre_ParMatmul_RowSizes.extracted+0x880> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
ADD %RSI,%RAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV 0x38(%RBP),%RDI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x30(%RBP),%R11 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JMP d1cc <hypre_ParMatmul_RowSizes.extracted+0xb0c> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
MOV %RDI,%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SHR $0x2,%R9 | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
MOV -0xc0(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA (%RSI,%RAX,8),%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV 0x60(%RBP),%R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JMP d24d <hypre_ParMatmul_RowSizes.extracted+0xb8d> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |