Function: hypre_BoomerAMGCoarsenPMIS.extracted | Module: libparcsr_ls.so | Source: par_coarsen.c:2516-2576 [...] | Coverage: 0.16% |
---|
Function: hypre_BoomerAMGCoarsenPMIS.extracted | Module: libparcsr_ls.so | Source: par_coarsen.c:2516-2576 [...] | Coverage: 0.16% |
---|
/home/eoseret/qaas_runs_CPU_9468/171-147-2675/intel/AMG/build/AMG/AMG/parcsr_ls/par_coarsen.c: 2516 - 2576 |
-------------------------------------------------------------------------------- |
2516: #pragma omp parallel private(ig,i) |
[...] |
2523: hypre_GetSimpleThreadPartition(&ig_begin, &ig_end, graph_size); |
2524: |
2525: HYPRE_Int ig_offd_begin, ig_offd_end; |
2526: hypre_GetSimpleThreadPartition(&ig_offd_begin, &ig_offd_end, graph_offd_size); |
2527: |
2528: for (ig = ig_begin; ig < ig_end; ig++) |
2529: { |
2530: i = graph_array[ig]; |
2531: |
2532: if (CF_marker[i]!=0) /* C or F point */ |
2533: { |
2534: /* the independent set subroutine needs measure 0 for |
2535: removed nodes */ |
2536: measure_array[i] = 0; |
2537: } |
2538: else |
2539: { |
2540: private_graph_size_cnt++; |
2541: } |
2542: } |
2543: |
2544: for (ig = ig_offd_begin; ig < ig_offd_end; ig++) |
2545: { |
2546: i = graph_array_offd[ig]; |
2547: |
2548: if (CF_marker_offd[i]!=0) /* C of F point */ |
2549: { |
2550: /* the independent set subroutine needs measure 0 for |
2551: removed nodes */ |
2552: measure_array[i + num_variables] = 0; |
2553: } |
2554: else |
2555: { |
2556: private_graph_offd_size_cnt++; |
2557: } |
2558: } |
2559: |
2560: hypre_prefix_sum_pair(&private_graph_size_cnt, &graph_size, &private_graph_offd_size_cnt, &graph_offd_size, prefix_sum_workspace); |
2561: |
2562: for (ig = ig_begin; ig < ig_end; ig++) |
2563: { |
2564: i = graph_array[ig]; |
2565: if (CF_marker[i]==0) |
2566: { |
2567: graph_array2[private_graph_size_cnt++] = i; |
2568: } |
2569: } |
2570: |
2571: for (ig = ig_offd_begin; ig < ig_offd_end; ig++) |
2572: { |
2573: i = graph_array_offd[ig]; |
2574: if (CF_marker_offd[i]==0) |
2575: { |
2576: graph_array_offd2[private_graph_offd_size_cnt++] = i; |
0x36d60 PUSH %RBP |
0x36d61 MOV %RSP,%RBP |
0x36d64 PUSH %R15 |
0x36d66 PUSH %R14 |
0x36d68 PUSH %R13 |
0x36d6a PUSH %R12 |
0x36d6c PUSH %RBX |
0x36d6d SUB $0x38,%RSP |
0x36d71 MOV %R9,%R13 |
0x36d74 MOV %R8,%RBX |
0x36d77 MOV %RCX,%R12 |
0x36d7a MOV %RDX,%R15 |
0x36d7d MOV 0x28(%RBP),%R14 |
0x36d81 MOV 0x20(%RBP),%RAX |
0x36d85 MOV (%RAX),%RDX |
0x36d88 LEA -0x40(%RBP),%RDI |
0x36d8c LEA -0x48(%RBP),%RSI |
0x36d90 CALL e050 <hypre_GetSimpleThreadPartition@plt> |
0x36d95 MOV (%R14),%RDX |
0x36d98 LEA -0x50(%RBP),%RDI |
0x36d9c LEA -0x58(%RBP),%RSI |
0x36da0 CALL e050 <hypre_GetSimpleThreadPartition@plt> |
0x36da5 MOV -0x40(%RBP),%RSI |
0x36da9 MOV -0x48(%RBP),%RAX |
0x36dad MOV %RAX,%RDI |
0x36db0 SUB %RSI,%RDI |
0x36db3 JLE 36e78 |
0x36db9 MOV (%R12),%RCX |
0x36dbd MOV %RDI,%RDX |
0x36dc0 AND $-0x8,%RDX |
0x36dc4 JE 3726a |
0x36dca LEA -0x1(%RDX),%R8 |
0x36dce MOV 0x10(%RBP),%R9 |
0x36dd2 LEA (%R9,%RSI,8),%R9 |
0x36dd6 VXORPD %XMM0,%XMM0,%XMM0 |
0x36dda XOR %R10D,%R10D |
0x36ddd VPCMPEQD %YMM1,%YMM1,%YMM1 |
0x36de1 VPXOR %XMM2,%XMM2,%XMM2 |
0x36de5 VPXOR %XMM3,%XMM3,%XMM3 |
0x36de9 JMP 36e39 |
0x36deb NOPL (%RAX,%RAX,1) |
(840) 0x36df0 VPTESTMQ %YMM7,%YMM7,%K1 |
(840) 0x36df6 KMOVQ %K1,%K2 |
(840) 0x36dfb VSCATTERQPD %YMM0,(%R11,%YMM6,8){%K2} |
(840) 0x36e02 VPTESTMQ %YMM5,%YMM5,%K2 |
(840) 0x36e08 KMOVQ %K2,%K3 |
(840) 0x36e0d VSCATTERQPD %YMM0,(%R11,%YMM4,8){%K3} |
(840) 0x36e14 VPSUBQ %YMM1,%YMM2,%YMM4 |
(840) 0x36e18 VMOVDQA64 %YMM2,%YMM4{%K1} |
(840) 0x36e1e VPSUBQ %YMM1,%YMM3,%YMM5 |
(840) 0x36e22 VMOVDQA64 %YMM3,%YMM5{%K2} |
(840) 0x36e28 ADD $0x8,%R10 |
(840) 0x36e2c VMOVDQA %YMM4,%YMM2 |
(840) 0x36e30 VMOVDQA %YMM5,%YMM3 |
(840) 0x36e34 CMP %R8,%R10 |
(840) 0x36e37 JA 36e7d |
(840) 0x36e39 VMOVDQU 0x20(%R9,%R10,8),%YMM4 |
(840) 0x36e40 KXNORW %K0,%K0,%K1 |
(840) 0x36e44 VPXOR %XMM5,%XMM5,%XMM5 |
(840) 0x36e48 VPGATHERQQ (%RCX,%YMM4,8),%YMM5{%K1} |
(840) 0x36e4f VMOVDQU (%R9,%R10,8),%YMM6 |
(840) 0x36e55 KXNORW %K0,%K0,%K1 |
(840) 0x36e59 VPXOR %XMM7,%XMM7,%XMM7 |
(840) 0x36e5d VPGATHERQQ (%RCX,%YMM6,8),%YMM7{%K1} |
(840) 0x36e64 VPOR %YMM5,%YMM7,%YMM8 |
(840) 0x36e68 VPTEST %YMM8,%YMM8 |
(840) 0x36e6d JE 36df0 |
(840) 0x36e6f MOV (%R13),%R11 |
(840) 0x36e73 JMP 36df0 |
0x36e78 XOR %R9D,%R9D |
0x36e7b JMP 36ea6 |
0x36e7d VPADDQ %YMM5,%YMM4,%YMM0 |
0x36e81 VEXTRACTI128 $0x1,%YMM0,%XMM1 |
0x36e87 VPADDQ %XMM1,%XMM0,%XMM0 |
0x36e8b VPSHUFD $-0x12,%XMM0,%XMM1 |
0x36e90 VPADDQ %XMM1,%XMM0,%XMM0 |
0x36e94 VMOVQ %XMM0,%R9 |
0x36e99 CMP %RDX,%RDI |
0x36e9c MOV 0x10(%RBP),%R8 |
0x36ea0 JNE 37273 |
0x36ea6 MOV %R12,-0x60(%RBP) |
0x36eaa MOV 0x30(%RBP),%R8 |
0x36eae MOV 0x18(%RBP),%R14 |
0x36eb2 MOV %R9,-0x30(%RBP) |
0x36eb6 MOV -0x50(%RBP),%R12 |
0x36eba MOV -0x58(%RBP),%RAX |
0x36ebe MOV %RAX,%RSI |
0x36ec1 SUB %R12,%RSI |
0x36ec4 JLE 3701e |
0x36eca MOV %RSI,%RCX |
0x36ecd AND $-0x10,%RCX |
0x36ed1 JE 372a8 |
0x36ed7 LEA -0x1(%RCX),%RDI |
0x36edb LEA (%R14,%R12,8),%R9 |
0x36edf VPXOR %XMM0,%XMM0,%XMM0 |
0x36ee3 XOR %R10D,%R10D |
0x36ee6 VPCMPEQD %YMM1,%YMM1,%YMM1 |
0x36eea VPXOR %XMM5,%XMM5,%XMM5 |
0x36eee VPXOR %XMM4,%XMM4,%XMM4 |
0x36ef2 VPXOR %XMM2,%XMM2,%XMM2 |
0x36ef6 VPXOR %XMM3,%XMM3,%XMM3 |
0x36efa JMP 36fa1 |
0x36eff NOP |
(838) 0x36f00 VPTESTMQ %YMM13,%YMM13,%K1 |
(838) 0x36f06 VPBROADCASTQ %RDX,%YMM13 |
(838) 0x36f0c VPADDQ %YMM13,%YMM12,%YMM12 |
(838) 0x36f11 KMOVQ %K1,%K2 |
(838) 0x36f16 VSCATTERQPD %YMM0,(%R11,%YMM12,8){%K2} |
(838) 0x36f1d VPSUBQ %YMM1,%YMM5,%YMM12 |
(838) 0x36f21 VMOVDQA64 %YMM5,%YMM12{%K1} |
(838) 0x36f27 VPTESTMQ %YMM7,%YMM7,%K1 |
(838) 0x36f2d VPTESTMQ %YMM11,%YMM11,%K2 |
(838) 0x36f33 VPTESTMQ %YMM10,%YMM10,%K3 |
(838) 0x36f39 VPADDQ %YMM13,%YMM8,%YMM5 |
(838) 0x36f3e VPSUBQ %YMM1,%YMM4,%YMM7 |
(838) 0x36f42 VMOVDQA64 %YMM4,%YMM7{%K3} |
(838) 0x36f48 VSCATTERQPD %YMM0,(%R11,%YMM5,8){%K3} |
(838) 0x36f4f VPADDQ %YMM13,%YMM9,%YMM4 |
(838) 0x36f54 KMOVQ %K2,%K3 |
(838) 0x36f59 VSCATTERQPD %YMM0,(%R11,%YMM4,8){%K3} |
(838) 0x36f60 VPADDQ %YMM6,%YMM13,%YMM4 |
(838) 0x36f64 KMOVQ %K1,%K3 |
(838) 0x36f69 VSCATTERQPD %YMM0,(%R11,%YMM4,8){%K3} |
(838) 0x36f70 VPSUBQ %YMM1,%YMM2,%YMM6 |
(838) 0x36f74 VMOVDQA64 %YMM2,%YMM6{%K2} |
(838) 0x36f7a VPSUBQ %YMM1,%YMM3,%YMM8 |
(838) 0x36f7e VMOVDQA64 %YMM3,%YMM8{%K1} |
(838) 0x36f84 ADD $0x10,%R10 |
(838) 0x36f88 VMOVDQA %YMM12,%YMM5 |
(838) 0x36f8c VMOVDQA %YMM7,%YMM4 |
(838) 0x36f90 VMOVDQA %YMM6,%YMM2 |
(838) 0x36f94 VMOVDQA %YMM8,%YMM3 |
(838) 0x36f98 CMP %RDI,%R10 |
(838) 0x36f9b JA 37022 |
(838) 0x36fa1 VMOVDQU 0x60(%R9,%R10,8),%YMM6 |
(838) 0x36fa8 VPXOR %XMM7,%XMM7,%XMM7 |
(838) 0x36fac KXNORW %K0,%K0,%K1 |
(838) 0x36fb0 VPGATHERQQ (%RBX,%YMM6,8),%YMM7{%K1} |
(838) 0x36fb7 VMOVDQU 0x20(%R9,%R10,8),%YMM8 |
(838) 0x36fbe VPXOR %XMM10,%XMM10,%XMM10 |
(838) 0x36fc3 KXNORW %K0,%K0,%K1 |
(838) 0x36fc7 VPGATHERQQ (%RBX,%YMM8,8),%YMM10{%K1} |
(838) 0x36fce VMOVDQU 0x40(%R9,%R10,8),%YMM9 |
(838) 0x36fd5 VPXOR %XMM11,%XMM11,%XMM11 |
(838) 0x36fda KXNORW %K0,%K0,%K1 |
(838) 0x36fde VPGATHERQQ (%RBX,%YMM9,8),%YMM11{%K1} |
(838) 0x36fe5 VMOVDQU (%R9,%R10,8),%YMM12 |
(838) 0x36feb VPXOR %XMM13,%XMM13,%XMM13 |
(838) 0x36ff0 KXNORW %K0,%K0,%K1 |
(838) 0x36ff4 VPGATHERQQ (%RBX,%YMM12,8),%YMM13{%K1} |
(838) 0x36ffb VPOR %YMM11,%YMM13,%YMM14 |
(838) 0x37000 VPTERNLOGQ $-0x2,%YMM7,%YMM10,%YMM14 |
(838) 0x37007 VPTEST %YMM14,%YMM14 |
(838) 0x3700c JE 36f00 |
(838) 0x37012 MOV (%R13),%R11 |
(838) 0x37016 MOV (%R15),%RDX |
(838) 0x37019 JMP 36f00 |
0x3701e XOR %EDI,%EDI |
0x37020 JMP 3704f |
0x37022 VPADDQ %YMM6,%YMM12,%YMM0 |
0x37026 VPADDQ %YMM7,%YMM8,%YMM1 |
0x3702a VPADDQ %YMM1,%YMM0,%YMM0 |
0x3702e VEXTRACTI128 $0x1,%YMM0,%XMM1 |
0x37034 VPADDQ %XMM1,%XMM0,%XMM0 |
0x37038 VPSHUFD $-0x12,%XMM0,%XMM1 |
0x3703d VPADDQ %XMM1,%XMM0,%XMM0 |
0x37041 VMOVQ %XMM0,%RDI |
0x37046 CMP %RCX,%RSI |
0x37049 JNE 372ac |
0x3704f MOV %RDI,-0x38(%RBP) |
0x37053 LEA -0x30(%RBP),%RDI |
0x37057 LEA -0x38(%RBP),%RDX |
0x3705b MOV 0x20(%RBP),%RSI |
0x3705f MOV 0x28(%RBP),%RCX |
0x37063 VZEROUPPER |
0x37066 CALL e6b0 <hypre_prefix_sum_pair@plt> |
0x3706b MOV -0x40(%RBP),%RAX |
0x3706f MOV -0x48(%RBP),%RCX |
0x37073 MOV %RCX,%RDI |
0x37076 SUB %RAX,%RDI |
0x37079 JLE 370d5 |
0x3707b MOV 0x38(%RBP),%RDX |
0x3707f MOV -0x60(%RBP),%RSI |
0x37083 MOV (%RSI),%RSI |
0x37086 CMP $0x4,%RDI |
0x3708a JAE 37144 |
0x37090 MOV %RDI,%R8 |
0x37093 AND $-0x4,%R8 |
0x37097 CMP %RDI,%R8 |
0x3709a MOV 0x10(%RBP),%R10 |
0x3709e JAE 370d5 |
0x370a0 ADD %R8,%RAX |
0x370a3 JMP 370b8 |
0x370a5 NOPW %CS:(%RAX,%RAX,1) |
(835) 0x370b0 INC %RAX |
(835) 0x370b3 CMP %RAX,%RCX |
(835) 0x370b6 JE 370d5 |
(835) 0x370b8 MOV (%R10,%RAX,8),%RDI |
(835) 0x370bc CMPQ $0,(%RSI,%RDI,8) |
(835) 0x370c1 JNE 370b0 |
(835) 0x370c3 MOV -0x30(%RBP),%R8 |
(835) 0x370c7 LEA 0x1(%R8),%R9 |
(835) 0x370cb MOV %R9,-0x30(%RBP) |
(835) 0x370cf MOV %RDI,(%RDX,%R8,8) |
(835) 0x370d3 JMP 370b0 |
0x370d5 MOV -0x50(%RBP),%RAX |
0x370d9 MOV -0x58(%RBP),%RCX |
0x370dd MOV %RCX,%RSI |
0x370e0 SUB %RAX,%RSI |
0x370e3 JLE 37135 |
0x370e5 MOV 0x40(%RBP),%RDX |
0x370e9 CMP $0x4,%RSI |
0x370ed JAE 371da |
0x370f3 MOV %RSI,%RDI |
0x370f6 AND $-0x4,%RDI |
0x370fa CMP %RSI,%RDI |
0x370fd JAE 37135 |
0x370ff ADD %RDI,%RAX |
0x37102 JMP 37118 |
0x37104 NOPW %CS:(%RAX,%RAX,1) |
(833) 0x37110 INC %RAX |
(833) 0x37113 CMP %RAX,%RCX |
(833) 0x37116 JE 37135 |
(833) 0x37118 MOV (%R14,%RAX,8),%RSI |
(833) 0x3711c CMPQ $0,(%RBX,%RSI,8) |
(833) 0x37121 JNE 37110 |
(833) 0x37123 MOV -0x38(%RBP),%RDI |
(833) 0x37127 LEA 0x1(%RDI),%R8 |
(833) 0x3712b MOV %R8,-0x38(%RBP) |
(833) 0x3712f MOV %RSI,(%RDX,%RDI,8) |
(833) 0x37133 JMP 37110 |
0x37135 ADD $0x38,%RSP |
0x37139 POP %RBX |
0x3713a POP %R12 |
0x3713c POP %R13 |
0x3713e POP %R14 |
0x37140 POP %R15 |
0x37142 POP %RBP |
0x37143 RET |
0x37144 MOV %RDI,%R8 |
0x37147 SHR $0x2,%R8 |
0x3714b MOV 0x10(%RBP),%R9 |
0x3714f LEA 0x18(%R9,%RAX,8),%R9 |
0x37154 JMP 3716d |
0x37156 NOPW %CS:(%RAX,%RAX,1) |
(836) 0x37160 ADD $0x20,%R9 |
(836) 0x37164 DEC %R8 |
(836) 0x37167 JE 37090 |
(836) 0x3716d MOV -0x18(%R9),%R10 |
(836) 0x37171 CMPQ $0,(%RSI,%R10,8) |
(836) 0x37176 JNE 37188 |
(836) 0x37178 MOV -0x30(%RBP),%R11 |
(836) 0x3717c LEA 0x1(%R11),%R15 |
(836) 0x37180 MOV %R15,-0x30(%RBP) |
(836) 0x37184 MOV %R10,(%RDX,%R11,8) |
(836) 0x37188 MOV -0x10(%R9),%R10 |
(836) 0x3718c CMPQ $0,(%RSI,%R10,8) |
(836) 0x37191 JNE 371a3 |
(836) 0x37193 MOV -0x30(%RBP),%R11 |
(836) 0x37197 LEA 0x1(%R11),%R15 |
(836) 0x3719b MOV %R15,-0x30(%RBP) |
(836) 0x3719f MOV %R10,(%RDX,%R11,8) |
(836) 0x371a3 MOV -0x8(%R9),%R10 |
(836) 0x371a7 CMPQ $0,(%RSI,%R10,8) |
(836) 0x371ac JNE 371be |
(836) 0x371ae MOV -0x30(%RBP),%R11 |
(836) 0x371b2 LEA 0x1(%R11),%R15 |
(836) 0x371b6 MOV %R15,-0x30(%RBP) |
(836) 0x371ba MOV %R10,(%RDX,%R11,8) |
(836) 0x371be MOV (%R9),%R10 |
(836) 0x371c1 CMPQ $0,(%RSI,%R10,8) |
(836) 0x371c6 JNE 37160 |
(836) 0x371c8 MOV -0x30(%RBP),%R11 |
(836) 0x371cc LEA 0x1(%R11),%R15 |
(836) 0x371d0 MOV %R15,-0x30(%RBP) |
(836) 0x371d4 MOV %R10,(%RDX,%R11,8) |
(836) 0x371d8 JMP 37160 |
0x371da MOV %RSI,%RDI |
0x371dd SHR $0x2,%RDI |
0x371e1 LEA 0x18(%R14,%RAX,8),%R8 |
0x371e6 JMP 371fd |
0x371e8 NOPL (%RAX,%RAX,1) |
(834) 0x371f0 ADD $0x20,%R8 |
(834) 0x371f4 DEC %RDI |
(834) 0x371f7 JE 370f3 |
(834) 0x371fd MOV -0x18(%R8),%R9 |
(834) 0x37201 CMPQ $0,(%RBX,%R9,8) |
(834) 0x37206 JNE 37218 |
(834) 0x37208 MOV -0x38(%RBP),%R10 |
(834) 0x3720c LEA 0x1(%R10),%R11 |
(834) 0x37210 MOV %R11,-0x38(%RBP) |
(834) 0x37214 MOV %R9,(%RDX,%R10,8) |
(834) 0x37218 MOV -0x10(%R8),%R9 |
(834) 0x3721c CMPQ $0,(%RBX,%R9,8) |
(834) 0x37221 JNE 37233 |
(834) 0x37223 MOV -0x38(%RBP),%R10 |
(834) 0x37227 LEA 0x1(%R10),%R11 |
(834) 0x3722b MOV %R11,-0x38(%RBP) |
(834) 0x3722f MOV %R9,(%RDX,%R10,8) |
(834) 0x37233 MOV -0x8(%R8),%R9 |
(834) 0x37237 CMPQ $0,(%RBX,%R9,8) |
(834) 0x3723c JNE 3724e |
(834) 0x3723e MOV -0x38(%RBP),%R10 |
(834) 0x37242 LEA 0x1(%R10),%R11 |
(834) 0x37246 MOV %R11,-0x38(%RBP) |
(834) 0x3724a MOV %R9,(%RDX,%R10,8) |
(834) 0x3724e MOV (%R8),%R9 |
(834) 0x37251 CMPQ $0,(%RBX,%R9,8) |
(834) 0x37256 JNE 371f0 |
(834) 0x37258 MOV -0x38(%RBP),%R10 |
(834) 0x3725c LEA 0x1(%R10),%R11 |
(834) 0x37260 MOV %R11,-0x38(%RBP) |
(834) 0x37264 MOV %R9,(%RDX,%R10,8) |
(834) 0x37268 JMP 371f0 |
0x3726a XOR %EDX,%EDX |
0x3726c XOR %R9D,%R9D |
0x3726f MOV 0x10(%RBP),%R8 |
0x37273 ADD %RSI,%RDX |
0x37276 JMP 37298 |
0x37278 NOPL (%RAX,%RAX,1) |
(839) 0x37280 MOV (%R13),%RDI |
(839) 0x37284 MOVQ $0,(%RDI,%RSI,8) |
(839) 0x3728c INC %RDX |
(839) 0x3728f CMP %RDX,%RAX |
(839) 0x37292 JE 36ea6 |
(839) 0x37298 MOV (%R8,%RDX,8),%RSI |
(839) 0x3729c CMPQ $0,(%RCX,%RSI,8) |
(839) 0x372a1 JNE 37280 |
(839) 0x372a3 INC %R9 |
(839) 0x372a6 JMP 3728c |
0x372a8 XOR %EDI,%EDI |
0x372aa XOR %ECX,%ECX |
0x372ac ADD %R12,%RCX |
0x372af JMP 372db |
0x372b1 NOPW %CS:(%RAX,%RAX,1) |
(837) 0x372c0 MOV (%R13),%RSI |
(837) 0x372c4 ADD (%R15),%RDX |
(837) 0x372c7 MOVQ $0,(%RSI,%RDX,8) |
(837) 0x372cf INC %RCX |
(837) 0x372d2 CMP %RCX,%RAX |
(837) 0x372d5 JE 3704f |
(837) 0x372db MOV (%R14,%RCX,8),%RDX |
(837) 0x372df CMPQ $0,(%RBX,%RDX,8) |
(837) 0x372e4 JNE 372c0 |
(837) 0x372e6 INC %RDI |
(837) 0x372e9 JMP 372cf |
0x372eb NOPL (%RAX,%RAX,1) |
Path / |
Source file and lines | par_coarsen.c:2516-2576 |
Module | libparcsr_ls.so |
nb instructions | 158 |
nb uops | 162 |
loop length | 601 |
used x86 registers | 15 |
used mmx registers | 0 |
used xmm registers | 6 |
used ymm registers | 8 |
used zmm registers | 0 |
nb stack references | 14 |
micro-operation queue | 27.00 cycles |
front end | 27.00 cycles |
P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 10.20 | 10.27 | 11.00 | 11.00 | 6.00 | 10.23 | 10.10 | 6.00 | 6.00 | 6.00 | 10.20 | 11.00 |
cycles | 10.20 | 10.27 | 11.00 | 11.00 | 6.00 | 10.23 | 10.10 | 6.00 | 6.00 | 6.00 | 10.20 | 11.00 |
Cycles executing div or sqrt instructions | NA |
FE+BE cycles | 26.45-26.50 |
Stall cycles | 0.00 |
Front-end | 27.00 |
Dispatch | 11.00 |
Overall L1 | 27.00 |
all | 37% |
load | 0% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 66% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 35% |
all | 100% |
load | NA (no load vectorizable/vectorized instructions) |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | NA (no add-sub vectorizable/vectorized instructions) |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 100% |
all | 38% |
load | 0% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 66% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 37% |
all | 19% |
load | 12% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 29% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 17% |
all | 25% |
load | NA (no load vectorizable/vectorized instructions) |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | NA (no add-sub vectorizable/vectorized instructions) |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 25% |
all | 19% |
load | 12% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 29% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 17% |
Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
PUSH %RBP | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
PUSH %R15 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R14 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R13 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R12 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %RBX | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
SUB $0x38,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %R9,%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %R8,%RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %RCX,%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %RDX,%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV 0x28(%RBP),%R14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x20(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RAX),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA -0x40(%RBP),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x48(%RBP),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
CALL e050 <hypre_GetSimpleThreadPartition@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
MOV (%R14),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA -0x50(%RBP),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x58(%RBP),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
CALL e050 <hypre_GetSimpleThreadPartition@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
MOV -0x40(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x48(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RAX,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %RSI,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE 36e78 <hypre_BoomerAMGCoarsenPMIS.extracted+0x118> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV (%R12),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RDI,%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x8,%RDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
JE 3726a <hypre_BoomerAMGCoarsenPMIS.extracted+0x50a> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
LEA -0x1(%RDX),%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV 0x10(%RBP),%R9 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA (%R9,%RSI,8),%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VXORPD %XMM0,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %R10D,%R10D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPCMPEQD %YMM1,%YMM1,%YMM1 | 1 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
VPXOR %XMM2,%XMM2,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPXOR %XMM3,%XMM3,%XMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP 36e39 <hypre_BoomerAMGCoarsenPMIS.extracted+0xd9> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %R9D,%R9D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP 36ea6 <hypre_BoomerAMGCoarsenPMIS.extracted+0x146> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
VPADDQ %YMM5,%YMM4,%YMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VEXTRACTI128 $0x1,%YMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
VPADDQ %XMM1,%XMM0,%XMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VPSHUFD $-0x12,%XMM0,%XMM1 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
VPADDQ %XMM1,%XMM0,%XMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VMOVQ %XMM0,%R9 | 1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 1 |
CMP %RDX,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV 0x10(%RBP),%R8 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JNE 37273 <hypre_BoomerAMGCoarsenPMIS.extracted+0x513> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %R12,-0x60(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV 0x30(%RBP),%R8 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x18(%RBP),%R14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %R9,-0x30(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV -0x50(%RBP),%R12 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x58(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RAX,%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %R12,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE 3701e <hypre_BoomerAMGCoarsenPMIS.extracted+0x2be> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RSI,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x10,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
JE 372a8 <hypre_BoomerAMGCoarsenPMIS.extracted+0x548> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
LEA -0x1(%RCX),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA (%R14,%R12,8),%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPXOR %XMM0,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %R10D,%R10D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPCMPEQD %YMM1,%YMM1,%YMM1 | 1 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
VPXOR %XMM5,%XMM5,%XMM5 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPXOR %XMM4,%XMM4,%XMM4 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPXOR %XMM2,%XMM2,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPXOR %XMM3,%XMM3,%XMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP 36fa1 <hypre_BoomerAMGCoarsenPMIS.extracted+0x241> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2.08 |
NOP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %EDI,%EDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP 3704f <hypre_BoomerAMGCoarsenPMIS.extracted+0x2ef> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
VPADDQ %YMM6,%YMM12,%YMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VPADDQ %YMM7,%YMM8,%YMM1 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VPADDQ %YMM1,%YMM0,%YMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VEXTRACTI128 $0x1,%YMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
VPADDQ %XMM1,%XMM0,%XMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VPSHUFD $-0x12,%XMM0,%XMM1 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
VPADDQ %XMM1,%XMM0,%XMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VMOVQ %XMM0,%RDI | 1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 1 |
CMP %RCX,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JNE 372ac <hypre_BoomerAMGCoarsenPMIS.extracted+0x54c> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RDI,-0x38(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
LEA -0x30(%RBP),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x38(%RBP),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV 0x20(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x28(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VZEROUPPER | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 |
CALL e6b0 <hypre_prefix_sum_pair@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
MOV -0x40(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x48(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RCX,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %RAX,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE 370d5 <hypre_BoomerAMGCoarsenPMIS.extracted+0x375> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV 0x38(%RBP),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x60(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RSI),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
CMP $0x4,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE 37144 <hypre_BoomerAMGCoarsenPMIS.extracted+0x3e4> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RDI,%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x4,%R8 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
CMP %RDI,%R8 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV 0x10(%RBP),%R10 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JAE 370d5 <hypre_BoomerAMGCoarsenPMIS.extracted+0x375> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
ADD %R8,%RAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JMP 370b8 <hypre_BoomerAMGCoarsenPMIS.extracted+0x358> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV -0x50(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x58(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RCX,%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %RAX,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE 37135 <hypre_BoomerAMGCoarsenPMIS.extracted+0x3d5> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV 0x40(%RBP),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
CMP $0x4,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE 371da <hypre_BoomerAMGCoarsenPMIS.extracted+0x47a> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RSI,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x4,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
CMP %RSI,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE 37135 <hypre_BoomerAMGCoarsenPMIS.extracted+0x3d5> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
ADD %RDI,%RAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JMP 37118 <hypre_BoomerAMGCoarsenPMIS.extracted+0x3b8> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
ADD $0x38,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
POP %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
RET | 1 | 0.50 | 0 | 0.33 | 0.33 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0.33 | 0 | 2.13 |
MOV %RDI,%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SHR $0x2,%R8 | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
MOV 0x10(%RBP),%R9 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA 0x18(%R9,%RAX,8),%R9 | 1 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
JMP 3716d <hypre_BoomerAMGCoarsenPMIS.extracted+0x40d> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV %RSI,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SHR $0x2,%RDI | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
LEA 0x18(%R14,%RAX,8),%R8 | 1 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
JMP 371fd <hypre_BoomerAMGCoarsenPMIS.extracted+0x49d> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %EDX,%EDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %R9D,%R9D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV 0x10(%RBP),%R8 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
ADD %RSI,%RDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JMP 37298 <hypre_BoomerAMGCoarsenPMIS.extracted+0x538> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %EDI,%EDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %ECX,%ECX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
ADD %R12,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JMP 372db <hypre_BoomerAMGCoarsenPMIS.extracted+0x57b> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
Source file and lines | par_coarsen.c:2516-2576 |
Module | libparcsr_ls.so |
nb instructions | 158 |
nb uops | 162 |
loop length | 601 |
used x86 registers | 15 |
used mmx registers | 0 |
used xmm registers | 6 |
used ymm registers | 8 |
used zmm registers | 0 |
nb stack references | 14 |
micro-operation queue | 27.00 cycles |
front end | 27.00 cycles |
P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 10.20 | 10.27 | 11.00 | 11.00 | 6.00 | 10.23 | 10.10 | 6.00 | 6.00 | 6.00 | 10.20 | 11.00 |
cycles | 10.20 | 10.27 | 11.00 | 11.00 | 6.00 | 10.23 | 10.10 | 6.00 | 6.00 | 6.00 | 10.20 | 11.00 |
Cycles executing div or sqrt instructions | NA |
FE+BE cycles | 26.45-26.50 |
Stall cycles | 0.00 |
Front-end | 27.00 |
Dispatch | 11.00 |
Overall L1 | 27.00 |
all | 37% |
load | 0% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 66% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 35% |
all | 100% |
load | NA (no load vectorizable/vectorized instructions) |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | NA (no add-sub vectorizable/vectorized instructions) |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 100% |
all | 38% |
load | 0% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 66% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 37% |
all | 19% |
load | 12% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 29% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 17% |
all | 25% |
load | NA (no load vectorizable/vectorized instructions) |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | NA (no add-sub vectorizable/vectorized instructions) |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 25% |
all | 19% |
load | 12% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 29% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 17% |
Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
PUSH %RBP | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
PUSH %R15 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R14 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R13 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R12 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %RBX | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
SUB $0x38,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %R9,%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %R8,%RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %RCX,%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %RDX,%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV 0x28(%RBP),%R14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x20(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RAX),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA -0x40(%RBP),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x48(%RBP),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
CALL e050 <hypre_GetSimpleThreadPartition@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
MOV (%R14),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA -0x50(%RBP),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x58(%RBP),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
CALL e050 <hypre_GetSimpleThreadPartition@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
MOV -0x40(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x48(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RAX,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %RSI,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE 36e78 <hypre_BoomerAMGCoarsenPMIS.extracted+0x118> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV (%R12),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RDI,%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x8,%RDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
JE 3726a <hypre_BoomerAMGCoarsenPMIS.extracted+0x50a> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
LEA -0x1(%RDX),%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV 0x10(%RBP),%R9 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA (%R9,%RSI,8),%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VXORPD %XMM0,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %R10D,%R10D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPCMPEQD %YMM1,%YMM1,%YMM1 | 1 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
VPXOR %XMM2,%XMM2,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPXOR %XMM3,%XMM3,%XMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP 36e39 <hypre_BoomerAMGCoarsenPMIS.extracted+0xd9> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %R9D,%R9D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP 36ea6 <hypre_BoomerAMGCoarsenPMIS.extracted+0x146> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
VPADDQ %YMM5,%YMM4,%YMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VEXTRACTI128 $0x1,%YMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
VPADDQ %XMM1,%XMM0,%XMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VPSHUFD $-0x12,%XMM0,%XMM1 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
VPADDQ %XMM1,%XMM0,%XMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VMOVQ %XMM0,%R9 | 1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 1 |
CMP %RDX,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV 0x10(%RBP),%R8 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JNE 37273 <hypre_BoomerAMGCoarsenPMIS.extracted+0x513> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %R12,-0x60(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV 0x30(%RBP),%R8 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x18(%RBP),%R14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %R9,-0x30(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV -0x50(%RBP),%R12 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x58(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RAX,%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %R12,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE 3701e <hypre_BoomerAMGCoarsenPMIS.extracted+0x2be> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RSI,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x10,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
JE 372a8 <hypre_BoomerAMGCoarsenPMIS.extracted+0x548> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
LEA -0x1(%RCX),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA (%R14,%R12,8),%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPXOR %XMM0,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %R10D,%R10D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPCMPEQD %YMM1,%YMM1,%YMM1 | 1 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
VPXOR %XMM5,%XMM5,%XMM5 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPXOR %XMM4,%XMM4,%XMM4 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPXOR %XMM2,%XMM2,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VPXOR %XMM3,%XMM3,%XMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP 36fa1 <hypre_BoomerAMGCoarsenPMIS.extracted+0x241> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2.08 |
NOP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %EDI,%EDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP 3704f <hypre_BoomerAMGCoarsenPMIS.extracted+0x2ef> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
VPADDQ %YMM6,%YMM12,%YMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VPADDQ %YMM7,%YMM8,%YMM1 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VPADDQ %YMM1,%YMM0,%YMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VEXTRACTI128 $0x1,%YMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
VPADDQ %XMM1,%XMM0,%XMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VPSHUFD $-0x12,%XMM0,%XMM1 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
VPADDQ %XMM1,%XMM0,%XMM0 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VMOVQ %XMM0,%RDI | 1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 1 |
CMP %RCX,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JNE 372ac <hypre_BoomerAMGCoarsenPMIS.extracted+0x54c> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RDI,-0x38(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
LEA -0x30(%RBP),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x38(%RBP),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV 0x20(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x28(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VZEROUPPER | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 |
CALL e6b0 <hypre_prefix_sum_pair@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
MOV -0x40(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x48(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RCX,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %RAX,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE 370d5 <hypre_BoomerAMGCoarsenPMIS.extracted+0x375> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV 0x38(%RBP),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x60(%RBP),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RSI),%RSI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
CMP $0x4,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE 37144 <hypre_BoomerAMGCoarsenPMIS.extracted+0x3e4> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RDI,%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x4,%R8 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
CMP %RDI,%R8 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV 0x10(%RBP),%R10 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
JAE 370d5 <hypre_BoomerAMGCoarsenPMIS.extracted+0x375> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
ADD %R8,%RAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JMP 370b8 <hypre_BoomerAMGCoarsenPMIS.extracted+0x358> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV -0x50(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x58(%RBP),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV %RCX,%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SUB %RAX,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JLE 37135 <hypre_BoomerAMGCoarsenPMIS.extracted+0x3d5> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV 0x40(%RBP),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
CMP $0x4,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE 371da <hypre_BoomerAMGCoarsenPMIS.extracted+0x47a> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RSI,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $-0x4,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
CMP %RSI,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE 37135 <hypre_BoomerAMGCoarsenPMIS.extracted+0x3d5> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
ADD %RDI,%RAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JMP 37118 <hypre_BoomerAMGCoarsenPMIS.extracted+0x3b8> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
ADD $0x38,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
POP %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
RET | 1 | 0.50 | 0 | 0.33 | 0.33 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0.33 | 0 | 2.13 |
MOV %RDI,%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SHR $0x2,%R8 | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
MOV 0x10(%RBP),%R9 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA 0x18(%R9,%RAX,8),%R9 | 1 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
JMP 3716d <hypre_BoomerAMGCoarsenPMIS.extracted+0x40d> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV %RSI,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
SHR $0x2,%RDI | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
LEA 0x18(%R14,%RAX,8),%R8 | 1 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
JMP 371fd <hypre_BoomerAMGCoarsenPMIS.extracted+0x49d> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %EDX,%EDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %R9D,%R9D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV 0x10(%RBP),%R8 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
ADD %RSI,%RDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JMP 37298 <hypre_BoomerAMGCoarsenPMIS.extracted+0x538> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %EDI,%EDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %ECX,%ECX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
ADD %R12,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JMP 372db <hypre_BoomerAMGCoarsenPMIS.extracted+0x57b> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
Name | Coverage (%) | Time (s) |
---|---|---|
▼hypre_BoomerAMGCoarsenPMIS.extracted– | 0.16 | 0.03 |
○Loop 836 - par_coarsen.c:2562-2567 - libparcsr_ls.so | 0.08 | 0.01 |
○Loop 840 - par_coarsen.c:2528-2540 - libparcsr_ls.so | 0.08 | 0.02 |
○Loop 834 - par_coarsen.c:2571-2576 - libparcsr_ls.so | 0 | 0 |
○Loop 835 - par_coarsen.c:2562-2567 - libparcsr_ls.so | 0 | 0 |
○Loop 837 - par_coarsen.c:2544-2556 - libparcsr_ls.so | 0 | 0 |
○Loop 838 - par_coarsen.c:2528-2556 - libparcsr_ls.so | 0 | 0 |
○Loop 839 - par_coarsen.c:2528-2540 - libparcsr_ls.so | 0 | 0 |
○Loop 833 - par_coarsen.c:2571-2576 - libparcsr_ls.so | 0 | 0 |