Function: clover_unpack_message_left_.DIR.OMP.PARALLEL.LOOP.2.split99 | Module: exec | Source: pack_kernel.f90:108-116 | Coverage: 0.03% |
---|
Function: clover_unpack_message_left_.DIR.OMP.PARALLEL.LOOP.2.split99 | Module: exec | Source: pack_kernel.f90:108-116 | Coverage: 0.03% |
---|
/scratch_na/users/xoserete/qaas_runs/171-214-9740/intel/CloverLeafFC/build/CloverLeafFC/CloverLeaf_ref/kernels/pack_kernel.f90: 108 - 116 |
-------------------------------------------------------------------------------- |
108: !$OMP PARALLEL DO PRIVATE(index) |
109: DO k=y_min-depth,y_max+y_inc+depth |
110: !$OMP SIMD |
111: DO j=1,depth |
112: index= buffer_offset + j+(k+depth-1)*depth |
113: field(x_min-j,k)=left_rcv_buffer(index) |
114: ENDDO |
115: ENDDO |
116: !$OMP END PARALLEL DO |
0x448ec0 PUSH %RBP |
0x448ec1 MOV %RSP,%RBP |
0x448ec4 PUSH %R15 |
0x448ec6 PUSH %R14 |
0x448ec8 PUSH %R13 |
0x448eca PUSH %R12 |
0x448ecc PUSH %RBX |
0x448ecd SUB $0x48,%RSP |
0x448ed1 MOV %R8,-0x58(%RBP) |
0x448ed5 MOV %RCX,-0x70(%RBP) |
0x448ed9 MOV 0x28(%RBP),%EAX |
0x448edc MOVL $0,-0x4c(%RBP) |
0x448ee3 TEST %EAX,%EAX |
0x448ee5 JS 448f46 |
0x448ee7 MOV %RDX,%RBX |
0x448eea MOV (%RDI),%ESI |
0x448eec MOVL $0,-0x34(%RBP) |
0x448ef3 MOV %EAX,-0x30(%RBP) |
0x448ef6 MOVL $0x1,-0x48(%RBP) |
0x448efd SUB $0x8,%RSP |
0x448f01 LEA -0x48(%RBP),%RAX |
0x448f05 LEA -0x4c(%RBP),%RCX |
0x448f09 LEA -0x34(%RBP),%R8 |
0x448f0d LEA -0x30(%RBP),%R9 |
0x448f11 MOV $0x74e530,%EDI |
0x448f16 MOV %ESI,-0x44(%RBP) |
0x448f19 MOV $0x22,%EDX |
0x448f1e PUSH $0x1 |
0x448f20 PUSH $0x1 |
0x448f22 PUSH %RAX |
0x448f23 CALL 4044c0 <__kmpc_for_static_init_4@plt> |
0x448f28 ADD $0x20,%RSP |
0x448f2c MOV -0x34(%RBP),%EAX |
0x448f2f MOV -0x30(%RBP),%EDX |
0x448f32 SUB %EAX,%EDX |
0x448f34 JAE 448f80 |
0x448f36 MOV $0x74e550,%EDI |
0x448f3b MOV -0x44(%RBP),%ESI |
0x448f3e VZEROUPPER |
0x448f41 CALL 4040b0 <__kmpc_for_static_fini@plt> |
0x448f46 ADD $0x48,%RSP |
0x448f4a POP %RBX |
0x448f4b POP %R12 |
0x448f4d POP %R13 |
0x448f4f POP %R14 |
0x448f51 POP %R15 |
0x448f53 POP %RBP |
0x448f54 RET |
0x448f55 NOPW %CS:(%RAX,%RAX,1) |
0x448f64 NOPW %CS:(%RAX,%RAX,1) |
0x448f73 NOPW %CS:(%RAX,%RAX,1) |
0x448f80 MOV %RAX,%RCX |
0x448f83 MOV -0x58(%RBP),%RAX |
0x448f87 MOV (%RAX),%ESI |
0x448f89 LEA -0x1(%RCX,%RBX,1),%EDI |
0x448f8d XOR %R8D,%R8D |
0x448f90 ADD %EBX,%ECX |
0x448f92 MOV %RCX,-0x68(%RBP) |
0x448f96 VMOVDQA64 0xc10a0(%RIP),%ZMM0 |
0x448fa0 VMOVDQA64 0xc10d6(%RIP),%ZMM1 |
0x448faa VMOVDQA64 0xc108c(%RIP),%ZMM2 |
0x448fb4 MOV %EDX,-0x2c(%RBP) |
0x448fb7 JMP 448fd8 |
0x448fb9 NOPL (%RAX) |
(438) 0x448fc0 MOV %ESI,%R10D |
(438) 0x448fc3 LEA 0x1(%R8),%EAX |
(438) 0x448fc7 INC %EDI |
(438) 0x448fc9 MOV %R10D,%ESI |
(438) 0x448fcc CMP %EDX,%R8D |
(438) 0x448fcf MOV %EAX,%R8D |
(438) 0x448fd2 JE 448f36 |
(438) 0x448fd8 TEST %ESI,%ESI |
(438) 0x448fda JLE 448fc0 |
(438) 0x448fdc MOV -0x68(%RBP),%RAX |
(438) 0x448fe0 LEA (%RAX,%R8,1),%R11D |
(438) 0x448fe4 MOV -0x70(%RBP),%RAX |
(438) 0x448fe8 MOVSXD (%RAX),%RCX |
(438) 0x448feb MOV -0x58(%RBP),%RAX |
(438) 0x448fef MOV (%RAX),%R10D |
(438) 0x448ff2 MOV 0x30ab9f(%RIP),%R9 |
(438) 0x448ff9 MOV 0x30abd0(%RIP),%R13 |
(438) 0x449000 MOV 0x10(%RBP),%RBX |
(438) 0x449004 MOV (%RBX),%R15 |
(438) 0x449007 MOV 0x38(%RBX),%RAX |
(438) 0x44900b MOV %RAX,-0x40(%RBP) |
(438) 0x44900f MOV 0x18(%RBP),%RDX |
(438) 0x449013 MOVSXD (%RDX),%RAX |
(438) 0x449016 MOV 0x50(%RBX),%RDX |
(438) 0x44901a MOV %ESI,%EBX |
(438) 0x44901c MOV %RBX,%R12 |
(438) 0x44901f MOVSXD %R11D,%R14 |
(438) 0x449022 MOV $-0x8,%ESI |
(438) 0x449027 AND %RSI,%R12 |
(438) 0x44902a JE 449100 |
(438) 0x449030 LEA (%R10,%RDI,1),%ESI |
(438) 0x449034 IMUL %R10D,%ESI |
(438) 0x449038 MOVSXD %ESI,%RSI |
(438) 0x44903b ADD %RCX,%RSI |
(438) 0x44903e LEA -0x1(%R11,%R10,1),%R11D |
(438) 0x449043 MOV %R10,-0x60(%RBP) |
(438) 0x449047 IMUL %R10D,%R11D |
(438) 0x44904b MOVSXD %R11D,%R11 |
(438) 0x44904e ADD %RCX,%R11 |
(438) 0x449051 VPBROADCASTQ %R13,%ZMM5 |
(438) 0x449057 MOV -0x40(%RBP),%RCX |
(438) 0x44905b VPBROADCASTQ %RCX,%ZMM4 |
(438) 0x449061 MOV %RAX,-0x40(%RBP) |
(438) 0x449065 XOR %ECX,%ECX |
(438) 0x449067 NOPW (%RAX,%RAX,1) |
(439) 0x449070 LEA (%RSI,%RCX,1),%R13 |
(439) 0x449074 VPBROADCASTQ %R13,%ZMM6 |
(439) 0x44907a VPADDQ %ZMM2,%ZMM6,%ZMM6 |
(439) 0x449080 VPMULLQ %ZMM6,%ZMM5,%ZMM6 |
(439) 0x449086 VXORPD %XMM7,%XMM7,%XMM7 |
(439) 0x44908a KXNORW %K0,%K0,%K1 |
(439) 0x44908e VGATHERQPD (%R9,%ZMM6,1),%ZMM7{%K1} |
(439) 0x449095 LEA 0x1(%R14),%R13 |
(439) 0x449099 IMUL %RDX,%R13 |
(439) 0x44909d VPBROADCASTQ %RAX,%ZMM6 |
(439) 0x4490a3 VPADDQ %ZMM1,%ZMM6,%ZMM6 |
(439) 0x4490a9 VPMULLQ %ZMM6,%ZMM4,%ZMM6 |
(439) 0x4490af LEA (%R15,%R13,1),%R10 |
(439) 0x4490b3 KXNORW %K0,%K0,%K1 |
(439) 0x4490b7 VSCATTERQPD %ZMM7,(%R10,%ZMM6,1){%K1} |
(439) 0x4490be ADD $0x8,%RCX |
(439) 0x4490c2 ADD $-0x8,%RAX |
(439) 0x4490c6 CMP %R12,%RCX |
(439) 0x4490c9 JB 449070 |
(438) 0x4490cb CMP %RBX,%R12 |
(438) 0x4490ce JNE 449180 |
(438) 0x4490d4 MOV -0x2c(%RBP),%EDX |
(438) 0x4490d7 MOV -0x60(%RBP),%R10 |
(438) 0x4490db JMP 448fc3 |
0x4490e0 NOPW %CS:(%RAX,%RAX,1) |
0x4490ef NOPW %CS:(%RAX,%RAX,1) |
0x4490fe XCHG %AX,%AX |
(438) 0x449100 VPBROADCASTQ %RBX,%ZMM8 |
(438) 0x449106 VPBROADCASTQ %R9,%ZMM6 |
(438) 0x44910c LEA -0x1(%R11,%R10,1),%ESI |
(438) 0x449111 IMUL %R10D,%ESI |
(438) 0x449115 MOVSXD %ESI,%R11 |
(438) 0x449118 ADD %RCX,%R11 |
(438) 0x44911b VPBROADCASTQ %R13,%ZMM5 |
(438) 0x449121 VPBROADCASTQ %R15,%ZMM7 |
(438) 0x449127 INC %R14 |
(438) 0x44912a IMUL %R14,%RDX |
(438) 0x44912e VPBROADCASTQ %RDX,%ZMM9 |
(438) 0x449134 VPBROADCASTQ %RAX,%ZMM10 |
(438) 0x44913a MOV -0x40(%RBP),%RAX |
(438) 0x44913e VPBROADCASTQ %RAX,%ZMM4 |
(438) 0x449144 XOR %R12D,%R12D |
(438) 0x449147 MOV -0x2c(%RBP),%EDX |
(438) 0x44914a JMP 4491a9 |
0x44914c NOPW %CS:(%RAX,%RAX,1) |
0x44915b NOPW %CS:(%RAX,%RAX,1) |
0x44916a NOPW %CS:(%RAX,%RAX,1) |
0x449179 NOPL (%RAX) |
(438) 0x449180 VPBROADCASTQ %R13,%ZMM9 |
(438) 0x449186 VPBROADCASTQ %R15,%ZMM7 |
(438) 0x44918c VPBROADCASTQ %R9,%ZMM6 |
(438) 0x449192 MOV -0x40(%RBP),%RAX |
(438) 0x449196 VPBROADCASTQ %RAX,%ZMM10 |
(438) 0x44919c VPBROADCASTQ %RBX,%ZMM8 |
(438) 0x4491a2 MOV -0x2c(%RBP),%EDX |
(438) 0x4491a5 MOV -0x60(%RBP),%R10 |
(438) 0x4491a9 VPBROADCASTQ %R12,%ZMM11 |
(438) 0x4491af VPSUBQ %ZMM11,%ZMM8,%ZMM8 |
(438) 0x4491b5 VPCMPNLEUQ %ZMM0,%ZMM8,%K1 |
(438) 0x4491bc ADD %R12,%R11 |
(438) 0x4491bf VPBROADCASTQ %R11,%ZMM8 |
(438) 0x4491c5 VPADDQ %ZMM0,%ZMM8,%ZMM8 |
(438) 0x4491cb VPMULLQ %ZMM8,%ZMM5,%ZMM5 |
(438) 0x4491d1 VPADDQ %ZMM5,%ZMM6,%ZMM5 |
(438) 0x4491d7 KMOVQ %K1,%K2 |
(438) 0x4491dc VPXOR %XMM6,%XMM6,%XMM6 |
(438) 0x4491e0 VGATHERQPD (,%ZMM5,1),%ZMM6{%K2} |
(438) 0x4491eb VPADDQ %ZMM9,%ZMM7,%ZMM5 |
(438) 0x4491f1 VMOVAPD %ZMM6,%ZMM3{%K1} |
(438) 0x4491f7 VPSUBQ %ZMM11,%ZMM10,%ZMM6 |
(438) 0x4491fd VPADDQ %ZMM1,%ZMM6,%ZMM6 |
(438) 0x449203 VPMULLQ %ZMM6,%ZMM4,%ZMM4 |
(438) 0x449209 VPADDQ %ZMM4,%ZMM5,%ZMM4 |
(438) 0x44920f VSCATTERQPD %ZMM3,(,%ZMM4,1){%K1} |
(438) 0x44921a JMP 448fc3 |
0x44921f NOPW %CS:(%RAX,%RAX,1) |
0x449229 NOPW %CS:(%RAX,%RAX,1) |
0x449233 NOPW %CS:(%RAX,%RAX,1) |
0x44923d NOPL (%RAX) |
Path / |
Source file and lines | pack_kernel.f90:108-116 |
Module | exec |
nb instructions | 75 |
nb uops | 78 |
loop length | 373 |
used x86 registers | 14 |
used mmx registers | 0 |
used xmm registers | 0 |
used ymm registers | 0 |
used zmm registers | 3 |
nb stack references | 10 |
micro-operation queue | 13.00 cycles |
front end | 13.00 cycles |
P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 2.10 | 2.00 | 5.67 | 5.67 | 10.00 | 2.00 | 1.90 | 10.00 | 10.00 | 10.00 | 2.00 | 5.67 |
cycles | 2.10 | 2.20 | 5.67 | 5.67 | 10.00 | 2.00 | 1.90 | 10.00 | 10.00 | 10.00 | 2.00 | 5.67 |
Cycles executing div or sqrt instructions | NA |
FE+BE cycles | 12.62-12.68 |
Stall cycles | 0.00 |
Front-end | 13.00 |
Dispatch | 10.00 |
Overall L1 | 13.00 |
all | 16% |
load | 42% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 25% |
all | 20% |
load | 46% |
store | 8% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 11% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 10% |
Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
PUSH %RBP | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
PUSH %R15 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R14 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R13 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R12 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %RBX | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
SUB $0x48,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %R8,-0x58(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV %RCX,-0x70(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV 0x28(%RBP),%EAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOVL $0,-0x4c(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
TEST %EAX,%EAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 2 | 0.20 |
JS 448f46 <pack_kernel_module_mp_clover_unpack_message_left_.DIR.OMP.PARALLEL.LOOP.2.split99+0x86> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RDX,%RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV (%RDI),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOVL $0,-0x34(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV %EAX,-0x30(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOVL $0x1,-0x48(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
SUB $0x8,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
LEA -0x48(%RBP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x4c(%RBP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x34(%RBP),%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x30(%RBP),%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV $0x74e530,%EDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV %ESI,-0x44(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV $0x22,%EDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
PUSH $0x1 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH $0x1 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %RAX | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
CALL 4044c0 <__kmpc_for_static_init_4@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
ADD $0x20,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV -0x34(%RBP),%EAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x30(%RBP),%EDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
SUB %EAX,%EDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE 448f80 <pack_kernel_module_mp_clover_unpack_message_left_.DIR.OMP.PARALLEL.LOOP.2.split99+0xc0> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV $0x74e550,%EDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV -0x44(%RBP),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VZEROUPPER | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 |
CALL 4040b0 <__kmpc_for_static_fini@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
ADD $0x48,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
POP %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
RET | 1 | 0.50 | 0 | 0.33 | 0.33 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0.33 | 0 | 2.13 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV %RAX,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV -0x58(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RAX),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA -0x1(%RCX,%RBX,1),%EDI | 1 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
XOR %R8D,%R8D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
ADD %EBX,%ECX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV %RCX,-0x68(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
VMOVDQA64 0xc10a0(%RIP),%ZMM0 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.50 |
VMOVDQA64 0xc10d6(%RIP),%ZMM1 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.50 |
VMOVDQA64 0xc108c(%RIP),%ZMM2 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.50 |
MOV %EDX,-0x2c(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
JMP 448fd8 <pack_kernel_module_mp_clover_unpack_message_left_.DIR.OMP.PARALLEL.LOOP.2.split99+0x118> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XCHG %AX,%AX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
Source file and lines | pack_kernel.f90:108-116 |
Module | exec |
nb instructions | 75 |
nb uops | 78 |
loop length | 373 |
used x86 registers | 14 |
used mmx registers | 0 |
used xmm registers | 0 |
used ymm registers | 0 |
used zmm registers | 3 |
nb stack references | 10 |
micro-operation queue | 13.00 cycles |
front end | 13.00 cycles |
P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 2.10 | 2.00 | 5.67 | 5.67 | 10.00 | 2.00 | 1.90 | 10.00 | 10.00 | 10.00 | 2.00 | 5.67 |
cycles | 2.10 | 2.20 | 5.67 | 5.67 | 10.00 | 2.00 | 1.90 | 10.00 | 10.00 | 10.00 | 2.00 | 5.67 |
Cycles executing div or sqrt instructions | NA |
FE+BE cycles | 12.62-12.68 |
Stall cycles | 0.00 |
Front-end | 13.00 |
Dispatch | 10.00 |
Overall L1 | 13.00 |
all | 16% |
load | 42% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 25% |
all | 20% |
load | 46% |
store | 8% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 11% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 10% |
Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
PUSH %RBP | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
PUSH %R15 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R14 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R13 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R12 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %RBX | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
SUB $0x48,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %R8,-0x58(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV %RCX,-0x70(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV 0x28(%RBP),%EAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOVL $0,-0x4c(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
TEST %EAX,%EAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 2 | 0.20 |
JS 448f46 <pack_kernel_module_mp_clover_unpack_message_left_.DIR.OMP.PARALLEL.LOOP.2.split99+0x86> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RDX,%RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV (%RDI),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOVL $0,-0x34(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV %EAX,-0x30(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOVL $0x1,-0x48(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
SUB $0x8,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
LEA -0x48(%RBP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x4c(%RBP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x34(%RBP),%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x30(%RBP),%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV $0x74e530,%EDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV %ESI,-0x44(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV $0x22,%EDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
PUSH $0x1 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH $0x1 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %RAX | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
CALL 4044c0 <__kmpc_for_static_init_4@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
ADD $0x20,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV -0x34(%RBP),%EAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x30(%RBP),%EDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
SUB %EAX,%EDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE 448f80 <pack_kernel_module_mp_clover_unpack_message_left_.DIR.OMP.PARALLEL.LOOP.2.split99+0xc0> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV $0x74e550,%EDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV -0x44(%RBP),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VZEROUPPER | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 |
CALL 4040b0 <__kmpc_for_static_fini@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
ADD $0x48,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
POP %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
RET | 1 | 0.50 | 0 | 0.33 | 0.33 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0.33 | 0 | 2.13 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV %RAX,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV -0x58(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RAX),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA -0x1(%RCX,%RBX,1),%EDI | 1 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
XOR %R8D,%R8D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
ADD %EBX,%ECX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV %RCX,-0x68(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
VMOVDQA64 0xc10a0(%RIP),%ZMM0 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.50 |
VMOVDQA64 0xc10d6(%RIP),%ZMM1 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.50 |
VMOVDQA64 0xc108c(%RIP),%ZMM2 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.50 |
MOV %EDX,-0x2c(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
JMP 448fd8 <pack_kernel_module_mp_clover_unpack_message_left_.DIR.OMP.PARALLEL.LOOP.2.split99+0x118> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XCHG %AX,%AX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
Name | Coverage (%) | Time (s) |
---|---|---|
▼clover_unpack_message_left_.DIR.OMP.PARALLEL.LOOP.2.split99– | 0.03 | 0.01 |
▼Loop 438 - pack_kernel.f90:109-113 - exec– | 0.03 | 0.02 |
○Loop 439 - pack_kernel.f90:111-113 - exec | 0 | 0 |