Function: clover_pack_message_right_.DIR.OMP.PARALLEL.LOOP.2.split100 | Module: exec | Source: pack_kernel.f90:155-163 | Coverage: 0.06% |
---|
Function: clover_pack_message_right_.DIR.OMP.PARALLEL.LOOP.2.split100 | Module: exec | Source: pack_kernel.f90:155-163 | Coverage: 0.06% |
---|
/scratch_na/users/xoserete/qaas_runs/171-214-9740/intel/CloverLeafFC/build/CloverLeafFC/CloverLeaf_ref/kernels/pack_kernel.f90: 155 - 163 |
-------------------------------------------------------------------------------- |
155: !$OMP PARALLEL DO PRIVATE(index) |
156: DO k=y_min-depth,y_max+y_inc+depth |
157: !$OMP SIMD |
158: DO j=1,depth |
159: index= buffer_offset + j+(k+depth-1)*depth |
160: right_snd_buffer(index)=field(x_max+1-j,k) |
161: ENDDO |
162: ENDDO |
163: !$OMP END PARALLEL DO |
0x449240 PUSH %RBP |
0x449241 MOV %RSP,%RBP |
0x449244 PUSH %R15 |
0x449246 PUSH %R14 |
0x449248 PUSH %R13 |
0x44924a PUSH %R12 |
0x44924c PUSH %RBX |
0x44924d SUB $0x58,%RSP |
0x449251 MOV %R8,-0x58(%RBP) |
0x449255 MOV %RCX,-0x70(%RBP) |
0x449259 MOV 0x28(%RBP),%EAX |
0x44925c MOVL $0,-0x4c(%RBP) |
0x449263 TEST %EAX,%EAX |
0x449265 JS 4492c6 |
0x449267 MOV %RDX,%RBX |
0x44926a MOV (%RDI),%ESI |
0x44926c MOVL $0,-0x34(%RBP) |
0x449273 MOV %EAX,-0x30(%RBP) |
0x449276 MOVL $0x1,-0x48(%RBP) |
0x44927d SUB $0x8,%RSP |
0x449281 LEA -0x48(%RBP),%RAX |
0x449285 LEA -0x4c(%RBP),%RCX |
0x449289 LEA -0x34(%RBP),%R8 |
0x44928d LEA -0x30(%RBP),%R9 |
0x449291 MOV $0x74e590,%EDI |
0x449296 MOV %ESI,-0x44(%RBP) |
0x449299 MOV $0x22,%EDX |
0x44929e PUSH $0x1 |
0x4492a0 PUSH $0x1 |
0x4492a2 PUSH %RAX |
0x4492a3 CALL 4044c0 <__kmpc_for_static_init_4@plt> |
0x4492a8 ADD $0x20,%RSP |
0x4492ac MOV -0x34(%RBP),%EAX |
0x4492af MOV -0x30(%RBP),%EDX |
0x4492b2 SUB %EAX,%EDX |
0x4492b4 JAE 449300 |
0x4492b6 MOV $0x74e5b0,%EDI |
0x4492bb MOV -0x44(%RBP),%ESI |
0x4492be VZEROUPPER |
0x4492c1 CALL 4040b0 <__kmpc_for_static_fini@plt> |
0x4492c6 ADD $0x58,%RSP |
0x4492ca POP %RBX |
0x4492cb POP %R12 |
0x4492cd POP %R13 |
0x4492cf POP %R14 |
0x4492d1 POP %R15 |
0x4492d3 POP %RBP |
0x4492d4 RET |
0x4492d5 NOPW %CS:(%RAX,%RAX,1) |
0x4492e4 NOPW %CS:(%RAX,%RAX,1) |
0x4492f3 NOPW %CS:(%RAX,%RAX,1) |
0x449300 MOV %RAX,%RCX |
0x449303 MOV -0x58(%RBP),%RAX |
0x449307 MOV (%RAX),%ESI |
0x449309 LEA -0x1(%RCX,%RBX,1),%EDI |
0x44930d XOR %R8D,%R8D |
0x449310 ADD %EBX,%ECX |
0x449312 MOV %RCX,-0x68(%RBP) |
0x449316 VMOVDQA64 0xc0da0(%RIP),%ZMM0 |
0x449320 VMOVDQA 0xc1c98(%RIP),%YMM1 |
0x449328 VPTERNLOGD $-0x1,%ZMM2,%ZMM2,%ZMM2 |
0x44932f VMOVDQA64 0xc0d87(%RIP),%ZMM3 |
0x449339 MOV %EDX,-0x2c(%RBP) |
0x44933c JMP 449358 |
0x44933e XCHG %AX,%AX |
(440) 0x449340 MOV %ESI,%R10D |
(440) 0x449343 LEA 0x1(%R8),%EAX |
(440) 0x449347 INC %EDI |
(440) 0x449349 MOV %R10D,%ESI |
(440) 0x44934c CMP %EDX,%R8D |
(440) 0x44934f MOV %EAX,%R8D |
(440) 0x449352 JE 4492b6 |
(440) 0x449358 TEST %ESI,%ESI |
(440) 0x44935a JLE 449340 |
(440) 0x44935c MOV -0x68(%RBP),%RAX |
(440) 0x449360 LEA (%RAX,%R8,1),%R12D |
(440) 0x449364 MOV -0x70(%RBP),%RAX |
(440) 0x449368 MOVSXD (%RAX),%RAX |
(440) 0x44936b MOV %RAX,-0x40(%RBP) |
(440) 0x44936f MOV -0x58(%RBP),%RCX |
(440) 0x449373 MOV (%RCX),%R10D |
(440) 0x449376 MOV 0x10(%RBP),%R9 |
(440) 0x44937a MOV (%R9),%RDX |
(440) 0x44937d MOV 0x38(%R9),%R13 |
(440) 0x449381 MOV 0x18(%RBP),%RCX |
(440) 0x449385 MOV (%RCX),%EAX |
(440) 0x449387 MOV 0x50(%R9),%R9 |
(440) 0x44938b MOV 0x30a96e(%RIP),%R15 |
(440) 0x449392 MOV 0x30a99f(%RIP),%RCX |
(440) 0x449399 MOV %ESI,%R14D |
(440) 0x44939c MOV %R14,%R11 |
(440) 0x44939f MOVSXD %R12D,%RBX |
(440) 0x4493a2 MOV $-0x8,%ESI |
(440) 0x4493a7 AND %RSI,%R11 |
(440) 0x4493aa JE 449480 |
(440) 0x4493b0 LEA (%R10,%RDI,1),%ESI |
(440) 0x4493b4 IMUL %R10D,%ESI |
(440) 0x4493b8 MOVSXD %ESI,%RSI |
(440) 0x4493bb MOV %RCX,-0x78(%RBP) |
(440) 0x4493bf MOV -0x40(%RBP),%RCX |
(440) 0x4493c3 ADD %RCX,%RSI |
(440) 0x4493c6 VPBROADCASTQ %R13,%ZMM5 |
(440) 0x4493cc LEA -0x1(%R12,%R10,1),%R12D |
(440) 0x4493d1 MOV %R10,-0x60(%RBP) |
(440) 0x4493d5 IMUL %R10D,%R12D |
(440) 0x4493d9 MOVSXD %R12D,%R12 |
(440) 0x4493dc ADD %RCX,%R12 |
(440) 0x4493df MOV -0x78(%RBP),%RCX |
(440) 0x4493e3 VPBROADCASTQ %RCX,%ZMM6 |
(440) 0x4493e9 MOV %EAX,-0x40(%RBP) |
(440) 0x4493ec XOR %ECX,%ECX |
(440) 0x4493ee XCHG %AX,%AX |
(441) 0x4493f0 LEA 0x1(%RBX),%R13 |
(441) 0x4493f4 IMUL %R9,%R13 |
(441) 0x4493f8 VPBROADCASTD %EAX,%YMM7 |
(441) 0x4493fe VPADDD %YMM1,%YMM7,%YMM7 |
(441) 0x449402 VPMOVSXDQ %YMM7,%ZMM7 |
(441) 0x449408 VPSUBQ %ZMM2,%ZMM7,%ZMM7 |
(441) 0x44940e VPMULLQ %ZMM7,%ZMM5,%ZMM7 |
(441) 0x449414 LEA (%RDX,%R13,1),%R10 |
(441) 0x449418 KXNORW %K0,%K0,%K1 |
(441) 0x44941c VXORPD %XMM8,%XMM8,%XMM8 |
(441) 0x449421 VGATHERQPD (%R10,%ZMM7,1),%ZMM8{%K1} |
(441) 0x449428 LEA (%RSI,%RCX,1),%R10 |
(441) 0x44942c VPBROADCASTQ %R10,%ZMM7 |
(441) 0x449432 VPADDQ %ZMM3,%ZMM7,%ZMM7 |
(441) 0x449438 VPMULLQ %ZMM7,%ZMM6,%ZMM7 |
(441) 0x44943e KXNORW %K0,%K0,%K1 |
(441) 0x449442 VSCATTERQPD %ZMM8,(%R15,%ZMM7,1){%K1} |
(441) 0x449449 ADD $0x8,%RCX |
(441) 0x44944d ADD $-0x8,%EAX |
(441) 0x449450 CMP %R11,%RCX |
(441) 0x449453 JB 4493f0 |
(440) 0x449455 CMP %R14,%R11 |
(440) 0x449458 JNE 449500 |
(440) 0x44945e MOV -0x2c(%RBP),%EDX |
(440) 0x449461 MOV -0x60(%RBP),%R10 |
(440) 0x449465 JMP 449343 |
0x44946a NOPW %CS:(%RAX,%RAX,1) |
0x449479 NOPL (%RAX) |
(440) 0x449480 VPBROADCASTQ %R14,%ZMM8 |
(440) 0x449486 VPBROADCASTQ %RDX,%ZMM9 |
(440) 0x44948c INC %RBX |
(440) 0x44948f IMUL %RBX,%R9 |
(440) 0x449493 VPBROADCASTQ %R9,%ZMM11 |
(440) 0x449499 VPBROADCASTD %EAX,%YMM10 |
(440) 0x44949f VPBROADCASTQ %R13,%ZMM5 |
(440) 0x4494a5 VPBROADCASTQ %R15,%ZMM7 |
(440) 0x4494ab LEA -0x1(%R12,%R10,1),%EDX |
(440) 0x4494b0 IMUL %R10D,%EDX |
(440) 0x4494b4 MOVSXD %EDX,%R12 |
(440) 0x4494b7 ADD -0x40(%RBP),%R12 |
(440) 0x4494bb VPBROADCASTQ %RCX,%ZMM6 |
(440) 0x4494c1 XOR %R11D,%R11D |
(440) 0x4494c4 MOV -0x2c(%RBP),%EDX |
(440) 0x4494c7 JMP 449528 |
0x4494c9 NOPW %CS:(%RAX,%RAX,1) |
0x4494d8 NOPW %CS:(%RAX,%RAX,1) |
0x4494e7 NOPW %CS:(%RAX,%RAX,1) |
0x4494f6 NOPW %CS:(%RAX,%RAX,1) |
(440) 0x449500 VPBROADCASTQ %R13,%ZMM11 |
(440) 0x449506 VPBROADCASTQ %R15,%ZMM7 |
(440) 0x44950c VPBROADCASTQ %RDX,%ZMM9 |
(440) 0x449512 MOV -0x40(%RBP),%EAX |
(440) 0x449515 VPBROADCASTD %EAX,%YMM10 |
(440) 0x44951b VPBROADCASTQ %R14,%ZMM8 |
(440) 0x449521 MOV -0x2c(%RBP),%EDX |
(440) 0x449524 MOV -0x60(%RBP),%R10 |
(440) 0x449528 VPADDQ %ZMM11,%ZMM9,%ZMM9 |
(440) 0x44952e VPBROADCASTQ %R11,%ZMM11 |
(440) 0x449534 VPSUBQ %ZMM11,%ZMM8,%ZMM8 |
(440) 0x44953a VPCMPNLEUQ %ZMM0,%ZMM8,%K1 |
(440) 0x449541 VPBROADCASTD %R11D,%YMM8 |
(440) 0x449547 VPSUBD %YMM8,%YMM10,%YMM8 |
(440) 0x44954c VPADDD %YMM1,%YMM8,%YMM8 |
(440) 0x449550 VPMOVSXDQ %YMM8,%ZMM8 |
(440) 0x449556 VPSUBQ %ZMM2,%ZMM8,%ZMM8 |
(440) 0x44955c VPMULLQ %ZMM8,%ZMM5,%ZMM5 |
(440) 0x449562 VPADDQ %ZMM5,%ZMM9,%ZMM5 |
(440) 0x449568 VPXOR %XMM8,%XMM8,%XMM8 |
(440) 0x44956d KMOVQ %K1,%K2 |
(440) 0x449572 VGATHERQPD (,%ZMM5,1),%ZMM8{%K2} |
(440) 0x44957d VMOVAPD %ZMM8,%ZMM4{%K1} |
(440) 0x449583 ADD %R11,%R12 |
(440) 0x449586 VPBROADCASTQ %R12,%ZMM5 |
(440) 0x44958c VPADDQ %ZMM0,%ZMM5,%ZMM5 |
(440) 0x449592 VPMULLQ %ZMM5,%ZMM6,%ZMM5 |
(440) 0x449598 VPADDQ %ZMM5,%ZMM7,%ZMM5 |
(440) 0x44959e VSCATTERQPD %ZMM4,(,%ZMM5,1){%K1} |
(440) 0x4495a9 JMP 449343 |
0x4495ae NOPW %CS:(%RAX,%RAX,1) |
0x4495b8 NOPL (%RAX,%RAX,1) |
Path / |
Source file and lines | pack_kernel.f90:155-163 |
Module | exec |
nb instructions | 73 |
nb uops | 76 |
loop length | 351 |
used x86 registers | 14 |
used mmx registers | 0 |
used xmm registers | 0 |
used ymm registers | 1 |
used zmm registers | 3 |
nb stack references | 10 |
micro-operation queue | 12.67 cycles |
front end | 12.67 cycles |
P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 2.30 | 2.20 | 5.67 | 5.67 | 10.00 | 2.20 | 2.10 | 10.00 | 10.00 | 10.00 | 2.20 | 5.67 |
cycles | 2.30 | 2.20 | 5.67 | 5.67 | 10.00 | 2.20 | 2.10 | 10.00 | 10.00 | 10.00 | 2.20 | 5.67 |
Cycles executing div or sqrt instructions | NA |
FE+BE cycles | 12.29-12.35 |
Stall cycles | 0.00 |
Front-end | 12.67 |
Dispatch | 10.00 |
Overall L1 | 12.67 |
all | 19% |
load | 42% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 40% |
all | 21% |
load | 39% |
store | 8% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 11% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 28% |
Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
PUSH %RBP | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
PUSH %R15 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R14 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R13 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R12 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %RBX | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
SUB $0x58,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %R8,-0x58(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV %RCX,-0x70(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV 0x28(%RBP),%EAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOVL $0,-0x4c(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
TEST %EAX,%EAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 2 | 0.20 |
JS 4492c6 <pack_kernel_module_mp_clover_pack_message_right_.DIR.OMP.PARALLEL.LOOP.2.split100+0x86> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RDX,%RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV (%RDI),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOVL $0,-0x34(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV %EAX,-0x30(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOVL $0x1,-0x48(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
SUB $0x8,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
LEA -0x48(%RBP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x4c(%RBP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x34(%RBP),%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x30(%RBP),%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV $0x74e590,%EDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV %ESI,-0x44(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV $0x22,%EDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
PUSH $0x1 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH $0x1 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %RAX | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
CALL 4044c0 <__kmpc_for_static_init_4@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
ADD $0x20,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV -0x34(%RBP),%EAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x30(%RBP),%EDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
SUB %EAX,%EDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE 449300 <pack_kernel_module_mp_clover_pack_message_right_.DIR.OMP.PARALLEL.LOOP.2.split100+0xc0> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV $0x74e5b0,%EDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV -0x44(%RBP),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VZEROUPPER | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 |
CALL 4040b0 <__kmpc_for_static_fini@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
ADD $0x58,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
POP %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
RET | 1 | 0.50 | 0 | 0.33 | 0.33 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0.33 | 0 | 2.13 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV %RAX,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV -0x58(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RAX),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA -0x1(%RCX,%RBX,1),%EDI | 1 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
XOR %R8D,%R8D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
ADD %EBX,%ECX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV %RCX,-0x68(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
VMOVDQA64 0xc0da0(%RIP),%ZMM0 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.50 |
VMOVDQA 0xc1c98(%RIP),%YMM1 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VPTERNLOGD $-0x1,%ZMM2,%ZMM2,%ZMM2 | 1 | 0.50 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
VMOVDQA64 0xc0d87(%RIP),%ZMM3 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.50 |
MOV %EDX,-0x2c(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
JMP 449358 <pack_kernel_module_mp_clover_pack_message_right_.DIR.OMP.PARALLEL.LOOP.2.split100+0x118> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
XCHG %AX,%AX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
Source file and lines | pack_kernel.f90:155-163 |
Module | exec |
nb instructions | 73 |
nb uops | 76 |
loop length | 351 |
used x86 registers | 14 |
used mmx registers | 0 |
used xmm registers | 0 |
used ymm registers | 1 |
used zmm registers | 3 |
nb stack references | 10 |
micro-operation queue | 12.67 cycles |
front end | 12.67 cycles |
P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 2.30 | 2.20 | 5.67 | 5.67 | 10.00 | 2.20 | 2.10 | 10.00 | 10.00 | 10.00 | 2.20 | 5.67 |
cycles | 2.30 | 2.20 | 5.67 | 5.67 | 10.00 | 2.20 | 2.10 | 10.00 | 10.00 | 10.00 | 2.20 | 5.67 |
Cycles executing div or sqrt instructions | NA |
FE+BE cycles | 12.29-12.35 |
Stall cycles | 0.00 |
Front-end | 12.67 |
Dispatch | 10.00 |
Overall L1 | 12.67 |
all | 19% |
load | 42% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 40% |
all | 21% |
load | 39% |
store | 8% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 11% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 28% |
Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
PUSH %RBP | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
PUSH %R15 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R14 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R13 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %R12 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %RBX | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
SUB $0x58,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV %R8,-0x58(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV %RCX,-0x70(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV 0x28(%RBP),%EAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOVL $0,-0x4c(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
TEST %EAX,%EAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 2 | 0.20 |
JS 4492c6 <pack_kernel_module_mp_clover_pack_message_right_.DIR.OMP.PARALLEL.LOOP.2.split100+0x86> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RDX,%RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV (%RDI),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOVL $0,-0x34(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV %EAX,-0x30(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOVL $0x1,-0x48(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
SUB $0x8,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
LEA -0x48(%RBP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x4c(%RBP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x34(%RBP),%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
LEA -0x30(%RBP),%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV $0x74e590,%EDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV %ESI,-0x44(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
MOV $0x22,%EDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
PUSH $0x1 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH $0x1 | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
PUSH %RAX | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 5-12 | 0.50 |
CALL 4044c0 <__kmpc_for_static_init_4@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
ADD $0x20,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV -0x34(%RBP),%EAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV -0x30(%RBP),%EDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
SUB %EAX,%EDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JAE 449300 <pack_kernel_module_mp_clover_pack_message_right_.DIR.OMP.PARALLEL.LOOP.2.split100+0xc0> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV $0x74e5b0,%EDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV -0x44(%RBP),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VZEROUPPER | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 |
CALL 4040b0 <__kmpc_for_static_fini@plt> | 2 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 0 | 1 |
ADD $0x58,%RSP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
POP %RBX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R12 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %R15 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
POP %RBP | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1-6 | 0.33 |
RET | 1 | 0.50 | 0 | 0.33 | 0.33 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0.33 | 0 | 2.13 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV %RAX,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
MOV -0x58(%RBP),%RAX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV (%RAX),%ESI | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
LEA -0x1(%RCX,%RBX,1),%EDI | 1 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
XOR %R8D,%R8D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
ADD %EBX,%ECX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV %RCX,-0x68(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
VMOVDQA64 0xc0da0(%RIP),%ZMM0 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.50 |
VMOVDQA 0xc1c98(%RIP),%YMM1 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VPTERNLOGD $-0x1,%ZMM2,%ZMM2,%ZMM2 | 1 | 0.50 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
VMOVDQA64 0xc0d87(%RIP),%ZMM3 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.50 |
MOV %EDX,-0x2c(%RBP) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
JMP 449358 <pack_kernel_module_mp_clover_pack_message_right_.DIR.OMP.PARALLEL.LOOP.2.split100+0x118> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 5.84 |
XCHG %AX,%AX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
NOPL (%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
Name | Coverage (%) | Time (s) |
---|---|---|
▼clover_pack_message_right_.DIR.OMP.PARALLEL.LOOP.2.split100– | 0.06 | 0.02 |
▼Loop 440 - pack_kernel.f90:156-160 - exec– | 0.06 | 0.03 |
○Loop 441 - pack_kernel.f90:158-160 - exec | 0 | 0 |