| Loop Id: 37 | Module: attention-avx512 | Source: attention.cpp:43-284 [...] | Coverage: 0.11% |
|---|
| Loop Id: 37 | Module: attention-avx512 | Source: attention.cpp:43-284 [...] | Coverage: 0.11% |
|---|
0x5da0 MOV 0x190(%RSP),%RAX |
0x5da8 MOV 0x1d0(%RSP),%RDX |
0x5db0 MOV 0x218(%RSP),%RSI |
0x5db8 INC %R13 |
0x5dbb MOV %R12,%RDI |
0x5dbe VMOVSS %XMM2,(%RAX,%RCX,4) |
0x5dc3 MOV 0x1a0(%RSP),%RCX |
0x5dcb ADD %RDX,%RSI |
0x5dce ADD %RDX,0x18(%RSP) |
0x5dd3 ADD %RDX,%RCX |
0x5dd6 MOV %RCX,%RAX |
0x5dd9 CMP 0x58(%RSP),%R12 |
0x5dde MOV %R13,%R12 |
0x5de1 MOV 0x158(%RSP),%R13 |
0x5de9 JE 5d50 |
0x5def MOV %R12,%R8 |
0x5df2 MOV %R12,%R9 |
0x5df5 AND $-0x4,%R8 |
0x5df9 AND $-0x20,%R9 |
0x5dfd MOV %RAX,0x1a0(%RSP) |
0x5e05 CMP $0x4,%R12 |
0x5e09 JAE 5e20 |
0x5e0b VMOVSS -0x4f63(%RIP),%XMM1 |
0x5e13 MOV 0x18(%RSP),%R13 |
0x5e18 XOR %EAX,%EAX |
0x5e1a JMP 5f10 |
0x5e20 MOV 0x18(%RSP),%R13 |
0x5e25 CMP $0x20,%R12 |
0x5e29 JAE 5e40 |
0x5e2b VMOVSS -0x4f83(%RIP),%XMM1 |
0x5e33 XOR %EAX,%EAX |
0x5e35 JMP 5ec2 |
0x5e40 VBROADCASTSS -0x4f99(%RIP),%YMM0 |
0x5e49 MOV $0x7ffffffffffffffc,%RAX |
0x5e53 XOR %ECX,%ECX |
0x5e55 ADD $-0x1c,%RAX |
0x5e59 AND %R12,%RAX |
0x5e5c VMOVAPS %YMM0,%YMM1 |
0x5e60 VMOVAPS %YMM0,%YMM2 |
0x5e64 VMOVAPS %YMM0,%YMM3 |
0x5e68 NOPL (%RAX,%RAX,1) |
(32) 0x5e70 VMAXPS -0x60(%RSI,%RCX,4),%YMM0,%YMM0 |
(32) 0x5e76 VMAXPS -0x40(%RSI,%RCX,4),%YMM1,%YMM1 |
(32) 0x5e7c VMAXPS -0x20(%RSI,%RCX,4),%YMM2,%YMM2 |
(32) 0x5e82 VMAXPS (%RSI,%RCX,4),%YMM3,%YMM3 |
(32) 0x5e87 ADD $0x20,%RCX |
(32) 0x5e8b CMP %RCX,%R9 |
(32) 0x5e8e JNE 5e70 |
0x5e90 VMAXPS %YMM1,%YMM0,%YMM0 |
0x5e94 VMAXPS %YMM3,%YMM2,%YMM1 |
0x5e98 VMAXPS %YMM1,%YMM0,%YMM0 |
0x5e9c VEXTRACTF128 $0x1,%YMM0,%XMM1 |
0x5ea2 VMAXPS %XMM1,%XMM0,%XMM0 |
0x5ea6 VSHUFPD $0x1,%XMM0,%XMM0,%XMM1 |
0x5eab VMAXPS %XMM1,%XMM0,%XMM0 |
0x5eaf VMOVSHDUP %XMM0,%XMM1 |
0x5eb3 VMAXSS %XMM1,%XMM0,%XMM1 |
0x5eb7 CMP %RAX,%R12 |
0x5eba JE 5f1f |
0x5ebc TEST $0x1c,%R12B |
0x5ec0 JE 5f10 |
0x5ec2 MOV $0x7ffffffffffffffc,%RDX |
0x5ecc VBROADCASTSS %XMM1,%XMM0 |
0x5ed1 MOV %RAX,%RCX |
0x5ed4 MOV %R12,%RAX |
0x5ed7 AND %RDX,%RAX |
0x5eda NOPW (%RAX,%RAX,1) |
(42) 0x5ee0 VMAXPS (%R13,%RCX,4),%XMM0,%XMM0 |
(42) 0x5ee7 ADD $0x4,%RCX |
(42) 0x5eeb CMP %RCX,%R8 |
(42) 0x5eee JNE 5ee0 |
0x5ef0 VSHUFPD $0x1,%XMM0,%XMM0,%XMM1 |
0x5ef5 VMAXPS %XMM1,%XMM0,%XMM0 |
0x5ef9 VMOVSHDUP %XMM0,%XMM1 |
0x5efd VMAXSS %XMM1,%XMM0,%XMM1 |
0x5f01 JMP 5f1a |
(41) 0x5f10 VMAXSS (%R13,%RAX,4),%XMM1,%XMM1 |
(41) 0x5f17 INC %RAX |
(41) 0x5f1a CMP %RAX,%R12 |
(41) 0x5f1d JNE 5f10 |
0x5f1f VMOVAPS %XMM1,0x160(%RSP) |
0x5f28 MOV %RSI,0x218(%RSP) |
0x5f30 MOV %RDI,0x210(%RSP) |
0x5f38 MOV %R8,0x180(%RSP) |
0x5f40 MOV %R13,0x18(%RSP) |
0x5f45 CMP $0x4,%R12 |
0x5f49 JAE 5f60 |
0x5f4b VXORPS %XMM2,%XMM2,%XMM2 |
0x5f4f XOR %EAX,%EAX |
0x5f51 JMP 665a |
0x5f60 CMP $0x20,%R12 |
0x5f64 JAE 5f80 |
0x5f66 VXORPS %XMM2,%XMM2,%XMM2 |
0x5f6a XOR %EAX,%EAX |
0x5f6c JMP 654f |
0x5f80 MOV $0x7ffffffffffffffc,%RAX |
0x5f8a VBROADCASTSS %XMM1,%YMM0 |
0x5f8f VXORPS %XMM1,%XMM1,%XMM1 |
0x5f93 MOV %R9,0x2a0(%RSP) |
0x5f9b ADD $-0x1c,%RAX |
0x5f9f AND %R12,%RAX |
0x5fa2 VMOVAPS %YMM0,0x2e0(%RSP) |
0x5fab VXORPS %XMM0,%XMM0,%XMM0 |
0x5faf MOV %RAX,0x1d8(%RSP) |
0x5fb7 XOR %EAX,%EAX |
0x5fb9 VMOVAPS %YMM0,0x220(%RSP) |
0x5fc2 VMOVAPS %YMM0,0x240(%RSP) |
0x5fcb VMOVAPS %YMM0,0x2c0(%RSP) |
0x5fd4 NOPW %CS:(%RAX,%RAX,1) |
(33) 0x5fe0 VMOVAPS %YMM1,0x300(%RSP) |
(33) 0x5fe9 VMOVUPS -0x60(%RSI,%RAX,4),%YMM0 |
(33) 0x5fef VMOVAPS 0x2e0(%RSP),%YMM4 |
(33) 0x5ff8 VMOVUPS -0x40(%RSI,%RAX,4),%YMM1 |
(33) 0x5ffe VMOVUPS -0x20(%RSI,%RAX,4),%YMM2 |
(33) 0x6004 VMOVUPS (%RSI,%RAX,4),%YMM3 |
(33) 0x6009 MOV %RAX,0x2a8(%RSP) |
(33) 0x6011 VSUBPS %YMM4,%YMM0,%YMM5 |
(33) 0x6015 VSUBPS %YMM4,%YMM1,%YMM0 |
(33) 0x6019 VMOVAPS %YMM0,0x20(%RSP) |
(33) 0x601f VSUBPS %YMM4,%YMM2,%YMM0 |
(33) 0x6023 VMOVAPS %YMM5,0x60(%RSP) |
(33) 0x6029 VMOVAPS %YMM0,0xc0(%RSP) |
(33) 0x6032 VSUBPS %YMM4,%YMM3,%YMM0 |
(33) 0x6036 VMOVAPS %YMM0,0x80(%RSP) |
(33) 0x603f VEXTRACTF128 $0x1,%YMM5,%XMM0 |
(33) 0x6045 VMOVAPS %XMM0,0xa0(%RSP) |
(33) 0x604e VZEROUPPER |
(33) 0x6051 CALL 7170 <@plt_start@+0x20> |
(33) 0x6056 VMOVAPS %XMM0,0x140(%RSP) |
(33) 0x605f VMOVSHDUP 0xa0(%RSP),%XMM0 |
(33) 0x6068 CALL 7170 <@plt_start@+0x20> |
(33) 0x606d VMOVAPS 0x140(%RSP),%XMM1 |
(33) 0x6076 VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(33) 0x607c VMOVAPS %XMM0,0x140(%RSP) |
(33) 0x6085 VPERMILPD $0x1,0xa0(%RSP),%XMM0 |
(33) 0x6090 CALL 7170 <@plt_start@+0x20> |
(33) 0x6095 VMOVAPS 0x140(%RSP),%XMM1 |
(33) 0x609e VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(33) 0x60a4 VMOVAPS %XMM0,0x140(%RSP) |
(33) 0x60ad VPERMILPS $-0x1,0xa0(%RSP),%XMM0 |
(33) 0x60b8 CALL 7170 <@plt_start@+0x20> |
(33) 0x60bd VMOVAPS 0x140(%RSP),%XMM1 |
(33) 0x60c6 VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(33) 0x60cc VMOVAPS %XMM0,0xa0(%RSP) |
(33) 0x60d5 VMOVAPS 0x60(%RSP),%YMM0 |
(33) 0x60db VZEROUPPER |
(33) 0x60de CALL 7170 <@plt_start@+0x20> |
(33) 0x60e3 VMOVAPS %XMM0,0x140(%RSP) |
(33) 0x60ec VMOVSHDUP 0x60(%RSP),%XMM0 |
(33) 0x60f2 CALL 7170 <@plt_start@+0x20> |
(33) 0x60f7 VMOVAPS 0x140(%RSP),%XMM1 |
(33) 0x6100 VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(33) 0x6106 VMOVAPS %XMM0,0x140(%RSP) |
(33) 0x610f VPERMILPD $0x1,0x60(%RSP),%XMM0 |
(33) 0x6117 CALL 7170 <@plt_start@+0x20> |
(33) 0x611c VMOVAPS 0x140(%RSP),%XMM1 |
(33) 0x6125 VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(33) 0x612b VMOVAPS %XMM0,0x140(%RSP) |
(33) 0x6134 VPERMILPS $-0x1,0x60(%RSP),%XMM0 |
(33) 0x613c CALL 7170 <@plt_start@+0x20> |
(33) 0x6141 VMOVAPS 0x140(%RSP),%XMM1 |
(33) 0x614a VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(33) 0x6150 VINSERTF128 $0x1,0xa0(%RSP),%YMM0,%YMM0 |
(33) 0x615b VMOVAPS 0x220(%RSP),%YMM1 |
(33) 0x6164 VADDPS %YMM1,%YMM0,%YMM1 |
(33) 0x6168 VMOVAPS 0x20(%RSP),%YMM0 |
(33) 0x616e VMOVAPS %YMM1,0x220(%RSP) |
(33) 0x6177 VEXTRACTF128 $0x1,%YMM0,%XMM0 |
(33) 0x617d VMOVAPS %XMM0,0x60(%RSP) |
(33) 0x6183 VZEROUPPER |
(33) 0x6186 CALL 7170 <@plt_start@+0x20> |
(33) 0x618b VMOVAPS %XMM0,0xa0(%RSP) |
(33) 0x6194 VMOVSHDUP 0x60(%RSP),%XMM0 |
(33) 0x619a CALL 7170 <@plt_start@+0x20> |
(33) 0x619f VMOVAPS 0xa0(%RSP),%XMM1 |
(33) 0x61a8 VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(33) 0x61ae VMOVAPS %XMM0,0xa0(%RSP) |
(33) 0x61b7 VPERMILPD $0x1,0x60(%RSP),%XMM0 |
(33) 0x61bf CALL 7170 <@plt_start@+0x20> |
(33) 0x61c4 VMOVAPS 0xa0(%RSP),%XMM1 |
(33) 0x61cd VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(33) 0x61d3 VMOVAPS %XMM0,0xa0(%RSP) |
(33) 0x61dc VPERMILPS $-0x1,0x60(%RSP),%XMM0 |
(33) 0x61e4 CALL 7170 <@plt_start@+0x20> |
(33) 0x61e9 VMOVAPS 0xa0(%RSP),%XMM1 |
(33) 0x61f2 VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(33) 0x61f8 VMOVAPS %XMM0,0x60(%RSP) |
(33) 0x61fe VMOVAPS 0x20(%RSP),%YMM0 |
(33) 0x6204 VZEROUPPER |
(33) 0x6207 CALL 7170 <@plt_start@+0x20> |
(33) 0x620c VMOVAPS %XMM0,0xa0(%RSP) |
(33) 0x6215 VMOVSHDUP 0x20(%RSP),%XMM0 |
(33) 0x621b CALL 7170 <@plt_start@+0x20> |
(33) 0x6220 VMOVAPS 0xa0(%RSP),%XMM1 |
(33) 0x6229 VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(33) 0x622f VMOVAPS %XMM0,0xa0(%RSP) |
(33) 0x6238 VPERMILPD $0x1,0x20(%RSP),%XMM0 |
(33) 0x6240 CALL 7170 <@plt_start@+0x20> |
(33) 0x6245 VMOVAPS 0xa0(%RSP),%XMM1 |
(33) 0x624e VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(33) 0x6254 VMOVAPS %XMM0,0xa0(%RSP) |
(33) 0x625d VPERMILPS $-0x1,0x20(%RSP),%XMM0 |
(33) 0x6265 CALL 7170 <@plt_start@+0x20> |
(33) 0x626a VMOVAPS 0xa0(%RSP),%XMM1 |
(33) 0x6273 VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(33) 0x6279 VINSERTF128 $0x1,0x60(%RSP),%YMM0,%YMM0 |
(33) 0x6281 VMOVAPS 0x240(%RSP),%YMM1 |
(33) 0x628a VADDPS %YMM1,%YMM0,%YMM1 |
(33) 0x628e VMOVAPS 0xc0(%RSP),%YMM0 |
(33) 0x6297 VMOVAPS %YMM1,0x240(%RSP) |
(33) 0x62a0 VEXTRACTF128 $0x1,%YMM0,%XMM0 |
(33) 0x62a6 VMOVAPS %XMM0,0x20(%RSP) |
(33) 0x62ac VZEROUPPER |
(33) 0x62af CALL 7170 <@plt_start@+0x20> |
(33) 0x62b4 VMOVAPS %XMM0,0x60(%RSP) |
(33) 0x62ba VMOVSHDUP 0x20(%RSP),%XMM0 |
(33) 0x62c0 CALL 7170 <@plt_start@+0x20> |
(33) 0x62c5 VMOVAPS 0x60(%RSP),%XMM1 |
(33) 0x62cb VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(33) 0x62d1 VMOVAPS %XMM0,0x60(%RSP) |
(33) 0x62d7 VPERMILPD $0x1,0x20(%RSP),%XMM0 |
(33) 0x62df CALL 7170 <@plt_start@+0x20> |
(33) 0x62e4 VMOVAPS 0x60(%RSP),%XMM1 |
(33) 0x62ea VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(33) 0x62f0 VMOVAPS %XMM0,0x60(%RSP) |
(33) 0x62f6 VPERMILPS $-0x1,0x20(%RSP),%XMM0 |
(33) 0x62fe CALL 7170 <@plt_start@+0x20> |
(33) 0x6303 VMOVAPS 0x60(%RSP),%XMM1 |
(33) 0x6309 VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(33) 0x630f VMOVAPS %XMM0,0x20(%RSP) |
(33) 0x6315 VMOVAPS 0xc0(%RSP),%YMM0 |
(33) 0x631e VZEROUPPER |
(33) 0x6321 CALL 7170 <@plt_start@+0x20> |
(33) 0x6326 VMOVAPS %XMM0,0x60(%RSP) |
(33) 0x632c VMOVSHDUP 0xc0(%RSP),%XMM0 |
(33) 0x6335 CALL 7170 <@plt_start@+0x20> |
(33) 0x633a VMOVAPS 0x60(%RSP),%XMM1 |
(33) 0x6340 VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(33) 0x6346 VMOVAPS %XMM0,0x60(%RSP) |
(33) 0x634c VPERMILPD $0x1,0xc0(%RSP),%XMM0 |
(33) 0x6357 CALL 7170 <@plt_start@+0x20> |
(33) 0x635c VMOVAPS 0x60(%RSP),%XMM1 |
(33) 0x6362 VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(33) 0x6368 VMOVAPS %XMM0,0x60(%RSP) |
(33) 0x636e VPERMILPS $-0x1,0xc0(%RSP),%XMM0 |
(33) 0x6379 CALL 7170 <@plt_start@+0x20> |
(33) 0x637e VMOVAPS 0x60(%RSP),%XMM1 |
(33) 0x6384 VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(33) 0x638a VINSERTF128 $0x1,0x20(%RSP),%YMM0,%YMM0 |
(33) 0x6392 VMOVAPS 0x2c0(%RSP),%YMM1 |
(33) 0x639b VADDPS %YMM1,%YMM0,%YMM1 |
(33) 0x639f VMOVAPS 0x80(%RSP),%YMM0 |
(33) 0x63a8 VMOVAPS %YMM1,0x2c0(%RSP) |
(33) 0x63b1 VEXTRACTF128 $0x1,%YMM0,%XMM0 |
(33) 0x63b7 VMOVAPS %XMM0,0xc0(%RSP) |
(33) 0x63c0 VZEROUPPER |
(33) 0x63c3 CALL 7170 <@plt_start@+0x20> |
(33) 0x63c8 VMOVAPS %XMM0,0x20(%RSP) |
(33) 0x63ce VMOVSHDUP 0xc0(%RSP),%XMM0 |
(33) 0x63d7 CALL 7170 <@plt_start@+0x20> |
(33) 0x63dc VMOVAPS 0x20(%RSP),%XMM1 |
(33) 0x63e2 VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(33) 0x63e8 VMOVAPS %XMM0,0x20(%RSP) |
(33) 0x63ee VPERMILPD $0x1,0xc0(%RSP),%XMM0 |
(33) 0x63f9 CALL 7170 <@plt_start@+0x20> |
(33) 0x63fe VMOVAPS 0x20(%RSP),%XMM1 |
(33) 0x6404 VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(33) 0x640a VMOVAPS %XMM0,0x20(%RSP) |
(33) 0x6410 VPERMILPS $-0x1,0xc0(%RSP),%XMM0 |
(33) 0x641b CALL 7170 <@plt_start@+0x20> |
(33) 0x6420 VMOVAPS 0x20(%RSP),%XMM1 |
(33) 0x6426 VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(33) 0x642c VMOVAPS %XMM0,0xc0(%RSP) |
(33) 0x6435 VMOVAPS 0x80(%RSP),%YMM0 |
(33) 0x643e VZEROUPPER |
(33) 0x6441 CALL 7170 <@plt_start@+0x20> |
(33) 0x6446 VMOVAPS %XMM0,0x20(%RSP) |
(33) 0x644c VMOVSHDUP 0x80(%RSP),%XMM0 |
(33) 0x6455 CALL 7170 <@plt_start@+0x20> |
(33) 0x645a VMOVAPS 0x20(%RSP),%XMM1 |
(33) 0x6460 VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(33) 0x6466 VMOVAPS %XMM0,0x20(%RSP) |
(33) 0x646c VPERMILPD $0x1,0x80(%RSP),%XMM0 |
(33) 0x6477 CALL 7170 <@plt_start@+0x20> |
(33) 0x647c VMOVAPS 0x20(%RSP),%XMM1 |
(33) 0x6482 VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(33) 0x6488 VMOVAPS %XMM0,0x20(%RSP) |
(33) 0x648e VPERMILPS $-0x1,0x80(%RSP),%XMM0 |
(33) 0x6499 CALL 7170 <@plt_start@+0x20> |
(33) 0x649e VMOVAPS 0x20(%RSP),%XMM2 |
(33) 0x64a4 VMOVAPS 0x300(%RSP),%YMM1 |
(33) 0x64ad MOV 0x2a8(%RSP),%RAX |
(33) 0x64b5 MOV 0x2a0(%RSP),%R9 |
(33) 0x64bd MOV 0x218(%RSP),%RSI |
(33) 0x64c5 ADD $0x20,%RAX |
(33) 0x64c9 VINSERTPS $0x30,%XMM0,%XMM2,%XMM0 |
(33) 0x64cf VINSERTF128 $0x1,0xc0(%RSP),%YMM0,%YMM0 |
(33) 0x64da VADDPS %YMM1,%YMM0,%YMM1 |
(33) 0x64de CMP %RAX,%R9 |
(33) 0x64e1 JNE 5fe0 |
0x64e7 VMOVAPS 0x240(%RSP),%YMM0 |
0x64f0 MOV 0x1d8(%RSP),%RAX |
0x64f8 VADDPS 0x220(%RSP),%YMM0,%YMM0 |
0x6501 VADDPS 0x2c0(%RSP),%YMM0,%YMM0 |
0x650a VADDPS %YMM0,%YMM1,%YMM0 |
0x650e VEXTRACTF128 $0x1,%YMM0,%XMM1 |
0x6514 VADDPS %XMM1,%XMM0,%XMM0 |
0x6518 VSHUFPD $0x1,%XMM0,%XMM0,%XMM1 |
0x651d VADDPS %XMM1,%XMM0,%XMM0 |
0x6521 VMOVSHDUP %XMM0,%XMM1 |
0x6525 VADDSS %XMM1,%XMM0,%XMM2 |
0x6529 CMP %RAX,%R12 |
0x652c JNE 653c |
0x652e VMOVAPS 0x160(%RSP),%XMM1 |
0x6537 JMP 66be |
0x653c VMOVAPS 0x160(%RSP),%XMM1 |
0x6545 TEST $0x1c,%R12B |
0x6549 JE 665a |
0x654f VXORPS %XMM0,%XMM0,%XMM0 |
0x6553 VBLENDPS $0x1,%XMM2,%XMM0,%XMM2 |
0x6559 VBROADCASTSS %XMM1,%XMM0 |
0x655e MOV $0x7ffffffffffffffc,%RDX |
0x6568 MOV %R12,%RCX |
0x656b MOV %RAX,%R13 |
0x656e AND %RDX,%RCX |
0x6571 MOV %RCX,0x1d8(%RSP) |
0x6579 VMOVAPS %XMM0,0x60(%RSP) |
0x657f NOP |
(40) 0x6580 MOV 0x18(%RSP),%RAX |
(40) 0x6585 VMOVAPS %XMM2,0xc0(%RSP) |
(40) 0x658e VMOVUPS (%RAX,%R13,4),%XMM0 |
(40) 0x6594 VSUBPS 0x60(%RSP),%XMM0,%XMM0 |
(40) 0x659a VMOVAPS %XMM0,0x80(%RSP) |
(40) 0x65a3 VZEROUPPER |
(40) 0x65a6 CALL 7170 <@plt_start@+0x20> |
(40) 0x65ab VMOVAPS %XMM0,0x20(%RSP) |
(40) 0x65b1 VMOVSHDUP 0x80(%RSP),%XMM0 |
(40) 0x65ba CALL 7170 <@plt_start@+0x20> |
(40) 0x65bf VMOVAPS 0x20(%RSP),%XMM1 |
(40) 0x65c5 VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(40) 0x65cb VMOVAPS %XMM0,0x20(%RSP) |
(40) 0x65d1 VPERMILPD $0x1,0x80(%RSP),%XMM0 |
(40) 0x65dc CALL 7170 <@plt_start@+0x20> |
(40) 0x65e1 VMOVAPS 0x20(%RSP),%XMM1 |
(40) 0x65e7 VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(40) 0x65ed VMOVAPS %XMM0,0x20(%RSP) |
(40) 0x65f3 VPERMILPS $-0x1,0x80(%RSP),%XMM0 |
(40) 0x65fe CALL 7170 <@plt_start@+0x20> |
(40) 0x6603 VMOVAPS 0x20(%RSP),%XMM1 |
(40) 0x6609 VMOVAPS 0xc0(%RSP),%XMM2 |
(40) 0x6612 ADD $0x4,%R13 |
(40) 0x6616 VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(40) 0x661c VADDPS %XMM2,%XMM0,%XMM2 |
(40) 0x6620 CMP %R13,0x180(%RSP) |
(40) 0x6628 JNE 6580 |
0x662e VSHUFPD $0x1,%XMM2,%XMM2,%XMM0 |
0x6633 MOV 0x1d8(%RSP),%RAX |
0x663b MOV 0x18(%RSP),%R13 |
0x6640 VADDPS %XMM0,%XMM2,%XMM0 |
0x6644 VMOVSHDUP %XMM0,%XMM1 |
0x6648 VADDSS %XMM1,%XMM0,%XMM2 |
0x664c VMOVAPS 0x160(%RSP),%XMM1 |
0x6655 CMP %RAX,%R12 |
0x6658 JE 66be |
0x665a MOV %R12,0x1e0(%RSP) |
0x6662 NOPW %CS:(%RAX,%RAX,1) |
(34) 0x6670 VMOVSS (%R13,%RAX,4),%XMM0 |
(34) 0x6677 MOV %R13,%R12 |
(34) 0x667a VMOVAPS %XMM2,0x80(%RSP) |
(34) 0x6683 MOV %RAX,%R13 |
(34) 0x6686 VSUBSS %XMM1,%XMM0,%XMM0 |
(34) 0x668a VZEROUPPER |
(34) 0x668d CALL 7170 <@plt_start@+0x20> |
(34) 0x6692 VMOVAPS 0x80(%RSP),%XMM2 |
(34) 0x669b VMOVAPS 0x160(%RSP),%XMM1 |
(34) 0x66a4 MOV %R13,%RAX |
(34) 0x66a7 MOV %R12,%R13 |
(34) 0x66aa MOV 0x1e0(%RSP),%R12 |
(34) 0x66b2 INC %RAX |
(34) 0x66b5 VADDSS %XMM2,%XMM0,%XMM2 |
(34) 0x66b9 CMP %RAX,%R12 |
(34) 0x66bc JNE 6670 |
0x66be VMOVAPS %XMM2,0x80(%RSP) |
0x66c7 CMP $0x4,%R12 |
0x66cb JAE 66e0 |
0x66cd MOV %R13,%RCX |
0x66d0 XOR %R13D,%R13D |
0x66d3 JMP 69f0 |
0x66e0 CMP $0x8,%R12 |
0x66e4 JAE 6810 |
0x66ea MOV 0x18(%RSP),%RCX |
0x66ef XOR %R13D,%R13D |
0x66f2 VBROADCASTSS %XMM1,%XMM0 |
0x66f7 VBROADCASTSS %XMM2,%XMM1 |
0x66fc MOV $0x7ffffffffffffffc,%RAX |
0x6706 MOV %R13,%RSI |
0x6709 MOV %R12,%R13 |
0x670c AND %RAX,%R13 |
0x670f VMOVAPS %XMM0,0xa0(%RSP) |
0x6718 VMOVAPS %XMM1,0x220(%RSP) |
0x6721 NOPW %CS:(%RAX,%RAX,1) |
(38) 0x6730 VMOVUPS (%RCX,%RSI,4),%XMM0 |
(38) 0x6735 MOV %RSI,0x20(%RSP) |
(38) 0x673a VSUBPS 0xa0(%RSP),%XMM0,%XMM0 |
(38) 0x6743 VMOVAPS %XMM0,0xc0(%RSP) |
(38) 0x674c VZEROUPPER |
(38) 0x674f CALL 7170 <@plt_start@+0x20> |
(38) 0x6754 VMOVAPS %XMM0,0x60(%RSP) |
(38) 0x675a VMOVSHDUP 0xc0(%RSP),%XMM0 |
(38) 0x6763 CALL 7170 <@plt_start@+0x20> |
(38) 0x6768 VMOVAPS 0x60(%RSP),%XMM1 |
(38) 0x676e VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(38) 0x6774 VMOVAPS %XMM0,0x60(%RSP) |
(38) 0x677a VPERMILPD $0x1,0xc0(%RSP),%XMM0 |
(38) 0x6785 CALL 7170 <@plt_start@+0x20> |
(38) 0x678a VMOVAPS 0x60(%RSP),%XMM1 |
(38) 0x6790 VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(38) 0x6796 VMOVAPS %XMM0,0x60(%RSP) |
(38) 0x679c VPERMILPS $-0x1,0xc0(%RSP),%XMM0 |
(38) 0x67a7 CALL 7170 <@plt_start@+0x20> |
(38) 0x67ac VMOVAPS 0x60(%RSP),%XMM1 |
(38) 0x67b2 MOV 0x20(%RSP),%RSI |
(38) 0x67b7 MOV 0x1a0(%RSP),%RDX |
(38) 0x67bf MOV 0x18(%RSP),%RCX |
(38) 0x67c4 VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(38) 0x67ca VDIVPS 0x220(%RSP),%XMM0,%XMM0 |
(38) 0x67d3 VMOVUPS %XMM0,(%RDX,%RSI,4) |
(38) 0x67d8 ADD $0x4,%RSI |
(38) 0x67dc CMP %RSI,0x180(%RSP) |
(38) 0x67e4 JNE 6730 |
0x67ea VMOVAPS 0x160(%RSP),%XMM1 |
0x67f3 VMOVAPS 0x80(%RSP),%XMM2 |
0x67fc CMP %R13,%R12 |
0x67ff JNE 69f0 |
0x6805 JMP 6a33 |
0x6810 MOV 0x18(%RSP),%RCX |
0x6815 MOV %R12,%RAX |
0x6818 AND $-0x8,%RAX |
0x681c VBROADCASTSS %XMM1,%YMM0 |
0x6821 VBROADCASTSS %XMM2,%YMM1 |
0x6826 MOV %RAX,0x220(%RSP) |
0x682e MOV $0x7ffffffffffffffc,%RAX |
0x6838 LEA -0x4(%RAX),%R13 |
0x683c VMOVAPS %YMM0,0x240(%RSP) |
0x6845 VMOVAPS %YMM1,0x2c0(%RSP) |
0x684e XOR %EAX,%EAX |
0x6850 AND %R12,%R13 |
0x6853 NOPW %CS:(%RAX,%RAX,1) |
(35) 0x6860 VMOVUPS (%RCX,%RAX,4),%YMM0 |
(35) 0x6865 MOV %RAX,0x60(%RSP) |
(35) 0x686a VSUBPS 0x240(%RSP),%YMM0,%YMM0 |
(35) 0x6873 VMOVAPS %YMM0,0xc0(%RSP) |
(35) 0x687c VEXTRACTF128 $0x1,%YMM0,%XMM0 |
(35) 0x6882 VMOVAPS %XMM0,0x20(%RSP) |
(35) 0x6888 VZEROUPPER |
(35) 0x688b CALL 7170 <@plt_start@+0x20> |
(35) 0x6890 VMOVAPS %XMM0,0xa0(%RSP) |
(35) 0x6899 VMOVSHDUP 0x20(%RSP),%XMM0 |
(35) 0x689f CALL 7170 <@plt_start@+0x20> |
(35) 0x68a4 VMOVAPS 0xa0(%RSP),%XMM1 |
(35) 0x68ad VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(35) 0x68b3 VMOVAPS %XMM0,0xa0(%RSP) |
(35) 0x68bc VPERMILPD $0x1,0x20(%RSP),%XMM0 |
(35) 0x68c4 CALL 7170 <@plt_start@+0x20> |
(35) 0x68c9 VMOVAPS 0xa0(%RSP),%XMM1 |
(35) 0x68d2 VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(35) 0x68d8 VMOVAPS %XMM0,0xa0(%RSP) |
(35) 0x68e1 VPERMILPS $-0x1,0x20(%RSP),%XMM0 |
(35) 0x68e9 CALL 7170 <@plt_start@+0x20> |
(35) 0x68ee VMOVAPS 0xa0(%RSP),%XMM1 |
(35) 0x68f7 VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(35) 0x68fd VMOVAPS %XMM0,0x20(%RSP) |
(35) 0x6903 VMOVAPS 0xc0(%RSP),%YMM0 |
(35) 0x690c VZEROUPPER |
(35) 0x690f CALL 7170 <@plt_start@+0x20> |
(35) 0x6914 VMOVAPS %XMM0,0xa0(%RSP) |
(35) 0x691d VMOVSHDUP 0xc0(%RSP),%XMM0 |
(35) 0x6926 CALL 7170 <@plt_start@+0x20> |
(35) 0x692b VMOVAPS 0xa0(%RSP),%XMM1 |
(35) 0x6934 VINSERTPS $0x10,%XMM0,%XMM1,%XMM0 |
(35) 0x693a VMOVAPS %XMM0,0xa0(%RSP) |
(35) 0x6943 VPERMILPD $0x1,0xc0(%RSP),%XMM0 |
(35) 0x694e CALL 7170 <@plt_start@+0x20> |
(35) 0x6953 VMOVAPS 0xa0(%RSP),%XMM1 |
(35) 0x695c VINSERTPS $0x20,%XMM0,%XMM1,%XMM0 |
(35) 0x6962 VMOVAPS %XMM0,0xa0(%RSP) |
(35) 0x696b VPERMILPS $-0x1,0xc0(%RSP),%XMM0 |
(35) 0x6976 CALL 7170 <@plt_start@+0x20> |
(35) 0x697b VMOVAPS 0xa0(%RSP),%XMM1 |
(35) 0x6984 MOV 0x60(%RSP),%RAX |
(35) 0x6989 MOV 0x1a0(%RSP),%RDX |
(35) 0x6991 MOV 0x18(%RSP),%RCX |
(35) 0x6996 VINSERTPS $0x30,%XMM0,%XMM1,%XMM0 |
(35) 0x699c VINSERTF128 $0x1,0x20(%RSP),%YMM0,%YMM0 |
(35) 0x69a4 VDIVPS 0x2c0(%RSP),%YMM0,%YMM0 |
(35) 0x69ad VMOVUPS %YMM0,(%RDX,%RAX,4) |
(35) 0x69b2 ADD $0x8,%RAX |
(35) 0x69b6 CMP %RAX,0x220(%RSP) |
(35) 0x69be JNE 6860 |
0x69c4 VMOVAPS 0x160(%RSP),%XMM1 |
0x69cd VMOVAPS 0x80(%RSP),%XMM2 |
0x69d6 CMP %R13,%R12 |
0x69d9 JE 6a33 |
0x69db TEST $0x4,%R12B |
0x69df JNE 66f2 |
0x69e5 NOPW %CS:(%RAX,%RAX,1) |
(39) 0x69f0 VMOVSS (%RCX,%R13,4),%XMM0 |
(39) 0x69f6 VSUBSS %XMM1,%XMM0,%XMM0 |
(39) 0x69fa VZEROUPPER |
(39) 0x69fd CALL 7170 <@plt_start@+0x20> |
(39) 0x6a02 VMOVAPS 0x80(%RSP),%XMM2 |
(39) 0x6a0b VMOVAPS 0x160(%RSP),%XMM1 |
(39) 0x6a14 MOV 0x1a0(%RSP),%RDX |
(39) 0x6a1c MOV 0x18(%RSP),%RCX |
(39) 0x6a21 VDIVSS %XMM2,%XMM0,%XMM0 |
(39) 0x6a25 VMOVSS %XMM0,(%RDX,%R13,4) |
(39) 0x6a2b INC %R13 |
(39) 0x6a2e CMP %R13,%R12 |
(39) 0x6a31 JNE 69f0 |
0x6a33 MOV 0x210(%RSP),%RCX |
0x6a3b MOV %R12,%R13 |
0x6a3e LEA 0x1(%RCX),%R12 |
0x6a42 CMP 0x58(%RSP),%R12 |
0x6a47 JAE 5da0 |
0x6a4d MOV %RCX,%RDI |
0x6a50 IMUL 0x298(%RSP),%RDI |
0x6a59 MOV 0x290(%RSP),%RDX |
0x6a61 LEA (,%RCX,4),%RAX |
0x6a69 XOR %ESI,%ESI |
0x6a6b SUB %RAX,%RDX |
0x6a6e ADD $0x4,%RDI |
0x6a72 ADD 0x100(%RSP),%RDI |
0x6a7a VZEROUPPER |
0x6a7d CALL 7180 <@plt_start@+0x30> |
0x6a82 VMOVAPS 0x80(%RSP),%XMM2 |
0x6a8b MOV 0x210(%RSP),%RCX |
0x6a93 JMP 5da0 |
/home/eoseret/llm-attention/attention.cpp: 43 - 284 |
-------------------------------------------------------------------------------- |
43: for (int row = 0; row < N; ++row) { |
44: const float *S_row = &S[row * N]; |
45: |
46: float max_val = -FLT_MAX; |
47: for (int idx = 0; idx <= row; ++idx) // vectorised |
48: if (S_row[idx] > max_val) max_val = S_row[idx]; |
49: |
50: float sum = 0.0f; |
51: #pragma clang loop vectorize(enable) |
52: for (int idx = 0; idx <= row; ++idx) // vectorised |
53: sum += expf(S_row[idx] - max_val); |
54: |
55: for (int idx = 0; idx <= row; ++idx) //vectorised |
56: P[row * N + idx] = expf(S_row[idx] - max_val) / sum; |
57: |
58: for (int idx = row + 1; idx < N; ++idx) |
59: P[row * N + idx] = 0.0f; |
60: |
61: D[row] = sum; |
[...] |
284: for (size_t r = 0; r < rept; r++) |
| Coverage (%) | Name | Source Location | Module |
|---|
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| Path / |
| Metric | Value |
|---|---|
| CQA speedup if no scalar integer | 2.14 |
| CQA speedup if FP arith vectorized | 2.01 |
| CQA speedup if fully vectorized | 7.70 |
| CQA speedup if no inter-iteration dependency | NA |
| CQA speedup if next bottleneck killed | 1.65 |
| Bottlenecks | micro-operation queue, |
| Function | main |
| Source | attention.cpp:43-44,attention.cpp:47-47,attention.cpp:52-52,attention.cpp:55-55,attention.cpp:58-61,attention.cpp:284-284 |
| Source loop unroll info | NA |
| Source loop unroll confidence level | NA |
| Unroll/vectorization loop type | NA |
| Unroll factor | NA |
| CQA cycles | 23.50 |
| CQA cycles if no scalar integer | 11.00 |
| CQA cycles if FP arith vectorized | 11.67 |
| CQA cycles if fully vectorized | 3.05 |
| Front-end cycles | 23.50 |
| P0 cycles | 11.67 |
| P1 cycles | 11.67 |
| P2 cycles | 11.67 |
| P3 cycles | 11.67 |
| P4 cycles | 11.67 |
| P5 cycles | 11.67 |
| P6 cycles | 14.25 |
| P7 cycles | 14.25 |
| P8 cycles | 14.25 |
| P9 cycles | 14.25 |
| P10 cycles | 8.50 |
| P11 cycles | 8.50 |
| P12 cycles | 8.50 |
| P13 cycles | 8.50 |
| P14 cycles | 6.00 |
| P15 cycles | 6.00 |
| DIV/SQRT cycles | 0.00 |
| Inter-iter dependencies cycles | NA |
| FE+BE cycles (UFS) | NA |
| Stall cycles (UFS) | NA |
| Nb insns | 195.00 |
| Nb uops | 188.00 |
| Nb loads | 34.00 |
| Nb stores | 23.00 |
| Nb stack references | 23.00 |
| FLOP/cycle | 1.62 |
| Nb FLOP add-sub | 38.00 |
| Nb FLOP mul | 0.00 |
| Nb FLOP fma | 0.00 |
| Nb FLOP div | 0.00 |
| Nb FLOP rcp | 0.00 |
| Nb FLOP sqrt | 0.00 |
| Nb FLOP rsqrt | 0.00 |
| Bytes/cycle | 32.34 |
| Bytes prefetched | 0.00 |
| Bytes loaded | 396.00 |
| Bytes stored | 364.00 |
| Stride 0 | NA |
| Stride 1 | NA |
| Stride n | NA |
| Stride unknown | NA |
| Stride indirect | NA |
| Vectorization ratio all | 44.44 |
| Vectorization ratio load | 45.83 |
| Vectorization ratio store | 47.83 |
| Vectorization ratio mul | NA |
| Vectorization ratio add_sub | 54.55 |
| Vectorization ratio fma | NA |
| Vectorization ratio div_sqrt | NA |
| Vectorization ratio other | 40.00 |
| Vector-efficiency ratio all | 20.09 |
| Vector-efficiency ratio load | 20.57 |
| Vector-efficiency ratio store | 24.73 |
| Vector-efficiency ratio mul | NA |
| Vector-efficiency ratio add_sub | 25.00 |
| Vector-efficiency ratio fma | NA |
| Vector-efficiency ratio div_sqrt | NA |
| Vector-efficiency ratio other | 17.79 |
| Metric | Value |
|---|---|
| CQA speedup if no scalar integer | 2.14 |
| CQA speedup if FP arith vectorized | 2.01 |
| CQA speedup if fully vectorized | 7.70 |
| CQA speedup if no inter-iteration dependency | NA |
| CQA speedup if next bottleneck killed | 1.65 |
| Bottlenecks | micro-operation queue, |
| Function | main |
| Source | attention.cpp:43-44,attention.cpp:47-47,attention.cpp:52-52,attention.cpp:55-55,attention.cpp:58-61,attention.cpp:284-284 |
| Source loop unroll info | NA |
| Source loop unroll confidence level | NA |
| Unroll/vectorization loop type | NA |
| Unroll factor | NA |
| CQA cycles | 23.50 |
| CQA cycles if no scalar integer | 11.00 |
| CQA cycles if FP arith vectorized | 11.67 |
| CQA cycles if fully vectorized | 3.05 |
| Front-end cycles | 23.50 |
| P0 cycles | 11.67 |
| P1 cycles | 11.67 |
| P2 cycles | 11.67 |
| P3 cycles | 11.67 |
| P4 cycles | 11.67 |
| P5 cycles | 11.67 |
| P6 cycles | 14.25 |
| P7 cycles | 14.25 |
| P8 cycles | 14.25 |
| P9 cycles | 14.25 |
| P10 cycles | 8.50 |
| P11 cycles | 8.50 |
| P12 cycles | 8.50 |
| P13 cycles | 8.50 |
| P14 cycles | 6.00 |
| P15 cycles | 6.00 |
| DIV/SQRT cycles | 0.00 |
| Inter-iter dependencies cycles | NA |
| FE+BE cycles (UFS) | NA |
| Stall cycles (UFS) | NA |
| Nb insns | 195.00 |
| Nb uops | 188.00 |
| Nb loads | 34.00 |
| Nb stores | 23.00 |
| Nb stack references | 23.00 |
| FLOP/cycle | 1.62 |
| Nb FLOP add-sub | 38.00 |
| Nb FLOP mul | 0.00 |
| Nb FLOP fma | 0.00 |
| Nb FLOP div | 0.00 |
| Nb FLOP rcp | 0.00 |
| Nb FLOP sqrt | 0.00 |
| Nb FLOP rsqrt | 0.00 |
| Bytes/cycle | 32.34 |
| Bytes prefetched | 0.00 |
| Bytes loaded | 396.00 |
| Bytes stored | 364.00 |
| Stride 0 | NA |
| Stride 1 | NA |
| Stride n | NA |
| Stride unknown | NA |
| Stride indirect | NA |
| Vectorization ratio all | 44.44 |
| Vectorization ratio load | 45.83 |
| Vectorization ratio store | 47.83 |
| Vectorization ratio mul | NA |
| Vectorization ratio add_sub | 54.55 |
| Vectorization ratio fma | NA |
| Vectorization ratio div_sqrt | NA |
| Vectorization ratio other | 40.00 |
| Vector-efficiency ratio all | 20.09 |
| Vector-efficiency ratio load | 20.57 |
| Vector-efficiency ratio store | 24.73 |
| Vector-efficiency ratio mul | NA |
| Vector-efficiency ratio add_sub | 25.00 |
| Vector-efficiency ratio fma | NA |
| Vector-efficiency ratio div_sqrt | NA |
| Vector-efficiency ratio other | 17.79 |
| Path / |
| Function | main |
| Source file and lines | attention.cpp:43-284 |
| Module | attention-avx512 |
| nb instructions | 195 |
| nb uops | 188 |
| loop length | 1050 |
| used x86 registers | 10 |
| used mmx registers | 0 |
| used xmm registers | 3 |
| used ymm registers | 4 |
| used zmm registers | 0 |
| nb stack references | 23 |
| micro-operation queue | 23.50 cycles |
| front end | 23.50 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| uops | 11.67 | 11.67 | 11.67 | 11.67 | 11.67 | 11.67 | 14.25 | 14.25 | 14.25 | 14.25 | 8.50 | 8.50 | 8.50 | 8.50 | 6.00 | 6.00 |
| cycles | 11.67 | 11.67 | 11.67 | 11.67 | 11.67 | 11.67 | 14.25 | 14.25 | 14.25 | 14.25 | 8.50 | 8.50 | 8.50 | 8.50 | 6.00 | 6.00 |
| Cycles executing div or sqrt instructions | NA |
| Front-end | 23.50 |
| Dispatch | 14.25 |
| Overall L1 | 23.50 |
| all | 1% |
| load | 0% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 3% |
| all | 77% |
| load | 78% |
| store | 91% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 75% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 71% |
| all | 44% |
| load | 45% |
| store | 47% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 54% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 40% |
| all | 12% |
| load | 12% |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 12% |
| all | 26% |
| load | 26% |
| store | 35% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 29% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 22% |
| all | 20% |
| load | 20% |
| store | 24% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 25% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 17% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| MOV 0x190(%RSP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV 0x1d0(%RSP),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| MOV 0x218(%RSP),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| INC %R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %R12,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| VMOVSS %XMM2,(%RAX,%RCX,4) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 1 | 0.50 | scal (6.3%) |
| MOV 0x1a0(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| ADD %RDX,%RSI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| ADD %RDX,0x18(%RSP) | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (12.5%) |
| ADD %RDX,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV %RCX,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| CMP 0x58(%RSP),%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
| MOV %R13,%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| MOV 0x158(%RSP),%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| JE 5d50 <main+0x20b0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV %R12,%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| MOV %R12,%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| AND $-0x4,%R8 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| AND $-0x20,%R9 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %RAX,0x1a0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| CMP $0x4,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 5e20 <main+0x2180> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VMOVSS -0x4f63(%RIP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (6.3%) |
| MOV 0x18(%RSP),%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| JMP 5f10 <main+0x2270> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| MOV 0x18(%RSP),%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| CMP $0x20,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 5e40 <main+0x21a0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VMOVSS -0x4f83(%RIP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (6.3%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| JMP 5ec2 <main+0x2222> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| VBROADCASTSS -0x4f99(%RIP),%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (6.3%) |
| MOV $0x7ffffffffffffffc,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| XOR %ECX,%ECX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| ADD $-0x1c,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| AND %R12,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VMOVAPS %YMM0,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
| VMOVAPS %YMM0,%YMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
| VMOVAPS %YMM0,%YMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
| NOPL (%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VMAXPS %YMM1,%YMM0,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VMAXPS %YMM3,%YMM2,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VMAXPS %YMM1,%YMM0,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VEXTRACTF128 $0x1,%YMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| VMAXPS %XMM1,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VSHUFPD $0x1,%XMM0,%XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
| VMAXPS %XMM1,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VMOVSHDUP %XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (12.5%) |
| VMAXSS %XMM1,%XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | scal (6.3%) |
| CMP %RAX,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JE 5f1f <main+0x227f> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| TEST $0x1c,%R12B | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| JE 5f10 <main+0x2270> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV $0x7ffffffffffffffc,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| VBROADCASTSS %XMM1,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (6.3%) |
| MOV %RAX,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| MOV %R12,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| AND %RDX,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| NOPW (%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VSHUFPD $0x1,%XMM0,%XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
| VMAXPS %XMM1,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VMOVSHDUP %XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (12.5%) |
| VMAXSS %XMM1,%XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | scal (6.3%) |
| JMP 5f1a <main+0x227a> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| VMOVAPS %XMM1,0x160(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (25.0%) |
| MOV %RSI,0x218(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV %RDI,0x210(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV %R8,0x180(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV %R13,0x18(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| CMP $0x4,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 5f60 <main+0x22c0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VXORPS %XMM2,%XMM2,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| JMP 665a <main+0x29ba> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| CMP $0x20,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 5f80 <main+0x22e0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VXORPS %XMM2,%XMM2,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| JMP 654f <main+0x28af> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| MOV $0x7ffffffffffffffc,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VBROADCASTSS %XMM1,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 1 | 0.50 | scal (6.3%) |
| VXORPS %XMM1,%XMM1,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
| MOV %R9,0x2a0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| ADD $-0x1c,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| AND %R12,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VMOVAPS %YMM0,0x2e0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| VXORPS %XMM0,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
| MOV %RAX,0x1d8(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| VMOVAPS %YMM0,0x220(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| VMOVAPS %YMM0,0x240(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| VMOVAPS %YMM0,0x2c0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VMOVAPS 0x240(%RSP),%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
| MOV 0x1d8(%RSP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| VADDPS 0x220(%RSP),%YMM0,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VADDPS 0x2c0(%RSP),%YMM0,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VADDPS %YMM0,%YMM1,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VEXTRACTF128 $0x1,%YMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| VADDPS %XMM1,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VSHUFPD $0x1,%XMM0,%XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
| VADDPS %XMM1,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VMOVSHDUP %XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (12.5%) |
| VADDSS %XMM1,%XMM0,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | scal (6.3%) |
| CMP %RAX,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JNE 653c <main+0x289c> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VMOVAPS 0x160(%RSP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| JMP 66be <main+0x2a1e> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| VMOVAPS 0x160(%RSP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| TEST $0x1c,%R12B | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| JE 665a <main+0x29ba> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VXORPS %XMM0,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
| VBLENDPS $0x1,%XMM2,%XMM0,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
| VBROADCASTSS %XMM1,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (6.3%) |
| MOV $0x7ffffffffffffffc,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %R12,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| MOV %RAX,%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| AND %RDX,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV %RCX,0x1d8(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| VMOVAPS %XMM0,0x60(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (25.0%) |
| NOP | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.09 | N/A |
| VSHUFPD $0x1,%XMM2,%XMM2,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
| MOV 0x1d8(%RSP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV 0x18(%RSP),%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| VADDPS %XMM0,%XMM2,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VMOVSHDUP %XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (12.5%) |
| VADDSS %XMM1,%XMM0,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | scal (6.3%) |
| VMOVAPS 0x160(%RSP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| CMP %RAX,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JE 66be <main+0x2a1e> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV %R12,0x1e0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VMOVAPS %XMM2,0x80(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (25.0%) |
| CMP $0x4,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 66e0 <main+0x2a40> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV %R13,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| XOR %R13D,%R13D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (6.3%) |
| JMP 69f0 <main+0x2d50> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| CMP $0x8,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 6810 <main+0x2b70> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV 0x18(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| XOR %R13D,%R13D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (6.3%) |
| VBROADCASTSS %XMM1,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (6.3%) |
| VBROADCASTSS %XMM2,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (6.3%) |
| MOV $0x7ffffffffffffffc,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV %R13,%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| MOV %R12,%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| AND %RAX,%R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| VMOVAPS %XMM0,0xa0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (25.0%) |
| VMOVAPS %XMM1,0x220(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (25.0%) |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VMOVAPS 0x160(%RSP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| VMOVAPS 0x80(%RSP),%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| CMP %R13,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JNE 69f0 <main+0x2d50> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| JMP 6a33 <main+0x2d93> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| MOV 0x18(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV %R12,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| AND $-0x8,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VBROADCASTSS %XMM1,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 1 | 0.50 | scal (6.3%) |
| VBROADCASTSS %XMM2,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 1 | 0.50 | scal (6.3%) |
| MOV %RAX,0x220(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV $0x7ffffffffffffffc,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| LEA -0x4(%RAX),%R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VMOVAPS %YMM0,0x240(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| VMOVAPS %YMM1,0x2c0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| AND %R12,%R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VMOVAPS 0x160(%RSP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| VMOVAPS 0x80(%RSP),%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| CMP %R13,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JE 6a33 <main+0x2d93> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| TEST $0x4,%R12B | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| JNE 66f2 <main+0x2a52> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| MOV 0x210(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV %R12,%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| LEA 0x1(%RCX),%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| CMP 0x58(%RSP),%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
| JAE 5da0 <main+0x2100> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV %RCX,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| IMUL 0x298(%RSP),%RDI | 1 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
| MOV 0x290(%RSP),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| LEA (,%RCX,4),%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| XOR %ESI,%ESI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (6.3%) |
| SUB %RAX,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| ADD $0x4,%RDI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| ADD 0x100(%RSP),%RDI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
| VZEROUPPER | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | vect (25.0%) |
| CALL 7180 <@plt_start@+0x30> | 2 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| VMOVAPS 0x80(%RSP),%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| MOV 0x210(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| JMP 5da0 <main+0x2100> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| Function | main |
| Source file and lines | attention.cpp:43-284 |
| Module | attention-avx512 |
| nb instructions | 195 |
| nb uops | 188 |
| loop length | 1050 |
| used x86 registers | 10 |
| used mmx registers | 0 |
| used xmm registers | 3 |
| used ymm registers | 4 |
| used zmm registers | 0 |
| nb stack references | 23 |
| micro-operation queue | 23.50 cycles |
| front end | 23.50 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| uops | 11.67 | 11.67 | 11.67 | 11.67 | 11.67 | 11.67 | 14.25 | 14.25 | 14.25 | 14.25 | 8.50 | 8.50 | 8.50 | 8.50 | 6.00 | 6.00 |
| cycles | 11.67 | 11.67 | 11.67 | 11.67 | 11.67 | 11.67 | 14.25 | 14.25 | 14.25 | 14.25 | 8.50 | 8.50 | 8.50 | 8.50 | 6.00 | 6.00 |
| Cycles executing div or sqrt instructions | NA |
| Front-end | 23.50 |
| Dispatch | 14.25 |
| Overall L1 | 23.50 |
| all | 1% |
| load | 0% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 3% |
| all | 77% |
| load | 78% |
| store | 91% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 75% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 71% |
| all | 44% |
| load | 45% |
| store | 47% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 54% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 40% |
| all | 12% |
| load | 12% |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 12% |
| all | 26% |
| load | 26% |
| store | 35% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 29% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 22% |
| all | 20% |
| load | 20% |
| store | 24% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 25% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 17% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| MOV 0x190(%RSP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV 0x1d0(%RSP),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| MOV 0x218(%RSP),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| INC %R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %R12,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| VMOVSS %XMM2,(%RAX,%RCX,4) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 1 | 0.50 | scal (6.3%) |
| MOV 0x1a0(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| ADD %RDX,%RSI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| ADD %RDX,0x18(%RSP) | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (12.5%) |
| ADD %RDX,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV %RCX,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| CMP 0x58(%RSP),%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
| MOV %R13,%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| MOV 0x158(%RSP),%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| JE 5d50 <main+0x20b0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV %R12,%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| MOV %R12,%R9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| AND $-0x4,%R8 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| AND $-0x20,%R9 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %RAX,0x1a0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| CMP $0x4,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 5e20 <main+0x2180> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VMOVSS -0x4f63(%RIP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (6.3%) |
| MOV 0x18(%RSP),%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| JMP 5f10 <main+0x2270> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| MOV 0x18(%RSP),%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| CMP $0x20,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 5e40 <main+0x21a0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VMOVSS -0x4f83(%RIP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (6.3%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| JMP 5ec2 <main+0x2222> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| VBROADCASTSS -0x4f99(%RIP),%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (6.3%) |
| MOV $0x7ffffffffffffffc,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| XOR %ECX,%ECX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| ADD $-0x1c,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| AND %R12,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VMOVAPS %YMM0,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
| VMOVAPS %YMM0,%YMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
| VMOVAPS %YMM0,%YMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
| NOPL (%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VMAXPS %YMM1,%YMM0,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VMAXPS %YMM3,%YMM2,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VMAXPS %YMM1,%YMM0,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VEXTRACTF128 $0x1,%YMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| VMAXPS %XMM1,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VSHUFPD $0x1,%XMM0,%XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
| VMAXPS %XMM1,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VMOVSHDUP %XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (12.5%) |
| VMAXSS %XMM1,%XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | scal (6.3%) |
| CMP %RAX,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JE 5f1f <main+0x227f> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| TEST $0x1c,%R12B | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| JE 5f10 <main+0x2270> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV $0x7ffffffffffffffc,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| VBROADCASTSS %XMM1,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (6.3%) |
| MOV %RAX,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| MOV %R12,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| AND %RDX,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| NOPW (%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VSHUFPD $0x1,%XMM0,%XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
| VMAXPS %XMM1,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VMOVSHDUP %XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (12.5%) |
| VMAXSS %XMM1,%XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | scal (6.3%) |
| JMP 5f1a <main+0x227a> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| VMOVAPS %XMM1,0x160(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (25.0%) |
| MOV %RSI,0x218(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV %RDI,0x210(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV %R8,0x180(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV %R13,0x18(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| CMP $0x4,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 5f60 <main+0x22c0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VXORPS %XMM2,%XMM2,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| JMP 665a <main+0x29ba> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| CMP $0x20,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 5f80 <main+0x22e0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VXORPS %XMM2,%XMM2,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| JMP 654f <main+0x28af> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| MOV $0x7ffffffffffffffc,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VBROADCASTSS %XMM1,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 1 | 0.50 | scal (6.3%) |
| VXORPS %XMM1,%XMM1,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
| MOV %R9,0x2a0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| ADD $-0x1c,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| AND %R12,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VMOVAPS %YMM0,0x2e0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| VXORPS %XMM0,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
| MOV %RAX,0x1d8(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| VMOVAPS %YMM0,0x220(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| VMOVAPS %YMM0,0x240(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| VMOVAPS %YMM0,0x2c0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VMOVAPS 0x240(%RSP),%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
| MOV 0x1d8(%RSP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| VADDPS 0x220(%RSP),%YMM0,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VADDPS 0x2c0(%RSP),%YMM0,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VADDPS %YMM0,%YMM1,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (50.0%) |
| VEXTRACTF128 $0x1,%YMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| VADDPS %XMM1,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VSHUFPD $0x1,%XMM0,%XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
| VADDPS %XMM1,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VMOVSHDUP %XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (12.5%) |
| VADDSS %XMM1,%XMM0,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | scal (6.3%) |
| CMP %RAX,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JNE 653c <main+0x289c> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VMOVAPS 0x160(%RSP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| JMP 66be <main+0x2a1e> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| VMOVAPS 0x160(%RSP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| TEST $0x1c,%R12B | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| JE 665a <main+0x29ba> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| VXORPS %XMM0,%XMM0,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
| VBLENDPS $0x1,%XMM2,%XMM0,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
| VBROADCASTSS %XMM1,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (6.3%) |
| MOV $0x7ffffffffffffffc,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %R12,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| MOV %RAX,%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| AND %RDX,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV %RCX,0x1d8(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| VMOVAPS %XMM0,0x60(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (25.0%) |
| NOP | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.09 | N/A |
| VSHUFPD $0x1,%XMM2,%XMM2,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
| MOV 0x1d8(%RSP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV 0x18(%RSP),%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| VADDPS %XMM0,%XMM2,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | vect (25.0%) |
| VMOVSHDUP %XMM0,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (12.5%) |
| VADDSS %XMM1,%XMM0,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 2 | 0.50 | scal (6.3%) |
| VMOVAPS 0x160(%RSP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| CMP %RAX,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JE 66be <main+0x2a1e> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV %R12,0x1e0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VMOVAPS %XMM2,0x80(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (25.0%) |
| CMP $0x4,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 66e0 <main+0x2a40> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV %R13,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| XOR %R13D,%R13D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (6.3%) |
| JMP 69f0 <main+0x2d50> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| CMP $0x8,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JAE 6810 <main+0x2b70> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV 0x18(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| XOR %R13D,%R13D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (6.3%) |
| VBROADCASTSS %XMM1,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (6.3%) |
| VBROADCASTSS %XMM2,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (6.3%) |
| MOV $0x7ffffffffffffffc,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV %R13,%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| MOV %R12,%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| AND %RAX,%R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| VMOVAPS %XMM0,0xa0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (25.0%) |
| VMOVAPS %XMM1,0x220(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (25.0%) |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VMOVAPS 0x160(%RSP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| VMOVAPS 0x80(%RSP),%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| CMP %R13,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JNE 69f0 <main+0x2d50> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| JMP 6a33 <main+0x2d93> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| MOV 0x18(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV %R12,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| AND $-0x8,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VBROADCASTSS %XMM1,%YMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 1 | 0.50 | scal (6.3%) |
| VBROADCASTSS %XMM2,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 1 | 0.50 | scal (6.3%) |
| MOV %RAX,0x220(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV $0x7ffffffffffffffc,%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| LEA -0x4(%RAX),%R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VMOVAPS %YMM0,0x240(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| VMOVAPS %YMM1,0x2c0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 4 | 0.50 | vect (50.0%) |
| XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| AND %R12,%R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| VMOVAPS 0x160(%RSP),%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| VMOVAPS 0x80(%RSP),%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| CMP %R13,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| JE 6a33 <main+0x2d93> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| TEST $0x4,%R12B | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| JNE 66f2 <main+0x2a52> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| MOV 0x210(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV %R12,%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| LEA 0x1(%RCX),%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| CMP 0x58(%RSP),%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
| JAE 5da0 <main+0x2100> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV %RCX,%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| IMUL 0x298(%RSP),%RDI | 1 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
| MOV 0x290(%RSP),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| LEA (,%RCX,4),%RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| XOR %ESI,%ESI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (6.3%) |
| SUB %RAX,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| ADD $0x4,%RDI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| ADD 0x100(%RSP),%RDI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
| VZEROUPPER | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | vect (25.0%) |
| CALL 7180 <@plt_start@+0x30> | 2 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| VMOVAPS 0x80(%RSP),%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
| MOV 0x210(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| JMP 5da0 <main+0x2100> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
