| Loop Id: 89 | Module: attention-clang-znver5-256 | Source: random.tcc:401-3367 [...] | Coverage: 0.28% |
|---|
| Loop Id: 89 | Module: attention-clang-znver5-256 | Source: random.tcc:401-3367 [...] | Coverage: 0.28% |
|---|
0x4870 MOV %R13,%RCX |
0x4873 INC %R13 |
0x4876 MOV $0x200b,%EDX |
0x487b MOV %R13,0x1818(%RSP) |
0x4883 MOV 0x498(%RSP,%RCX,8),%RCX |
0x488b BEXTR %RDX,%RCX,%RDX |
0x4890 XOR %RCX,%RDX |
0x4893 MOV %EDX,%ECX |
0x4895 SAL $0x7,%ECX |
0x4898 AND $-0x62d3a980,%ECX |
0x489e XOR %RDX,%RCX |
0x48a1 MOV %ECX,%EDX |
0x48a3 SAL $0xf,%EDX |
0x48a6 AND $-0x103a0000,%EDX |
0x48ac XOR %RCX,%RDX |
0x48af MOV %RDX,%RCX |
0x48b2 SHR $0x12,%RCX |
0x48b6 XOR %RDX,%RCX |
0x48b9 DEC %RAX |
0x48bc VCVTUSI2SS %RCX,%XMM15,%XMM2 |
0x48c2 VFMADD231SS %XMM2,%XMM1,%XMM0 |
0x48c7 VMULSS 0x3739(%RIP),%XMM1,%XMM1 |
0x48cf JE 4ce0 |
0x48d5 CMP $0x270,%R13 |
0x48dc JB 4870 |
0x48de VPBROADCASTQ 0x3759(%RIP),%YMM14 |
0x48e7 VPBROADCASTQ 0x3758(%RIP),%YMM15 |
0x48f0 VPBROADCASTQ 0x3756(%RIP),%YMM16 |
0x48fa VPBROADCASTQ 0x3754(%RIP),%YMM17 |
0x4904 VPBROADCASTQ %RBX,%YMM2 |
0x490a XOR %ECX,%ECX |
0x490c NOPL (%RAX) |
(90) 0x4910 VMOVDQU 0x4a0(%RSP,%RCX,8),%YMM3 |
(90) 0x4919 VMOVDQU 0x4c0(%RSP,%RCX,8),%YMM4 |
(90) 0x4922 VMOVDQU 0x4e0(%RSP,%RCX,8),%YMM5 |
(90) 0x492b VALIGNQ $0x3,%YMM2,%YMM3,%YMM6 |
(90) 0x4932 VMOVDQU 0x500(%RSP,%RCX,8),%YMM2 |
(90) 0x493b VALIGNQ $0x3,%YMM3,%YMM4,%YMM7 |
(90) 0x4942 VALIGNQ $0x3,%YMM4,%YMM5,%YMM8 |
(90) 0x4949 VPAND %YMM3,%YMM15,%YMM10 |
(90) 0x494d VPAND %YMM4,%YMM15,%YMM11 |
(90) 0x4951 VPAND %YMM5,%YMM15,%YMM12 |
(90) 0x4955 VPTESTMQ %YMM16,%YMM3,%K1 |
(90) 0x495b VPTESTMQ %YMM16,%YMM4,%K2 |
(90) 0x4961 VPTESTMQ %YMM16,%YMM5,%K3 |
(90) 0x4967 VPTERNLOGQ $-0x8,%YMM14,%YMM6,%YMM10 |
(90) 0x496e VPTERNLOGQ $-0x8,%YMM14,%YMM7,%YMM11 |
(90) 0x4975 VPTERNLOGQ $-0x8,%YMM14,%YMM8,%YMM12 |
(90) 0x497c VPSRLQ $0x1,%YMM10,%YMM6 |
(90) 0x4982 VPSRLQ $0x1,%YMM11,%YMM7 |
(90) 0x4988 VPSRLQ $0x1,%YMM12,%YMM8 |
(90) 0x498e VPXOR 0x1100(%RSP,%RCX,8),%YMM6,%YMM6 |
(90) 0x4997 VPXOR 0x1120(%RSP,%RCX,8),%YMM7,%YMM7 |
(90) 0x49a0 VPXOR 0x1140(%RSP,%RCX,8),%YMM8,%YMM8 |
(90) 0x49a9 VALIGNQ $0x3,%YMM5,%YMM2,%YMM9 |
(90) 0x49b0 VPAND %YMM2,%YMM15,%YMM13 |
(90) 0x49b4 VPTESTMQ %YMM16,%YMM2,%K4 |
(90) 0x49ba VPTERNLOGQ $-0x8,%YMM14,%YMM9,%YMM13 |
(90) 0x49c1 VPSRLQ $0x1,%YMM13,%YMM9 |
(90) 0x49c7 VPXOR 0x1160(%RSP,%RCX,8),%YMM9,%YMM9 |
(90) 0x49d0 VPXORQ %YMM17,%YMM6,%YMM6{%K1} |
(90) 0x49d6 VPXORQ %YMM17,%YMM7,%YMM7{%K2} |
(90) 0x49dc VPXORQ %YMM17,%YMM8,%YMM8{%K3} |
(90) 0x49e2 VMOVDQU %YMM6,0x498(%RSP,%RCX,8) |
(90) 0x49eb VMOVDQU %YMM7,0x4b8(%RSP,%RCX,8) |
(90) 0x49f4 VMOVDQU %YMM8,0x4d8(%RSP,%RCX,8) |
(90) 0x49fd VPXORQ %YMM17,%YMM9,%YMM9{%K4} |
(90) 0x4a03 VMOVDQU %YMM9,0x4f8(%RSP,%RCX,8) |
(90) 0x4a0c ADD $0x10,%RCX |
(90) 0x4a10 CMP $0xe0,%RCX |
(90) 0x4a17 JNE 4910 |
0x4a1d MOV 0xba0(%RSP),%RDX |
0x4a25 VEXTRACTI128 $0x1,%YMM2,%XMM2 |
0x4a2b MOV 0xba8(%RSP),%RCX |
0x4a33 VPBROADCASTQ 0x3604(%RIP),%XMM8 |
0x4a3c VPBROADCASTQ 0x3603(%RIP),%XMM9 |
0x4a45 VPBROADCASTQ 0x3602(%RIP),%XMM10 |
0x4a4e VPBROADCASTQ 0x3601(%RIP),%XMM11 |
0x4a57 VPEXTRQ $0x1,%XMM2,%RSI |
0x4a5d AND $-0x80000000,%RSI |
0x4a64 MOV %EDX,%EDI |
0x4a66 AND $0x7ffffffe,%EDI |
0x4a6c OR %RSI,%RDI |
0x4a6f MOV %EDX,%ESI |
0x4a71 AND $0x1,%ESI |
0x4a74 AND $-0x80000000,%RDX |
0x4a7b SHR $0x1,%RDI |
0x4a7e XOR 0x1800(%RSP),%RDI |
0x4a86 NEG %ESI |
0x4a88 AND %R12D,%ESI |
0x4a8b XOR %RDI,%RSI |
0x4a8e MOV %RSI,0xb98(%RSP) |
0x4a96 MOV %ECX,%ESI |
0x4a98 AND $0x7ffffffe,%ESI |
0x4a9e OR %RDX,%RSI |
0x4aa1 MOV %ECX,%EDX |
0x4aa3 AND $0x1,%EDX |
0x4aa6 AND $-0x80000000,%RCX |
0x4aad SHR $0x1,%RSI |
0x4ab0 XOR 0x1808(%RSP),%RSI |
0x4ab8 NEG %EDX |
0x4aba AND %R12D,%EDX |
0x4abd XOR %RSI,%RDX |
0x4ac0 MOV %RDX,0xba0(%RSP) |
0x4ac8 MOV 0xbb0(%RSP),%RDX |
0x4ad0 MOV %EDX,%ESI |
0x4ad2 VPBROADCASTQ %RDX,%XMM2 |
0x4ad8 AND $0x7ffffffe,%EDX |
0x4ade AND $0x1,%ESI |
0x4ae1 OR %RCX,%RDX |
0x4ae4 NEG %ESI |
0x4ae6 MOV $0xee,%ECX |
0x4aeb SHR $0x1,%RDX |
0x4aee XOR 0x1810(%RSP),%RDX |
0x4af6 AND %R12D,%ESI |
0x4af9 XOR %RDX,%RSI |
0x4afc MOV %RSI,0xba8(%RSP) |
0x4b04 NOPW %CS:(%RAX,%RAX,1) |
(91) 0x4b10 VMOVDQU 0x448(%RSP,%RCX,8),%XMM3 |
(91) 0x4b19 VMOVDQU 0x458(%RSP,%RCX,8),%XMM4 |
(91) 0x4b22 VMOVDQU 0x468(%RSP,%RCX,8),%XMM5 |
(91) 0x4b2b VMOVDQU 0x478(%RSP,%RCX,8),%XMM6 |
(91) 0x4b34 VPALIGNR $0x8,%XMM2,%XMM3,%XMM2 |
(91) 0x4b3a VPAND %XMM3,%XMM9,%XMM7 |
(91) 0x4b3e VPTESTMQ %XMM10,%XMM3,%K1 |
(91) 0x4b44 VPTERNLOGQ $-0x8,%XMM8,%XMM2,%XMM7 |
(91) 0x4b4b VPSRLQ $0x1,%XMM7,%XMM2 |
(91) 0x4b50 VPXOR -0x2d8(%RSP,%RCX,8),%XMM2,%XMM2 |
(91) 0x4b59 VPXORQ %XMM11,%XMM2,%XMM2{%K1} |
(91) 0x4b5f VPTESTMQ %XMM10,%XMM4,%K1 |
(91) 0x4b65 VMOVDQU %XMM2,0x440(%RSP,%RCX,8) |
(91) 0x4b6e VPALIGNR $0x8,%XMM3,%XMM4,%XMM2 |
(91) 0x4b74 VPAND %XMM4,%XMM9,%XMM3 |
(91) 0x4b78 VPTERNLOGQ $-0x8,%XMM8,%XMM2,%XMM3 |
(91) 0x4b7f VPSRLQ $0x1,%XMM3,%XMM2 |
(91) 0x4b84 VPXOR -0x2c8(%RSP,%RCX,8),%XMM2,%XMM2 |
(91) 0x4b8d VPAND %XMM5,%XMM9,%XMM3 |
(91) 0x4b91 VPXORQ %XMM11,%XMM2,%XMM2{%K1} |
(91) 0x4b97 VPTESTMQ %XMM10,%XMM5,%K1 |
(91) 0x4b9d VMOVDQU %XMM2,0x450(%RSP,%RCX,8) |
(91) 0x4ba6 VPALIGNR $0x8,%XMM4,%XMM5,%XMM2 |
(91) 0x4bac VPTERNLOGQ $-0x8,%XMM8,%XMM2,%XMM3 |
(91) 0x4bb3 VPSRLQ $0x1,%XMM3,%XMM2 |
(91) 0x4bb8 VPXOR -0x2b8(%RSP,%RCX,8),%XMM2,%XMM2 |
(91) 0x4bc1 VPAND %XMM6,%XMM9,%XMM3 |
(91) 0x4bc5 VPXORQ %XMM11,%XMM2,%XMM2{%K1} |
(91) 0x4bcb VPTESTMQ %XMM10,%XMM6,%K1 |
(91) 0x4bd1 VMOVDQU %XMM2,0x460(%RSP,%RCX,8) |
(91) 0x4bda VPALIGNR $0x8,%XMM5,%XMM6,%XMM2 |
(91) 0x4be0 VPTERNLOGQ $-0x8,%XMM8,%XMM2,%XMM3 |
(91) 0x4be7 VPSRLQ $0x1,%XMM3,%XMM2 |
(91) 0x4bec VPXOR -0x2a8(%RSP,%RCX,8),%XMM2,%XMM2 |
(91) 0x4bf5 VPXORQ %XMM11,%XMM2,%XMM2{%K1} |
(91) 0x4bfb VMOVDQU %XMM2,0x470(%RSP,%RCX,8) |
(91) 0x4c04 VMOVDQU 0x488(%RSP,%RCX,8),%XMM3 |
(91) 0x4c0d VPALIGNR $0x8,%XMM6,%XMM3,%XMM2 |
(91) 0x4c13 VPAND %XMM3,%XMM9,%XMM4 |
(91) 0x4c17 VPTESTMQ %XMM10,%XMM3,%K1 |
(91) 0x4c1d VPTERNLOGQ $-0x8,%XMM8,%XMM2,%XMM4 |
(91) 0x4c24 VPSRLQ $0x1,%XMM4,%XMM2 |
(91) 0x4c29 VPXOR -0x298(%RSP,%RCX,8),%XMM2,%XMM2 |
(91) 0x4c32 VPXORQ %XMM11,%XMM2,%XMM2{%K1} |
(91) 0x4c38 VMOVDQU %XMM2,0x480(%RSP,%RCX,8) |
(91) 0x4c41 VMOVDQU 0x498(%RSP,%RCX,8),%XMM2 |
(91) 0x4c4a VPALIGNR $0x8,%XMM3,%XMM2,%XMM3 |
(91) 0x4c50 VPAND %XMM2,%XMM9,%XMM4 |
(91) 0x4c54 VPTESTMQ %XMM10,%XMM2,%K1 |
(91) 0x4c5a VPTERNLOGQ $-0x8,%XMM8,%XMM3,%XMM4 |
(91) 0x4c61 VPSRLQ $0x1,%XMM4,%XMM3 |
(91) 0x4c66 VPXOR -0x288(%RSP,%RCX,8),%XMM3,%XMM3 |
(91) 0x4c6f VPXORQ %XMM11,%XMM3,%XMM3{%K1} |
(91) 0x4c75 VMOVDQU %XMM3,0x490(%RSP,%RCX,8) |
(91) 0x4c7e ADD $0xc,%RCX |
(91) 0x4c82 CMP $0x27a,%RCX |
(91) 0x4c89 JNE 4b10 |
0x4c8f MOV 0x1810(%RSP),%RCX |
0x4c97 MOV 0x498(%RSP),%RBX |
0x4c9f XOR %R13D,%R13D |
0x4ca2 MOV %EBX,%EDX |
0x4ca4 AND %R14,%RCX |
0x4ca7 AND $0x7ffffffe,%EDX |
0x4cad OR %RCX,%RDX |
0x4cb0 MOV %EBX,%ECX |
0x4cb2 AND $0x1,%ECX |
0x4cb5 SHR $0x1,%RDX |
0x4cb8 XOR 0x10f8(%RSP),%RDX |
0x4cc0 NEG %ECX |
0x4cc2 AND %R12D,%ECX |
0x4cc5 XOR %RDX,%RCX |
0x4cc8 MOV %RCX,0x1810(%RSP) |
0x4cd0 JMP 4870 |
/usr/lib/gcc/x86_64-redhat-linux/11/../../../../include/c++/11/bits/random.tcc: 401 - 3367 |
-------------------------------------------------------------------------------- |
401: for (size_t __k = 0; __k < (__n - __m); ++__k) |
402: { |
403: _UIntType __y = ((_M_x[__k] & __upper_mask) |
404: | (_M_x[__k + 1] & __lower_mask)); |
405: _M_x[__k] = (_M_x[__k + __m] ^ (__y >> 1) |
406: ^ ((__y & 0x01) ? __a : 0)); |
407: } |
408: |
409: for (size_t __k = (__n - __m); __k < (__n - 1); ++__k) |
410: { |
411: _UIntType __y = ((_M_x[__k] & __upper_mask) |
412: | (_M_x[__k + 1] & __lower_mask)); |
413: _M_x[__k] = (_M_x[__k + (__m - __n)] ^ (__y >> 1) |
414: ^ ((__y & 0x01) ? __a : 0)); |
415: } |
416: |
417: _UIntType __y = ((_M_x[__n - 1] & __upper_mask) |
418: | (_M_x[0] & __lower_mask)); |
419: _M_x[__n - 1] = (_M_x[__m - 1] ^ (__y >> 1) |
420: ^ ((__y & 0x01) ? __a : 0)); |
[...] |
455: if (_M_p >= state_size) |
456: _M_gen_rand(); |
457: |
458: // Calculate o(x(i)). |
459: result_type __z = _M_x[_M_p++]; |
460: __z ^= (__z >> __u) & __d; |
461: __z ^= (__z << __s) & __b; |
462: __z ^= (__z << __t) & __c; |
463: __z ^= (__z >> __l); |
[...] |
3364: for (size_t __k = __m; __k != 0; --__k) |
3365: { |
3366: __sum += _RealType(__urng() - __urng.min()) * __tmp; |
3367: __tmp *= __r; |
| Coverage (%) | Name | Source Location | Module |
|---|
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| Path / |
| Metric | Value |
|---|---|
| CQA speedup if no scalar integer | 2.05 |
| CQA speedup if FP arith vectorized | 2.04 |
| CQA speedup if fully vectorized | 12.09 |
| CQA speedup if no inter-iteration dependency | NA |
| CQA speedup if next bottleneck killed | 1.35 |
| Bottlenecks | |
| Function | main |
| Source | random.tcc:401-406,random.tcc:409-409,random.tcc:413-413,random.tcc:417-420,random.tcc:455-455,random.tcc:459-463,random.tcc:3364-3367 |
| Source loop unroll info | NA |
| Source loop unroll confidence level | NA |
| Unroll/vectorization loop type | NA |
| Unroll factor | NA |
| CQA cycles | 8.19 |
| CQA cycles if no scalar integer | 4.00 |
| CQA cycles if FP arith vectorized | 4.01 |
| CQA cycles if fully vectorized | 0.68 |
| Front-end cycles | 7.75 |
| P0 cycles | 5.75 |
| P1 cycles | 5.75 |
| P2 cycles | 5.75 |
| P3 cycles | 5.75 |
| P4 cycles | 5.75 |
| P5 cycles | 5.75 |
| P6 cycles | 3.38 |
| P7 cycles | 3.38 |
| P8 cycles | 3.38 |
| P9 cycles | 3.38 |
| P10 cycles | 2.38 |
| P11 cycles | 2.50 |
| P12 cycles | 2.13 |
| P13 cycles | 2.00 |
| P14 cycles | 0.25 |
| P15 cycles | 0.25 |
| DIV/SQRT cycles | 0.00 |
| Inter-iter dependencies cycles | 4 |
| FE+BE cycles (UFS) | NA |
| Stall cycles (UFS) | NA |
| Nb insns | 60.00 |
| Nb uops | 62.00 |
| Nb loads | 10.50 |
| Nb stores | 3.00 |
| Nb stack references | 5.50 |
| FLOP/cycle | 0.37 |
| Nb FLOP add-sub | 0.00 |
| Nb FLOP mul | 1.00 |
| Nb FLOP fma | 1.00 |
| Nb FLOP div | 0.00 |
| Nb FLOP rcp | 0.00 |
| Nb FLOP sqrt | 0.00 |
| Nb FLOP rsqrt | 0.00 |
| Bytes/cycle | 10.10 |
| Bytes prefetched | 0.00 |
| Bytes loaded | 80.00 |
| Bytes stored | 24.00 |
| Stride 0 | 0.00 |
| Stride 1 | 0.00 |
| Stride n | 0.00 |
| Stride unknown | 2.00 |
| Stride indirect | 0.00 |
| Vectorization ratio all | 0.77 |
| Vectorization ratio load | 0.00 |
| Vectorization ratio store | 0.00 |
| Vectorization ratio mul | 0.00 |
| Vectorization ratio add_sub | 0.00 |
| Vectorization ratio fma | 0.00 |
| Vectorization ratio div_sqrt | NA |
| Vectorization ratio other | 0.91 |
| Vector-efficiency ratio all | 9.69 |
| Vector-efficiency ratio load | 9.18 |
| Vector-efficiency ratio store | 12.50 |
| Vector-efficiency ratio mul | 6.25 |
| Vector-efficiency ratio add_sub | 12.50 |
| Vector-efficiency ratio fma | 6.25 |
| Vector-efficiency ratio div_sqrt | NA |
| Vector-efficiency ratio other | 9.52 |
| Metric | Value |
|---|---|
| CQA speedup if no scalar integer | 3.09 |
| CQA speedup if FP arith vectorized | 1.78 |
| CQA speedup if fully vectorized | 11.52 |
| CQA speedup if no inter-iteration dependency | NA |
| CQA speedup if next bottleneck killed | 1.38 |
| Bottlenecks | micro-operation queue, |
| Function | main |
| Source | random.tcc:401-406,random.tcc:409-409,random.tcc:413-413,random.tcc:417-420,random.tcc:455-455,random.tcc:459-463,random.tcc:3364-3367 |
| Source loop unroll info | NA |
| Source loop unroll confidence level | NA |
| Unroll/vectorization loop type | NA |
| Unroll factor | NA |
| CQA cycles | 12.38 |
| CQA cycles if no scalar integer | 4.00 |
| CQA cycles if FP arith vectorized | 6.94 |
| CQA cycles if fully vectorized | 1.07 |
| Front-end cycles | 12.38 |
| P0 cycles | 9.00 |
| P1 cycles | 9.00 |
| P2 cycles | 9.00 |
| P3 cycles | 9.00 |
| P4 cycles | 9.00 |
| P5 cycles | 9.00 |
| P6 cycles | 6.00 |
| P7 cycles | 6.00 |
| P8 cycles | 6.00 |
| P9 cycles | 6.00 |
| P10 cycles | 3.75 |
| P11 cycles | 4.00 |
| P12 cycles | 3.75 |
| P13 cycles | 3.50 |
| P14 cycles | 0.50 |
| P15 cycles | 0.50 |
| DIV/SQRT cycles | 0.00 |
| Inter-iter dependencies cycles | 4 |
| FE+BE cycles (UFS) | NA |
| Stall cycles (UFS) | NA |
| Nb insns | 95.00 |
| Nb uops | 99.00 |
| Nb loads | 19.00 |
| Nb stores | 5.00 |
| Nb stack references | 10.00 |
| FLOP/cycle | 0.24 |
| Nb FLOP add-sub | 0.00 |
| Nb FLOP mul | 1.00 |
| Nb FLOP fma | 1.00 |
| Nb FLOP div | 0.00 |
| Nb FLOP rcp | 0.00 |
| Nb FLOP sqrt | 0.00 |
| Nb FLOP rsqrt | 0.00 |
| Bytes/cycle | 15.19 |
| Bytes prefetched | 0.00 |
| Bytes loaded | 148.00 |
| Bytes stored | 40.00 |
| Stride 0 | 0.00 |
| Stride 1 | 0.00 |
| Stride n | 0.00 |
| Stride unknown | 2.00 |
| Stride indirect | 0.00 |
| Vectorization ratio all | 1.54 |
| Vectorization ratio load | 0.00 |
| Vectorization ratio store | 0.00 |
| Vectorization ratio mul | 0.00 |
| Vectorization ratio add_sub | NA |
| Vectorization ratio fma | 0.00 |
| Vectorization ratio div_sqrt | NA |
| Vectorization ratio other | 1.82 |
| Vector-efficiency ratio all | 10.29 |
| Vector-efficiency ratio load | 12.11 |
| Vector-efficiency ratio store | 12.50 |
| Vector-efficiency ratio mul | 6.25 |
| Vector-efficiency ratio add_sub | NA |
| Vector-efficiency ratio fma | 6.25 |
| Vector-efficiency ratio div_sqrt | NA |
| Vector-efficiency ratio other | 10.11 |
| Metric | Value |
|---|---|
| CQA speedup if no scalar integer | 1.00 |
| CQA speedup if FP arith vectorized | 3.72 |
| CQA speedup if fully vectorized | 14.29 |
| CQA speedup if no inter-iteration dependency | NA |
| CQA speedup if next bottleneck killed | 1.28 |
| Bottlenecks | |
| Function | main |
| Source | random.tcc:401-406,random.tcc:409-409,random.tcc:413-413,random.tcc:417-420,random.tcc:455-455,random.tcc:459-463,random.tcc:3364-3367 |
| Source loop unroll info | NA |
| Source loop unroll confidence level | NA |
| Unroll/vectorization loop type | NA |
| Unroll factor | NA |
| CQA cycles | 4.00 |
| CQA cycles if no scalar integer | 4.00 |
| CQA cycles if FP arith vectorized | 1.07 |
| CQA cycles if fully vectorized | 0.28 |
| Front-end cycles | 3.13 |
| P0 cycles | 2.50 |
| P1 cycles | 2.50 |
| P2 cycles | 2.50 |
| P3 cycles | 2.50 |
| P4 cycles | 2.50 |
| P5 cycles | 2.50 |
| P6 cycles | 0.75 |
| P7 cycles | 0.75 |
| P8 cycles | 0.75 |
| P9 cycles | 0.75 |
| P10 cycles | 1.00 |
| P11 cycles | 1.00 |
| P12 cycles | 0.50 |
| P13 cycles | 0.50 |
| P14 cycles | 0.00 |
| P15 cycles | 0.00 |
| DIV/SQRT cycles | 0.00 |
| Inter-iter dependencies cycles | 4 |
| FE+BE cycles (UFS) | NA |
| Stall cycles (UFS) | NA |
| Nb insns | 25.00 |
| Nb uops | 25.00 |
| Nb loads | 2.00 |
| Nb stores | 1.00 |
| Nb stack references | 1.00 |
| FLOP/cycle | 0.75 |
| Nb FLOP add-sub | 0.00 |
| Nb FLOP mul | 1.00 |
| Nb FLOP fma | 1.00 |
| Nb FLOP div | 0.00 |
| Nb FLOP rcp | 0.00 |
| Nb FLOP sqrt | 0.00 |
| Nb FLOP rsqrt | 0.00 |
| Bytes/cycle | 5.00 |
| Bytes prefetched | 0.00 |
| Bytes loaded | 12.00 |
| Bytes stored | 8.00 |
| Stride 0 | NA |
| Stride 1 | NA |
| Stride n | NA |
| Stride unknown | NA |
| Stride indirect | NA |
| Vectorization ratio all | 0.00 |
| Vectorization ratio load | 0.00 |
| Vectorization ratio store | 0.00 |
| Vectorization ratio mul | 0.00 |
| Vectorization ratio add_sub | 0.00 |
| Vectorization ratio fma | 0.00 |
| Vectorization ratio div_sqrt | NA |
| Vectorization ratio other | 0.00 |
| Vector-efficiency ratio all | 9.09 |
| Vector-efficiency ratio load | 6.25 |
| Vector-efficiency ratio store | 12.50 |
| Vector-efficiency ratio mul | 6.25 |
| Vector-efficiency ratio add_sub | 12.50 |
| Vector-efficiency ratio fma | 6.25 |
| Vector-efficiency ratio div_sqrt | NA |
| Vector-efficiency ratio other | 8.93 |
| Path / |
| Function | main |
| Source file and lines | random.tcc:401-3367 |
| Module | attention-clang-znver5-256 |
| nb instructions | 60 |
| nb uops | 62 |
| loop length | 291.50 |
| used x86 registers | 7.50 |
| used mmx registers | 0 |
| used xmm registers | 6 |
| used ymm registers | 2.50 |
| used zmm registers | 0 |
| nb stack references | 5.50 |
| micro-operation queue | 7.75 cycles |
| front end | 7.75 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| uops | 5.75 | 5.75 | 5.75 | 5.75 | 5.75 | 5.75 | 3.38 | 3.38 | 3.38 | 3.38 | 2.38 | 2.50 | 2.13 | 2.00 | 0.25 | 0.25 |
| cycles | 5.75 | 5.75 | 5.75 | 5.75 | 5.75 | 5.75 | 3.38 | 3.38 | 3.38 | 3.38 | 2.38 | 2.50 | 2.13 | 2.00 | 0.25 | 0.25 |
| Cycles executing div or sqrt instructions | NA |
| Longest recurrence chain latency (RecMII) | 4.00 |
| Front-end | 7.75 |
| Dispatch | 5.75 |
| Data deps. | 4.00 |
| Overall L1 | 8.19 |
| all | 0% |
| load | 0% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 0% |
| all | 0% |
| load | 0% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 0% |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | 0% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 0% |
| load | 0% |
| store | 0% |
| mul | 0% |
| add-sub | 0% |
| fma | 0% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 0% |
| all | 10% |
| load | 12% |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 9% |
| all | 6% |
| load | 6% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 6% |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | 6% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 9% |
| load | 9% |
| store | 12% |
| mul | 6% |
| add-sub | 12% |
| fma | 6% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 9% |
| Function | main |
| Source file and lines | random.tcc:401-3367 |
| Module | attention-clang-znver5-256 |
| nb instructions | 95 |
| nb uops | 99 |
| loop length | 473 |
| used x86 registers | 10 |
| used mmx registers | 0 |
| used xmm registers | 8 |
| used ymm registers | 5 |
| used zmm registers | 0 |
| nb stack references | 10 |
| micro-operation queue | 12.38 cycles |
| front end | 12.38 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| uops | 9.00 | 9.00 | 9.00 | 9.00 | 9.00 | 9.00 | 6.00 | 6.00 | 6.00 | 6.00 | 3.75 | 4.00 | 3.75 | 3.50 | 0.50 | 0.50 |
| cycles | 9.00 | 9.00 | 9.00 | 9.00 | 9.00 | 9.00 | 6.00 | 6.00 | 6.00 | 6.00 | 3.75 | 4.00 | 3.75 | 3.50 | 0.50 | 0.50 |
| Cycles executing div or sqrt instructions | NA |
| Longest recurrence chain latency (RecMII) | 4.00 |
| Front-end | 12.38 |
| Dispatch | 9.00 |
| Data deps. | 4.00 |
| Overall L1 | 12.38 |
| all | 1% |
| load | 0% |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 1% |
| all | 0% |
| load | 0% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 0% |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | 0% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 1% |
| load | 0% |
| store | 0% |
| mul | 0% |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | 0% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 1% |
| all | 10% |
| load | 12% |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 10% |
| all | 6% |
| load | 6% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 6% |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | 6% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 10% |
| load | 12% |
| store | 12% |
| mul | 6% |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | 6% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 10% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| MOV %R13,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| INC %R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV $0x200b,%EDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| MOV %R13,0x1818(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV 0x498(%RSP,%RCX,8),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| BEXTR %RDX,%RCX,%RDX | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | N/A |
| XOR %RCX,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %EDX,%ECX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| SAL $0x7,%ECX | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | N/A |
| AND $-0x62d3a980,%ECX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| XOR %RDX,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV %ECX,%EDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | scal (6.3%) |
| SAL $0xf,%EDX | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | scal (6.3%) |
| AND $-0x103a0000,%EDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| XOR %RCX,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %RDX,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| SHR $0x12,%RCX | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | N/A |
| XOR %RDX,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| DEC %RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VCVTUSI2SS %RCX,%XMM15,%XMM2 | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3-9 | 1 | scal (12.5%) |
| VFMADD231SS %XMM2,%XMM1,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (6.3%) |
| VMULSS 0x3739(%RIP),%XMM1,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 3 | 0.50 | scal (6.3%) |
| JE 4ce0 <main+0x1f10> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-0.50 | N/A |
| CMP $0x270,%R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| JB 4870 <main+0x1aa0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-0.50 | N/A |
| VPBROADCASTQ 0x3759(%RIP),%YMM14 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 2 | 0.50 | scal (12.5%) |
| VPBROADCASTQ 0x3758(%RIP),%YMM15 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 2 | 0.50 | scal (12.5%) |
| VPBROADCASTQ 0x3756(%RIP),%YMM16 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 2 | 0.50 | scal (12.5%) |
| VPBROADCASTQ 0x3754(%RIP),%YMM17 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 2 | 0.50 | scal (12.5%) |
| VPBROADCASTQ %RBX,%YMM2 | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 2 | 1 | scal (12.5%) |
| XOR %ECX,%ECX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| NOPL (%RAX) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| MOV 0xba0(%RSP),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| VEXTRACTI128 $0x1,%YMM2,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 3 | 0.25 | vect (25.0%) |
| MOV 0xba8(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| VPBROADCASTQ 0x3604(%RIP),%XMM8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 2 | 0.50 | scal (12.5%) |
| VPBROADCASTQ 0x3603(%RIP),%XMM9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 2 | 0.50 | scal (12.5%) |
| VPBROADCASTQ 0x3602(%RIP),%XMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 2 | 0.50 | scal (12.5%) |
| VPBROADCASTQ 0x3601(%RIP),%XMM11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 2 | 0.50 | scal (12.5%) |
| VPEXTRQ $0x1,%XMM2,%RSI | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0.50 | 0.50 | 7 | 0.50 | scal (12.5%) |
| AND $-0x80000000,%RSI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %EDX,%EDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | scal (6.3%) |
| AND $0x7ffffffe,%EDI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| OR %RSI,%RDI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %EDX,%ESI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | scal (6.3%) |
| AND $0x1,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| AND $-0x80000000,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| SHR $0x1,%RDI | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | scal (12.5%) |
| XOR 0x1800(%RSP),%RDI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
| NEG %ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| AND %R12D,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| XOR %RDI,%RSI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %RSI,0xb98(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV %ECX,%ESI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | scal (6.3%) |
| AND $0x7ffffffe,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| OR %RDX,%RSI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %ECX,%EDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | scal (6.3%) |
| AND $0x1,%EDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| AND $-0x80000000,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| SHR $0x1,%RSI | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | scal (12.5%) |
| XOR 0x1808(%RSP),%RSI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
| NEG %EDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| AND %R12D,%EDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| XOR %RSI,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %RDX,0xba0(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV 0xbb0(%RSP),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| MOV %EDX,%ESI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | scal (6.3%) |
| VPBROADCASTQ %RDX,%XMM2 | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 2 | 1 | scal (12.5%) |
| AND $0x7ffffffe,%EDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| AND $0x1,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| OR %RCX,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| NEG %ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| MOV $0xee,%ECX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| SHR $0x1,%RDX | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | scal (12.5%) |
| XOR 0x1810(%RSP),%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
| AND %R12D,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| XOR %RDX,%RSI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %RSI,0xba8(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| NOPW %CS:(%RAX,%RAX,1) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| MOV 0x1810(%RSP),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV 0x498(%RSP),%RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| XOR %R13D,%R13D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| MOV %EBX,%EDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | scal (6.3%) |
| AND %R14,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| AND $0x7ffffffe,%EDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| OR %RCX,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %EBX,%ECX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| AND $0x1,%ECX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| SHR $0x1,%RDX | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | scal (12.5%) |
| XOR 0x10f8(%RSP),%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
| NEG %ECX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| AND %R12D,%ECX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| XOR %RDX,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV %RCX,0x1810(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| JMP 4870 <main+0x1aa0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| Function | main |
| Source file and lines | random.tcc:401-3367 |
| Module | attention-clang-znver5-256 |
| nb instructions | 25 |
| nb uops | 25 |
| loop length | 110 |
| used x86 registers | 5 |
| used mmx registers | 0 |
| used xmm registers | 4 |
| used ymm registers | 0 |
| used zmm registers | 0 |
| nb stack references | 1 |
| micro-operation queue | 3.13 cycles |
| front end | 3.13 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| uops | 2.50 | 2.50 | 2.50 | 2.50 | 2.50 | 2.50 | 0.75 | 0.75 | 0.75 | 0.75 | 1.00 | 1.00 | 0.50 | 0.50 | 0.00 | 0.00 |
| cycles | 2.50 | 2.50 | 2.50 | 2.50 | 2.50 | 2.50 | 0.75 | 0.75 | 0.75 | 0.75 | 1.00 | 1.00 | 0.50 | 0.50 | 0.00 | 0.00 |
| Cycles executing div or sqrt instructions | NA |
| Longest recurrence chain latency (RecMII) | 4.00 |
| Front-end | 3.13 |
| Dispatch | 2.50 |
| Data deps. | 4.00 |
| Overall L1 | 4.00 |
| all | 0% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | 0% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 0% |
| all | 0% |
| load | 0% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 0% |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | 0% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 0% |
| load | 0% |
| store | 0% |
| mul | 0% |
| add-sub | 0% |
| fma | 0% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 0% |
| all | 9% |
| load | NA (no load vectorizable/vectorized instructions) |
| store | 12% |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | 12% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 8% |
| all | 6% |
| load | 6% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 6% |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | 6% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 9% |
| load | 6% |
| store | 12% |
| mul | 6% |
| add-sub | 12% |
| fma | 6% |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | 8% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| MOV %R13,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| INC %R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV $0x200b,%EDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| MOV %R13,0x1818(%RSP) | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
| MOV 0x498(%RSP,%RCX,8),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| BEXTR %RDX,%RCX,%RDX | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | N/A |
| XOR %RCX,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %EDX,%ECX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| SAL $0x7,%ECX | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | N/A |
| AND $-0x62d3a980,%ECX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| XOR %RDX,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| MOV %ECX,%EDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | scal (6.3%) |
| SAL $0xf,%EDX | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | scal (6.3%) |
| AND $-0x103a0000,%EDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| XOR %RCX,%RDX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| MOV %RDX,%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| SHR $0x12,%RCX | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | N/A |
| XOR %RDX,%RCX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| DEC %RAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (12.5%) |
| VCVTUSI2SS %RCX,%XMM15,%XMM2 | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3-9 | 1 | scal (12.5%) |
| VFMADD231SS %XMM2,%XMM1,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (6.3%) |
| VMULSS 0x3739(%RIP),%XMM1,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 3 | 0.50 | scal (6.3%) |
| JE 4ce0 <main+0x1f10> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-0.50 | N/A |
| CMP $0x270,%R13 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| JB 4870 <main+0x1aa0> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-0.50 | N/A |
