| Function: advanceVelocity._omp_fn.0 | Module: exec | Source: timestep.c:71-78 | Coverage (incl. loops): 1.94% | (excl. loops): 0.00% |
|---|
| Function: advanceVelocity._omp_fn.0 | Module: exec | Source: timestep.c:71-78 | Coverage (incl. loops): 1.94% | (excl. loops): 0.00% |
|---|
/beegfs/hackathon/users/eoseret/qaas_runs_test/gmz12.benchmarkcenter.megware.com/177-374-1600/CoMD/build/CoMD/CoMD/src-openmp/timestep.c: 71 - 78 |
-------------------------------------------------------------------------------- |
71: #pragma omp parallel for |
72: for (int iBox=0; iBox<nBoxes; iBox++) |
73: { |
74: for (int iOff=MAXATOMS*iBox,ii=0; ii<s->boxes->nAtoms[iBox]; ii++,iOff++) |
75: { |
76: s->atoms->p[iOff][0] += dt*s->atoms->f[iOff][0]; |
77: s->atoms->p[iOff][1] += dt*s->atoms->f[iOff][1]; |
78: s->atoms->p[iOff][2] += dt*s->atoms->f[iOff][2]; |
0x40ea40 PUSH %RBP |
0x40ea41 MOV %RSP,%RBP |
0x40ea44 PUSH %R12 |
0x40ea46 PUSH %RBX |
0x40ea47 MOV %RDI,%RBX |
0x40ea4a CALL 4032d0 <omp_get_num_threads@plt> |
0x40ea4f MOV %EAX,%R12D |
0x40ea52 CALL 403280 <omp_get_thread_num@plt> |
0x40ea57 MOV %EAX,%ESI |
0x40ea59 MOV 0x10(%RBX),%EAX |
0x40ea5c CLTD |
0x40ea5d IDIV %R12D |
0x40ea60 CMP %EDX,%ESI |
0x40ea62 JL 40eea0 |
0x40ea68 IMUL %EAX,%ESI |
0x40ea6b ADD %EDX,%ESI |
0x40ea6d ADD %ESI,%EAX |
0x40ea6f CMP %EAX,%ESI |
0x40ea71 JGE 40ee9b |
0x40ea77 MOV (%RBX),%R11 |
0x40ea7a MOVSXD %ESI,%R12 |
0x40ea7d VMOVSD 0x8(%RBX),%XMM0 |
0x40ea82 LEA (%R12,%R12,2),%R8 |
0x40ea86 SAL $0x9,%R8 |
0x40ea8a MOV 0x18(%R11),%RCX |
0x40ea8e MOV 0x78(%RCX),%R10 |
0x40ea92 NOPW %CS:(%RAX,%RAX,1) |
0x40ea9d NOPL (%RAX) |
(99) 0x40eaa0 MOVSXD (%R10,%R12,4),%RDI |
(99) 0x40eaa4 TEST %EDI,%EDI |
(99) 0x40eaa6 JLE 40ee88 |
(99) 0x40eaac MOV 0x20(%R11),%R9 |
(99) 0x40eab0 LEA (%RDI,%RDI,2),%RBX |
(99) 0x40eab4 LEA -0x18(,%RBX,8),%RDI |
(99) 0x40eabc SHR $0x3,%RDI |
(99) 0x40eac0 MOV 0x20(%R9),%RSI |
(99) 0x40eac4 MOV 0x28(%R9),%RCX |
(99) 0x40eac8 MOV $0xaaaaaaaaaaaaaab,%R9 |
(99) 0x40ead2 IMUL %R9,%RDI |
(99) 0x40ead6 INC %RDI |
(99) 0x40ead9 ADD %R8,%RSI |
(99) 0x40eadc ADD %R8,%RCX |
(99) 0x40eadf AND $0x7,%EDI |
(99) 0x40eae2 LEA (%RSI,%RBX,8),%RDX |
(99) 0x40eae6 JE 40ecaa |
(99) 0x40eaec CMP $0x1,%RDI |
(99) 0x40eaf0 JE 40ec6a |
(99) 0x40eaf6 CMP $0x2,%RDI |
(99) 0x40eafa JE 40ec33 |
(99) 0x40eb00 CMP $0x3,%RDI |
(99) 0x40eb04 JE 40ebfc |
(99) 0x40eb0a CMP $0x4,%RDI |
(99) 0x40eb0e JE 40ebc5 |
(99) 0x40eb14 CMP $0x5,%RDI |
(99) 0x40eb18 JE 40eb8e |
(99) 0x40eb1a CMP $0x6,%RDI |
(99) 0x40eb1e JE 40eb57 |
(99) 0x40eb20 VMOVSD (%RCX),%XMM1 |
(99) 0x40eb24 ADD $0x18,%RSI |
(99) 0x40eb28 VFMADD213SD -0x18(%RSI),%XMM0,%XMM1 |
(99) 0x40eb2e ADD $0x18,%RCX |
(99) 0x40eb32 VMOVSD %XMM1,-0x18(%RSI) |
(99) 0x40eb37 VMOVSD -0x10(%RCX),%XMM2 |
(99) 0x40eb3c VFMADD213SD -0x10(%RSI),%XMM0,%XMM2 |
(99) 0x40eb42 VMOVSD %XMM2,-0x10(%RSI) |
(99) 0x40eb47 VMOVSD -0x8(%RCX),%XMM3 |
(99) 0x40eb4c VFMADD213SD -0x8(%RSI),%XMM0,%XMM3 |
(99) 0x40eb52 VMOVSD %XMM3,-0x8(%RSI) |
(99) 0x40eb57 VMOVSD (%RCX),%XMM4 |
(99) 0x40eb5b ADD $0x18,%RSI |
(99) 0x40eb5f VFMADD213SD -0x18(%RSI),%XMM0,%XMM4 |
(99) 0x40eb65 ADD $0x18,%RCX |
(99) 0x40eb69 VMOVSD %XMM4,-0x18(%RSI) |
(99) 0x40eb6e VMOVSD -0x10(%RCX),%XMM5 |
(99) 0x40eb73 VFMADD213SD -0x10(%RSI),%XMM0,%XMM5 |
(99) 0x40eb79 VMOVSD %XMM5,-0x10(%RSI) |
(99) 0x40eb7e VMOVSD -0x8(%RCX),%XMM6 |
(99) 0x40eb83 VFMADD213SD -0x8(%RSI),%XMM0,%XMM6 |
(99) 0x40eb89 VMOVSD %XMM6,-0x8(%RSI) |
(99) 0x40eb8e VMOVSD (%RCX),%XMM7 |
(99) 0x40eb92 ADD $0x18,%RSI |
(99) 0x40eb96 VFMADD213SD -0x18(%RSI),%XMM0,%XMM7 |
(99) 0x40eb9c ADD $0x18,%RCX |
(99) 0x40eba0 VMOVSD %XMM7,-0x18(%RSI) |
(99) 0x40eba5 VMOVSD -0x10(%RCX),%XMM8 |
(99) 0x40ebaa VFMADD213SD -0x10(%RSI),%XMM0,%XMM8 |
(99) 0x40ebb0 VMOVSD %XMM8,-0x10(%RSI) |
(99) 0x40ebb5 VMOVSD -0x8(%RCX),%XMM9 |
(99) 0x40ebba VFMADD213SD -0x8(%RSI),%XMM0,%XMM9 |
(99) 0x40ebc0 VMOVSD %XMM9,-0x8(%RSI) |
(99) 0x40ebc5 VMOVSD (%RCX),%XMM10 |
(99) 0x40ebc9 ADD $0x18,%RSI |
(99) 0x40ebcd VFMADD213SD -0x18(%RSI),%XMM0,%XMM10 |
(99) 0x40ebd3 ADD $0x18,%RCX |
(99) 0x40ebd7 VMOVSD %XMM10,-0x18(%RSI) |
(99) 0x40ebdc VMOVSD -0x10(%RCX),%XMM11 |
(99) 0x40ebe1 VFMADD213SD -0x10(%RSI),%XMM0,%XMM11 |
(99) 0x40ebe7 VMOVSD %XMM11,-0x10(%RSI) |
(99) 0x40ebec VMOVSD -0x8(%RCX),%XMM12 |
(99) 0x40ebf1 VFMADD213SD -0x8(%RSI),%XMM0,%XMM12 |
(99) 0x40ebf7 VMOVSD %XMM12,-0x8(%RSI) |
(99) 0x40ebfc VMOVSD (%RCX),%XMM13 |
(99) 0x40ec00 ADD $0x18,%RSI |
(99) 0x40ec04 VFMADD213SD -0x18(%RSI),%XMM0,%XMM13 |
(99) 0x40ec0a ADD $0x18,%RCX |
(99) 0x40ec0e VMOVSD %XMM13,-0x18(%RSI) |
(99) 0x40ec13 VMOVSD -0x10(%RCX),%XMM14 |
(99) 0x40ec18 VFMADD213SD -0x10(%RSI),%XMM0,%XMM14 |
(99) 0x40ec1e VMOVSD %XMM14,-0x10(%RSI) |
(99) 0x40ec23 VMOVSD -0x8(%RCX),%XMM15 |
(99) 0x40ec28 VFMADD213SD -0x8(%RSI),%XMM0,%XMM15 |
(99) 0x40ec2e VMOVSD %XMM15,-0x8(%RSI) |
(99) 0x40ec33 VMOVSD (%RCX),%XMM1 |
(99) 0x40ec37 ADD $0x18,%RSI |
(99) 0x40ec3b VFMADD213SD -0x18(%RSI),%XMM0,%XMM1 |
(99) 0x40ec41 ADD $0x18,%RCX |
(99) 0x40ec45 VMOVSD %XMM1,-0x18(%RSI) |
(99) 0x40ec4a VMOVSD -0x10(%RCX),%XMM2 |
(99) 0x40ec4f VFMADD213SD -0x10(%RSI),%XMM0,%XMM2 |
(99) 0x40ec55 VMOVSD %XMM2,-0x10(%RSI) |
(99) 0x40ec5a VMOVSD -0x8(%RCX),%XMM3 |
(99) 0x40ec5f VFMADD213SD -0x8(%RSI),%XMM0,%XMM3 |
(99) 0x40ec65 VMOVSD %XMM3,-0x8(%RSI) |
(99) 0x40ec6a VMOVSD (%RCX),%XMM4 |
(99) 0x40ec6e ADD $0x18,%RSI |
(99) 0x40ec72 VFMADD213SD -0x18(%RSI),%XMM0,%XMM4 |
(99) 0x40ec78 VMOVSD %XMM4,-0x18(%RSI) |
(99) 0x40ec7d VMOVSD 0x8(%RCX),%XMM5 |
(99) 0x40ec82 VFMADD213SD -0x10(%RSI),%XMM0,%XMM5 |
(99) 0x40ec88 VMOVSD %XMM5,-0x10(%RSI) |
(99) 0x40ec8d VMOVSD 0x10(%RCX),%XMM6 |
(99) 0x40ec92 ADD $0x18,%RCX |
(99) 0x40ec96 VFMADD213SD -0x8(%RSI),%XMM0,%XMM6 |
(99) 0x40ec9c VMOVSD %XMM6,-0x8(%RSI) |
(99) 0x40eca1 CMP %RDX,%RSI |
(99) 0x40eca4 JE 40ee88 |
(100) 0x40ecaa VMOVSD (%RCX),%XMM7 |
(100) 0x40ecae ADD $0xc0,%RSI |
(100) 0x40ecb5 VFMADD213SD -0xc0(%RSI),%XMM0,%XMM7 |
(100) 0x40ecbe VMOVSD %XMM7,-0xc0(%RSI) |
(100) 0x40ecc6 VMOVSD 0x8(%RCX),%XMM8 |
(100) 0x40eccb VFMADD213SD -0xb8(%RSI),%XMM0,%XMM8 |
(100) 0x40ecd4 VMOVSD %XMM8,-0xb8(%RSI) |
(100) 0x40ecdc VMOVSD 0x10(%RCX),%XMM9 |
(100) 0x40ece1 VFMADD213SD -0xb0(%RSI),%XMM0,%XMM9 |
(100) 0x40ecea VMOVSD %XMM9,-0xb0(%RSI) |
(100) 0x40ecf2 VMOVSD 0x18(%RCX),%XMM10 |
(100) 0x40ecf7 VFMADD213SD -0xa8(%RSI),%XMM0,%XMM10 |
(100) 0x40ed00 VMOVSD %XMM10,-0xa8(%RSI) |
(100) 0x40ed08 VMOVSD 0x20(%RCX),%XMM11 |
(100) 0x40ed0d VFMADD213SD -0xa0(%RSI),%XMM0,%XMM11 |
(100) 0x40ed16 VMOVSD %XMM11,-0xa0(%RSI) |
(100) 0x40ed1e VMOVSD 0x28(%RCX),%XMM12 |
(100) 0x40ed23 VFMADD213SD -0x98(%RSI),%XMM0,%XMM12 |
(100) 0x40ed2c VMOVSD %XMM12,-0x98(%RSI) |
(100) 0x40ed34 VMOVSD 0x30(%RCX),%XMM13 |
(100) 0x40ed39 VFMADD213SD -0x90(%RSI),%XMM0,%XMM13 |
(100) 0x40ed42 VMOVSD %XMM13,-0x90(%RSI) |
(100) 0x40ed4a VMOVSD 0x38(%RCX),%XMM14 |
(100) 0x40ed4f VFMADD213SD -0x88(%RSI),%XMM0,%XMM14 |
(100) 0x40ed58 VMOVSD %XMM14,-0x88(%RSI) |
(100) 0x40ed60 VMOVSD 0x40(%RCX),%XMM15 |
(100) 0x40ed65 VFMADD213SD -0x80(%RSI),%XMM0,%XMM15 |
(100) 0x40ed6b VMOVSD %XMM15,-0x80(%RSI) |
(100) 0x40ed70 VMOVSD 0x48(%RCX),%XMM1 |
(100) 0x40ed75 VFMADD213SD -0x78(%RSI),%XMM0,%XMM1 |
(100) 0x40ed7b VMOVSD %XMM1,-0x78(%RSI) |
(100) 0x40ed80 VMOVSD 0x50(%RCX),%XMM2 |
(100) 0x40ed85 VFMADD213SD -0x70(%RSI),%XMM0,%XMM2 |
(100) 0x40ed8b VMOVSD %XMM2,-0x70(%RSI) |
(100) 0x40ed90 VMOVSD 0x58(%RCX),%XMM3 |
(100) 0x40ed95 VFMADD213SD -0x68(%RSI),%XMM0,%XMM3 |
(100) 0x40ed9b VMOVSD %XMM3,-0x68(%RSI) |
(100) 0x40eda0 VMOVSD 0x60(%RCX),%XMM4 |
(100) 0x40eda5 VFMADD213SD -0x60(%RSI),%XMM0,%XMM4 |
(100) 0x40edab VMOVSD %XMM4,-0x60(%RSI) |
(100) 0x40edb0 VMOVSD 0x68(%RCX),%XMM5 |
(100) 0x40edb5 VFMADD213SD -0x58(%RSI),%XMM0,%XMM5 |
(100) 0x40edbb VMOVSD %XMM5,-0x58(%RSI) |
(100) 0x40edc0 VMOVSD 0x70(%RCX),%XMM6 |
(100) 0x40edc5 VFMADD213SD -0x50(%RSI),%XMM0,%XMM6 |
(100) 0x40edcb VMOVSD %XMM6,-0x50(%RSI) |
(100) 0x40edd0 VMOVSD 0x78(%RCX),%XMM7 |
(100) 0x40edd5 VFMADD213SD -0x48(%RSI),%XMM0,%XMM7 |
(100) 0x40eddb VMOVSD %XMM7,-0x48(%RSI) |
(100) 0x40ede0 VMOVSD 0x80(%RCX),%XMM8 |
(100) 0x40ede8 VFMADD213SD -0x40(%RSI),%XMM0,%XMM8 |
(100) 0x40edee VMOVSD %XMM8,-0x40(%RSI) |
(100) 0x40edf3 VMOVSD 0x88(%RCX),%XMM9 |
(100) 0x40edfb VFMADD213SD -0x38(%RSI),%XMM0,%XMM9 |
(100) 0x40ee01 VMOVSD %XMM9,-0x38(%RSI) |
(100) 0x40ee06 VMOVSD 0x90(%RCX),%XMM10 |
(100) 0x40ee0e VFMADD213SD -0x30(%RSI),%XMM0,%XMM10 |
(100) 0x40ee14 VMOVSD %XMM10,-0x30(%RSI) |
(100) 0x40ee19 VMOVSD 0x98(%RCX),%XMM11 |
(100) 0x40ee21 VFMADD213SD -0x28(%RSI),%XMM0,%XMM11 |
(100) 0x40ee27 VMOVSD %XMM11,-0x28(%RSI) |
(100) 0x40ee2c VMOVSD 0xa0(%RCX),%XMM12 |
(100) 0x40ee34 VFMADD213SD -0x20(%RSI),%XMM0,%XMM12 |
(100) 0x40ee3a VMOVSD %XMM12,-0x20(%RSI) |
(100) 0x40ee3f VMOVSD 0xa8(%RCX),%XMM13 |
(100) 0x40ee47 VFMADD213SD -0x18(%RSI),%XMM0,%XMM13 |
(100) 0x40ee4d VMOVSD %XMM13,-0x18(%RSI) |
(100) 0x40ee52 VMOVSD 0xb0(%RCX),%XMM14 |
(100) 0x40ee5a VFMADD213SD -0x10(%RSI),%XMM0,%XMM14 |
(100) 0x40ee60 VMOVSD %XMM14,-0x10(%RSI) |
(100) 0x40ee65 VMOVSD 0xb8(%RCX),%XMM15 |
(100) 0x40ee6d ADD $0xc0,%RCX |
(100) 0x40ee74 VFMADD213SD -0x8(%RSI),%XMM0,%XMM15 |
(100) 0x40ee7a VMOVSD %XMM15,-0x8(%RSI) |
(100) 0x40ee7f CMP %RDX,%RSI |
(100) 0x40ee82 JNE 40ecaa |
(99) 0x40ee88 INC %R12 |
(99) 0x40ee8b ADD $0x600,%R8 |
(99) 0x40ee92 CMP %R12D,%EAX |
(99) 0x40ee95 JG 40eaa0 |
0x40ee9b POP %RBX |
0x40ee9c POP %R12 |
0x40ee9e POP %RBP |
0x40ee9f RET |
0x40eea0 INC %EAX |
0x40eea2 XOR %EDX,%EDX |
0x40eea4 JMP 40ea68 |
0x40eea9 NOPL (%RAX) |
| Coverage (%) | Name | Source Location | Module |
|---|---|---|---|
| ○98.83 | gomp_thread_start | team.c:130 | libgomp.so.1.0.0 |
| ○1.17 | GOMP_parallel | libgomp.h:982 | libgomp.so.1.0.0 |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| min | med | avg | max |
|---|---|---|---|
| Percentile Index | 10 | 20 | 30 | 40 | 50 | 60 | 70 | 80 | 90 | 100 |
|---|---|---|---|---|---|---|---|---|---|---|
| Value |
| Path / |
The code analyzed by CQA in that panel excludes loops and represents 0.00% of application time for run gcc_10
| Source file and lines | timestep.c:71-78 |
| Module | exec |
| nb instructions | 36 |
| nb uops | 35 |
| loop length | 117 |
| used x86 registers | 12 |
| used mmx registers | 0 |
| used xmm registers | 1 |
| used ymm registers | 0 |
| used zmm registers | 0 |
| nb stack references | 0 |
| micro-operation queue | 4.38 cycles |
| front end | 4.38 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| uops | 2.67 | 2.67 | 2.67 | 3.00 | 3.00 | 3.00 | 1.75 | 1.75 | 1.75 | 1.75 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 |
| cycles | 2.67 | 2.67 | 2.67 | 3.00 | 3.00 | 3.00 | 1.75 | 1.75 | 1.75 | 1.75 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 |
| Cycles executing div or sqrt instructions | 6.00 |
| Front-end | 4.38 |
| Dispatch | 3.00 |
| DIV/SQRT | 6.00 |
| Overall L1 | 6.00 |
| all | 0% |
| load | 0% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 0% |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 0% |
| all | 0% |
| load | 0% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 0% |
| load | 0% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 0% |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | 0% |
| other | 0% |
| all | 7% |
| load | 12% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 6% |
| add-sub | 6% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 7% |
| all | 12% |
| load | 12% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 8% |
| load | 12% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 6% |
| add-sub | 6% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | 6% |
| other | 8% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| PUSH %RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| PUSH %R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
| PUSH %RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RDI,%RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| CALL 4032d0 <omp_get_num_threads@plt> | 2 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| MOV %EAX,%R12D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| CALL 403280 <omp_get_thread_num@plt> | 2 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| MOV %EAX,%ESI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (6.3%) |
| MOV 0x10(%RBX),%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| CLTD | scal (6.3%) | |||||||||||||||||||
| IDIV %R12D | 2 | 0 | 0 | 0 | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 9-14 | 6 | scal (6.3%) |
| CMP %EDX,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| JL 40eea0 <advanceVelocity._omp_fn.0+0x460> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| IMUL %EAX,%ESI | 1 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | scal (6.3%) |
| ADD %EDX,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| ADD %ESI,%EAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| CMP %EAX,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| JGE 40ee9b <advanceVelocity._omp_fn.0+0x45b> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV (%RBX),%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOVSXD %ESI,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VMOVSD 0x8(%RBX),%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (12.5%) |
| LEA (%R12,%R12,2),%R8 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| SAL $0x9,%R8 | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | scal (12.5%) |
| MOV 0x18(%R11),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV 0x78(%RCX),%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| NOPL (%RAX) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| POP %RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
| POP %R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
| POP %RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
| RET | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
| INC %EAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| XOR %EDX,%EDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (6.3%) |
| JMP 40ea68 <advanceVelocity._omp_fn.0+0x28> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| NOPL (%RAX) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
The code analyzed by CQA in that panel excludes loops and represents 0.00% of application time for run gcc_10
| Source file and lines | timestep.c:71-78 |
| Module | exec |
| nb instructions | 36 |
| nb uops | 35 |
| loop length | 117 |
| used x86 registers | 12 |
| used mmx registers | 0 |
| used xmm registers | 1 |
| used ymm registers | 0 |
| used zmm registers | 0 |
| nb stack references | 0 |
| micro-operation queue | 4.38 cycles |
| front end | 4.38 cycles |
| P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| uops | 2.67 | 2.67 | 2.67 | 3.00 | 3.00 | 3.00 | 1.75 | 1.75 | 1.75 | 1.75 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 |
| cycles | 2.67 | 2.67 | 2.67 | 3.00 | 3.00 | 3.00 | 1.75 | 1.75 | 1.75 | 1.75 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 |
| Cycles executing div or sqrt instructions | 6.00 |
| Front-end | 4.38 |
| Dispatch | 3.00 |
| DIV/SQRT | 6.00 |
| Overall L1 | 6.00 |
| all | 0% |
| load | 0% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 0% |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 0% |
| all | 0% |
| load | 0% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 0% |
| load | 0% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 0% |
| add-sub | 0% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | 0% |
| other | 0% |
| all | 7% |
| load | 12% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 6% |
| add-sub | 6% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| other | 7% |
| all | 12% |
| load | 12% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | NA (no mul vectorizable/vectorized instructions) |
| add-sub | NA (no add-sub vectorizable/vectorized instructions) |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
| other | NA (no other vectorizable/vectorized instructions) |
| all | 8% |
| load | 12% |
| store | NA (no store vectorizable/vectorized instructions) |
| mul | 6% |
| add-sub | 6% |
| fma | NA (no fma vectorizable/vectorized instructions) |
| div/sqrt | 6% |
| other | 8% |
| Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | P12 | P13 | P14 | P15 | Latency | Recip. throughput | Vectorization |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| PUSH %RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RSP,%RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (12.5%) |
| PUSH %R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
| PUSH %RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
| MOV %RDI,%RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| CALL 4032d0 <omp_get_num_threads@plt> | 2 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| MOV %EAX,%R12D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
| CALL 403280 <omp_get_thread_num@plt> | 2 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| MOV %EAX,%ESI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (6.3%) |
| MOV 0x10(%RBX),%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| CLTD | scal (6.3%) | |||||||||||||||||||
| IDIV %R12D | 2 | 0 | 0 | 0 | 2 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 9-14 | 6 | scal (6.3%) |
| CMP %EDX,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| JL 40eea0 <advanceVelocity._omp_fn.0+0x460> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| IMUL %EAX,%ESI | 1 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | scal (6.3%) |
| ADD %EDX,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| ADD %ESI,%EAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| CMP %EAX,%ESI | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | scal (6.3%) |
| JGE 40ee9b <advanceVelocity._omp_fn.0+0x45b> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33-1 | N/A |
| MOV (%RBX),%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOVSXD %ESI,%R12 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| VMOVSD 0x8(%RBX),%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (12.5%) |
| LEA (%R12,%R12,2),%R8 | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| SAL $0x9,%R8 | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 | scal (12.5%) |
| MOV 0x18(%R11),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | N/A |
| MOV 0x78(%RCX),%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.25 | scal (12.5%) |
| NOPW %CS:(%RAX,%RAX,1) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| NOPL (%RAX) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| POP %RBX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
| POP %R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
| POP %RBP | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
| RET | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
| INC %EAX | 1 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0.17 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 | N/A |
| XOR %EDX,%EDX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | scal (6.3%) |
| JMP 40ea68 <advanceVelocity._omp_fn.0+0x28> | 1 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | N/A |
| NOPL (%RAX) | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.13 | N/A |
| Name | Coverage (%) | Time (s) |
|---|---|---|
| ▼advanceVelocity._omp_fn.0– | 1.94 | 0.14 |
| ▼Loop 99 - timestep.c:74-78 - exec– | 0.69 | 0.03 |
| ○Loop 100 - timestep.c:74-78 - exec | 1.24 | 0.05 |
