Loop Id: 827 | Module: exec | Source: MultiBsplineRef.hpp:226-262 [...] | Coverage: 0.01% |
---|
Loop Id: 827 | Module: exec | Source: MultiBsplineRef.hpp:226-262 [...] | Coverage: 0.01% |
---|
0x443650 MOV -0x1b0(%RBP),%R12 |
0x443657 LEA 0x1(%R12),%RAX |
0x44365c MOV %RAX,-0x40(%RBP) |
0x443660 MOV -0x88(%RBP),%RAX |
0x443667 MOV -0x1d8(%RBP),%RDI |
0x44366e ADD %RAX,%RDI |
0x443671 MOV -0x1e8(%RBP),%R13 |
0x443678 ADD %RAX,%R13 |
0x44367b MOV -0x1e0(%RBP),%R15 |
0x443682 ADD %RAX,%R15 |
0x443685 ADD %RAX,-0x68(%RBP) |
0x443689 MOV -0x1d0(%RBP),%R10 |
0x443690 ADD %RAX,%R10 |
0x443693 MOV -0x1c8(%RBP),%R11 |
0x44369a ADD %RAX,%R11 |
0x44369d MOV -0x1c0(%RBP),%R8 |
0x4436a4 ADD %RAX,%R8 |
0x4436a7 MOV -0x1b8(%RBP),%RSI |
0x4436ae ADD %RAX,%RSI |
0x4436b1 CMP $0x3,%R12 |
0x4436b5 MOV -0x40(%RBP),%RAX |
0x4436b9 JE 443d70 |
0x4436bf MOVSD -0x280(%RBP,%RAX,8),%XMM0 |
0x4436c8 MOVSD %XMM0,-0x200(%RBP) |
0x4436d0 MOVSD -0x2c0(%RBP,%RAX,8),%XMM0 |
0x4436d9 MOVSD %XMM0,-0x1f8(%RBP) |
0x4436e1 MOV %RAX,-0x1b0(%RBP) |
0x4436e8 MOVSD -0x430(%RBP,%RAX,8),%XMM0 |
0x4436f1 MOVSD %XMM0,-0x1f0(%RBP) |
0x4436f9 MOV %RSI,-0x1b8(%RBP) |
0x443700 MOV %R8,-0x1c0(%RBP) |
0x443707 MOV %R8,%RAX |
0x44370a MOV %R11,-0x1c8(%RBP) |
0x443711 MOV %R11,-0xa0(%RBP) |
0x443718 MOV %R10,-0x1d0(%RBP) |
0x44371f MOV -0x68(%RBP),%R8 |
0x443723 MOV %R8,-0x120(%RBP) |
0x44372a MOV %R15,-0x1e0(%RBP) |
0x443731 MOV %R15,-0x118(%RBP) |
0x443738 MOV %R13,-0x1e8(%RBP) |
0x44373f MOV %RDI,-0x1d8(%RBP) |
0x443746 MOV %RDI,%R11 |
0x443749 XOR %EDI,%EDI |
0x44374b MOV %RDI,-0xa8(%RBP) |
0x443752 JMP 4437bc |
(828) 0x443760 MOV -0xa8(%RBP),%R12 |
(828) 0x443767 LEA 0x1(%R12),%R8 |
(828) 0x44376c MOV -0x98(%RBP),%RDI |
(828) 0x443773 MOV %R11,%R15 |
(828) 0x443776 MOV -0x128(%RBP),%R11 |
(828) 0x44377d ADD %RDI,%R11 |
(828) 0x443780 MOV -0x130(%RBP),%R13 |
(828) 0x443787 ADD %RDI,%R13 |
(828) 0x44378a ADD %RDI,-0x118(%RBP) |
(828) 0x443791 ADD %RDI,-0x120(%RBP) |
(828) 0x443798 ADD %RDI,%R10 |
(828) 0x44379b ADD %RDI,%R15 |
(828) 0x44379e MOV %R15,-0xa0(%RBP) |
(828) 0x4437a5 ADD %RDI,%RAX |
(828) 0x4437a8 ADD %RDI,%RSI |
(828) 0x4437ab CMP $0x3,%R12 |
(828) 0x4437af MOV %R8,-0xa8(%RBP) |
(828) 0x4437b6 JE 443650 |
(828) 0x4437bc MOV %R11,-0x128(%RBP) |
(828) 0x4437c3 MOV %R13,-0x130(%RBP) |
(828) 0x4437ca CMPL $0,-0x70(%RBP) |
(828) 0x4437ce MOV -0xa0(%RBP),%R11 |
(828) 0x4437d5 JLE 443760 |
(828) 0x4437d7 MOV -0xa8(%RBP),%RDI |
(828) 0x4437de MOVSD -0x410(%RBP,%RDI,8),%XMM6 |
(828) 0x4437e7 MOVAPD %XMM6,%XMM7 |
(828) 0x4437eb MULSD -0x200(%RBP),%XMM7 |
(828) 0x4437f3 MOVAPD %XMM6,%XMM3 |
(828) 0x4437f7 MOVSD -0x1f8(%RBP),%XMM0 |
(828) 0x4437ff MULSD %XMM0,%XMM3 |
(828) 0x443803 MOVSD -0x1f0(%RBP),%XMM1 |
(828) 0x44380b MULSD %XMM1,%XMM6 |
(828) 0x44380f MOVSD -0x2a0(%RBP,%RDI,8),%XMM13 |
(828) 0x443819 MOVAPD %XMM13,%XMM2 |
(828) 0x44381e MULSD %XMM0,%XMM2 |
(828) 0x443822 MULSD %XMM1,%XMM13 |
(828) 0x443827 MOVSD -0x260(%RBP,%RDI,8),%XMM15 |
(828) 0x443831 MULSD %XMM1,%XMM15 |
(828) 0x443836 MOV -0x90(%RBP),%RDI |
(828) 0x44383d TEST %RDI,%RDI |
(828) 0x443840 MOVAPD %XMM3,-0x40(%RBP) |
(828) 0x443845 JE 443b40 |
(828) 0x44384b MOV %RAX,-0x208(%RBP) |
(828) 0x443852 MOV %RSI,-0x210(%RBP) |
(828) 0x443859 MOVAPD %XMM7,%XMM1 |
(828) 0x44385d UNPCKLPD %XMM7,%XMM1 |
(828) 0x443861 MOVAPD %XMM1,-0x3b0(%RBP) |
(828) 0x443869 MOVAPD %XMM2,-0x2f0(%RBP) |
(828) 0x443871 UNPCKLPD %XMM2,%XMM2 |
(828) 0x443875 MOVAPD %XMM2,-0x150(%RBP) |
(828) 0x44387d UNPCKLPD %XMM3,%XMM3 |
(828) 0x443881 MOVAPD -0x2d0(%RBP),%XMM4 |
(828) 0x443889 MOVAPD %XMM15,-0x2e0(%RBP) |
(828) 0x443892 UNPCKLPD %XMM15,%XMM15 |
(828) 0x443897 MOVAPD %XMM13,-0x300(%RBP) |
(828) 0x4438a0 UNPCKLPD %XMM13,%XMM13 |
(828) 0x4438a5 MOVAPD %XMM6,%XMM12 |
(828) 0x4438aa UNPCKLPD %XMM6,%XMM12 |
(828) 0x4438af XOR %R13D,%R13D |
(828) 0x4438b2 MOV -0x50(%RBP),%R15 |
(828) 0x4438b6 MOV -0x118(%RBP),%RSI |
(828) 0x4438bd MOV -0x120(%RBP),%R12 |
(828) 0x4438c4 MOV -0x130(%RBP),%RAX |
(828) 0x4438cb MOV -0x128(%RBP),%R11 |
(828) 0x4438d2 NOPW %CS:(%RAX,%RAX,1) |
(830) 0x4438e0 MOVUPD (%R12,%R13,8),%XMM11 |
(830) 0x4438e6 MOVUPD (%RSI,%R13,8),%XMM1 |
(830) 0x4438ec MOVUPD (%RAX,%R13,8),%XMM14 |
(830) 0x4438f2 MOVUPD (%R11,%R13,8),%XMM2 |
(830) 0x4438f8 MOVAPD %XMM11,%XMM9 |
(830) 0x4438fd MULPD -0x3a0(%RBP),%XMM9 |
(830) 0x443906 MOVAPD %XMM1,%XMM10 |
(830) 0x44390b MULPD -0x390(%RBP),%XMM10 |
(830) 0x443914 ADDPD %XMM9,%XMM10 |
(830) 0x443919 MOVAPD %XMM14,%XMM8 |
(830) 0x44391e MULPD -0x380(%RBP),%XMM8 |
(830) 0x443927 MOVAPD -0x370(%RBP),%XMM9 |
(830) 0x443930 MULPD %XMM2,%XMM9 |
(830) 0x443935 ADDPD %XMM8,%XMM9 |
(830) 0x44393a ADDPD %XMM10,%XMM9 |
(830) 0x44393f MOVAPD %XMM11,%XMM8 |
(830) 0x443944 MULPD -0x360(%RBP),%XMM8 |
(830) 0x44394d MOVAPD %XMM1,%XMM0 |
(830) 0x443951 MULPD -0x350(%RBP),%XMM0 |
(830) 0x443959 MOVAPD %XMM14,%XMM5 |
(830) 0x44395e MULPD -0x340(%RBP),%XMM5 |
(830) 0x443966 ADDPD %XMM8,%XMM5 |
(830) 0x44396b MOVAPD %XMM2,%XMM10 |
(830) 0x443970 MULPD -0x240(%RBP),%XMM10 |
(830) 0x443979 ADDPD %XMM0,%XMM10 |
(830) 0x44397e MULPD %XMM4,%XMM10 |
(830) 0x443983 ADDPD %XMM5,%XMM10 |
(830) 0x443988 MULPD -0x330(%RBP),%XMM11 |
(830) 0x443991 MULPD -0x320(%RBP),%XMM1 |
(830) 0x443999 ADDPD %XMM11,%XMM1 |
(830) 0x44399e MULPD -0x310(%RBP),%XMM14 |
(830) 0x4439a7 MULPD %XMM4,%XMM2 |
(830) 0x4439ab ADDPD %XMM14,%XMM2 |
(830) 0x4439b0 ADDPD %XMM1,%XMM2 |
(830) 0x4439b4 MOVAPD -0x3b0(%RBP),%XMM0 |
(830) 0x4439bc MULPD %XMM9,%XMM0 |
(830) 0x4439c1 LEA (%RBX,%R13,8),%R8 |
(830) 0x4439c5 MOVUPD (%RBX,%R13,8),%XMM1 |
(830) 0x4439cb ADDPD %XMM0,%XMM1 |
(830) 0x4439cf MOVUPD %XMM1,(%RBX,%R13,8) |
(830) 0x4439d5 MOVAPD -0x150(%RBP),%XMM0 |
(830) 0x4439dd MULPD %XMM9,%XMM0 |
(830) 0x4439e2 MOVUPD (%R9,%R8,1),%XMM1 |
(830) 0x4439e8 ADDPD %XMM0,%XMM1 |
(830) 0x4439ec MOVUPD %XMM1,(%R9,%R8,1) |
(830) 0x4439f2 LEA (%R8,%R9,1),%R8 |
(830) 0x4439f6 MOVAPD %XMM10,%XMM0 |
(830) 0x4439fb MULPD %XMM3,%XMM0 |
(830) 0x4439ff MOVUPD (%R9,%R8,1),%XMM1 |
(830) 0x443a05 ADDPD %XMM0,%XMM1 |
(830) 0x443a09 MOVUPD %XMM1,(%R9,%R8,1) |
(830) 0x443a0f LEA (%R8,%R9,1),%R8 |
(830) 0x443a13 MOVAPD %XMM15,%XMM0 |
(830) 0x443a18 MULPD %XMM9,%XMM0 |
(830) 0x443a1d MOVUPD (%R9,%R8,1),%XMM1 |
(830) 0x443a23 ADDPD %XMM0,%XMM1 |
(830) 0x443a27 MOVUPD %XMM1,(%R9,%R8,1) |
(830) 0x443a2d LEA (%R8,%R9,1),%R8 |
(830) 0x443a31 MOVAPD %XMM10,%XMM0 |
(830) 0x443a36 MULPD %XMM13,%XMM0 |
(830) 0x443a3b MOVUPD (%R9,%R8,1),%XMM1 |
(830) 0x443a41 ADDPD %XMM0,%XMM1 |
(830) 0x443a45 MOVUPD %XMM1,(%R9,%R8,1) |
(830) 0x443a4b LEA (%R8,%R9,1),%R8 |
(830) 0x443a4f MULPD %XMM12,%XMM2 |
(830) 0x443a54 MOVUPD (%R9,%R8,1),%XMM0 |
(830) 0x443a5a ADDPD %XMM2,%XMM0 |
(830) 0x443a5e MOVUPD %XMM0,(%R9,%R8,1) |
(830) 0x443a64 MOVAPD %XMM9,%XMM0 |
(830) 0x443a69 MULPD %XMM3,%XMM0 |
(830) 0x443a6d MOVUPD (%R15,%R13,8),%XMM1 |
(830) 0x443a73 ADDPD %XMM0,%XMM1 |
(830) 0x443a77 MOVUPD %XMM1,(%R15,%R13,8) |
(830) 0x443a7d MOVAPD %XMM9,%XMM0 |
(830) 0x443a82 MULPD %XMM13,%XMM0 |
(830) 0x443a87 MOVUPD (%RDX,%R13,8),%XMM1 |
(830) 0x443a8d ADDPD %XMM0,%XMM1 |
(830) 0x443a91 MOVUPD %XMM1,(%RDX,%R13,8) |
(830) 0x443a97 MULPD %XMM12,%XMM10 |
(830) 0x443a9c MOVUPD (%RCX,%R13,8),%XMM0 |
(830) 0x443aa2 ADDPD %XMM10,%XMM0 |
(830) 0x443aa7 MOVUPD %XMM0,(%RCX,%R13,8) |
(830) 0x443aad MULPD %XMM12,%XMM9 |
(830) 0x443ab2 MOVUPD (%R14,%R13,8),%XMM0 |
(830) 0x443ab8 ADDPD %XMM9,%XMM0 |
(830) 0x443abd MOVUPD %XMM0,(%R14,%R13,8) |
(830) 0x443ac3 ADD $0x2,%R13 |
(830) 0x443ac7 CMP %RDI,%R13 |
(830) 0x443aca JL 4438e0 |
(828) 0x443ad0 MOV %RDI,%R13 |
(828) 0x443ad3 MOV %RDI,%R8 |
(828) 0x443ad6 MOV -0x70(%RBP),%RDI |
(828) 0x443ada CMP %RDI,%R8 |
(828) 0x443add MOVAPD -0xd0(%RBP),%XMM12 |
(828) 0x443ae6 MOVAPD -0x240(%RBP),%XMM5 |
(828) 0x443aee MOVAPD -0x80(%RBP),%XMM9 |
(828) 0x443af4 MOVAPD -0x60(%RBP),%XMM10 |
(828) 0x443afa MOVAPD -0xc0(%RBP),%XMM11 |
(828) 0x443b03 MOV -0x210(%RBP),%RSI |
(828) 0x443b0a MOV -0x208(%RBP),%RAX |
(828) 0x443b11 MOV -0xa0(%RBP),%R11 |
(828) 0x443b18 MOVAPD -0x300(%RBP),%XMM13 |
(828) 0x443b21 MOVAPD -0x2f0(%RBP),%XMM8 |
(828) 0x443b2a MOVAPD -0x2e0(%RBP),%XMM4 |
(828) 0x443b32 JNE 443b60 |
(828) 0x443b34 JMP 443760 |
(828) 0x443b40 XOR %R13D,%R13D |
(828) 0x443b43 MOV -0x70(%RBP),%RDI |
(828) 0x443b47 MOV -0x50(%RBP),%R15 |
(828) 0x443b4b MOVAPD %XMM2,%XMM8 |
(828) 0x443b50 MOVAPD %XMM15,%XMM4 |
(828) 0x443b55 NOPW %CS:(%RAX,%RAX,1) |
(829) 0x443b60 MOVAPD %XMM6,%XMM15 |
(829) 0x443b65 MOVAPD %XMM8,%XMM14 |
(829) 0x443b6a MOVSD (%RSI,%R13,8),%XMM8 |
(829) 0x443b70 MOVSD (%RAX,%R13,8),%XMM2 |
(829) 0x443b76 MOVSD (%R11,%R13,8),%XMM0 |
(829) 0x443b7c MOVSD (%R10,%R13,8),%XMM3 |
(829) 0x443b82 MOVAPD %XMM4,%XMM6 |
(829) 0x443b86 MOVAPD %XMM13,%XMM4 |
(829) 0x443b8b MOVAPD %XMM7,%XMM13 |
(829) 0x443b90 MOVAPD %XMM12,%XMM7 |
(829) 0x443b95 MOVAPD %XMM5,%XMM12 |
(829) 0x443b9a MOVAPD %XMM8,%XMM5 |
(829) 0x443b9f UNPCKLPD %XMM2,%XMM5 |
(829) 0x443ba3 MOVAPD %XMM0,%XMM1 |
(829) 0x443ba7 UNPCKLPD %XMM9,%XMM1 |
(829) 0x443bac MOVAPD -0x160(%RBP),%XMM9 |
(829) 0x443bb5 UNPCKLPD %XMM3,%XMM9 |
(829) 0x443bba MULPD %XMM1,%XMM9 |
(829) 0x443bbf MULPD %XMM10,%XMM5 |
(829) 0x443bc4 ADDPD %XMM9,%XMM5 |
(829) 0x443bc9 MOVAPD %XMM5,%XMM1 |
(829) 0x443bcd UNPCKHPD %XMM5,%XMM1 |
(829) 0x443bd1 ADDSD %XMM5,%XMM1 |
(829) 0x443bd5 MOVAPD %XMM8,%XMM5 |
(829) 0x443bda MULSD %XMM11,%XMM5 |
(829) 0x443bdf MOVAPD %XMM2,%XMM10 |
(829) 0x443be4 MULSD -0xe0(%RBP),%XMM10 |
(829) 0x443bed MOVAPD %XMM0,%XMM11 |
(829) 0x443bf2 MULSD -0x3f0(%RBP),%XMM11 |
(829) 0x443bfb ADDSD %XMM5,%XMM11 |
(829) 0x443c00 MOVAPD %XMM12,%XMM5 |
(829) 0x443c05 MOVAPD %XMM7,%XMM12 |
(829) 0x443c0a MOVAPD %XMM13,%XMM7 |
(829) 0x443c0f MOVAPD %XMM4,%XMM13 |
(829) 0x443c14 MOVAPD %XMM6,%XMM4 |
(829) 0x443c18 MOVAPD %XMM3,%XMM9 |
(829) 0x443c1d MULSD %XMM5,%XMM9 |
(829) 0x443c22 ADDSD %XMM10,%XMM9 |
(829) 0x443c27 MOVAPD -0x60(%RBP),%XMM10 |
(829) 0x443c2d MULSD %XMM12,%XMM9 |
(829) 0x443c32 ADDSD %XMM11,%XMM9 |
(829) 0x443c37 MOVAPD -0xc0(%RBP),%XMM11 |
(829) 0x443c40 MULSD -0x3e0(%RBP),%XMM8 |
(829) 0x443c49 MULSD -0x3d0(%RBP),%XMM2 |
(829) 0x443c51 ADDSD %XMM8,%XMM2 |
(829) 0x443c56 MOVAPD %XMM14,%XMM8 |
(829) 0x443c5b MOVAPD %XMM15,%XMM6 |
(829) 0x443c60 MOVAPD -0x40(%RBP),%XMM14 |
(829) 0x443c66 MULSD -0x3c0(%RBP),%XMM0 |
(829) 0x443c6e MULSD %XMM12,%XMM3 |
(829) 0x443c73 ADDSD %XMM0,%XMM3 |
(829) 0x443c77 MOVAPD %XMM7,%XMM0 |
(829) 0x443c7b MULSD %XMM1,%XMM0 |
(829) 0x443c7f ADDSD (%RBX,%R13,8),%XMM0 |
(829) 0x443c85 LEA (%RBX,%R13,8),%R8 |
(829) 0x443c89 MOVSD %XMM0,(%RBX,%R13,8) |
(829) 0x443c8f MOVAPD %XMM8,%XMM0 |
(829) 0x443c94 MULSD %XMM1,%XMM0 |
(829) 0x443c98 ADDSD (%R9,%R8,1),%XMM0 |
(829) 0x443c9e MOVSD %XMM0,(%R9,%R8,1) |
(829) 0x443ca4 LEA (%R8,%R9,1),%R8 |
(829) 0x443ca8 MOVAPD %XMM9,%XMM0 |
(829) 0x443cad MULSD %XMM14,%XMM0 |
(829) 0x443cb2 ADDSD (%R9,%R8,1),%XMM0 |
(829) 0x443cb8 MOVSD %XMM0,(%R9,%R8,1) |
(829) 0x443cbe LEA (%R8,%R9,1),%R8 |
(829) 0x443cc2 MOVAPD %XMM4,%XMM0 |
(829) 0x443cc6 MULSD %XMM1,%XMM0 |
(829) 0x443cca ADDSD (%R9,%R8,1),%XMM0 |
(829) 0x443cd0 MOVSD %XMM0,(%R9,%R8,1) |
(829) 0x443cd6 LEA (%R8,%R9,1),%R8 |
(829) 0x443cda MOVAPD %XMM9,%XMM0 |
(829) 0x443cdf MULSD %XMM13,%XMM0 |
(829) 0x443ce4 ADDSD (%R9,%R8,1),%XMM0 |
(829) 0x443cea ADDSD %XMM2,%XMM3 |
(829) 0x443cee MOVSD %XMM0,(%R9,%R8,1) |
(829) 0x443cf4 LEA (%R8,%R9,1),%R8 |
(829) 0x443cf8 MULSD %XMM15,%XMM3 |
(829) 0x443cfd ADDSD (%R9,%R8,1),%XMM3 |
(829) 0x443d03 MOVSD %XMM3,(%R9,%R8,1) |
(829) 0x443d09 MOVAPD %XMM1,%XMM0 |
(829) 0x443d0d MULSD %XMM14,%XMM0 |
(829) 0x443d12 ADDSD (%R15,%R13,8),%XMM0 |
(829) 0x443d18 MOVSD %XMM0,(%R15,%R13,8) |
(829) 0x443d1e MOVAPD %XMM1,%XMM0 |
(829) 0x443d22 MULSD %XMM13,%XMM0 |
(829) 0x443d27 ADDSD (%RDX,%R13,8),%XMM0 |
(829) 0x443d2d MOVSD %XMM0,(%RDX,%R13,8) |
(829) 0x443d33 MULSD %XMM15,%XMM9 |
(829) 0x443d38 ADDSD (%RCX,%R13,8),%XMM9 |
(829) 0x443d3e MOVSD %XMM9,(%RCX,%R13,8) |
(829) 0x443d44 MOVAPD -0x80(%RBP),%XMM9 |
(829) 0x443d4a MULSD %XMM15,%XMM1 |
(829) 0x443d4f ADDSD (%R14,%R13,8),%XMM1 |
(829) 0x443d55 MOVSD %XMM1,(%R14,%R13,8) |
(829) 0x443d5b INC %R13 |
(829) 0x443d5e CMP %R13,%RDI |
(829) 0x443d61 JNE 443b60 |
(828) 0x443d67 JMP 443760 |
/beegfs/hackathon/users/eoseret/qaas_runs/170-855-3059/intel/miniqmc/build/miniqmc/src/Numerics/OhmmsPETE/TinyVector.h: 61 - 61 |
-------------------------------------------------------------------------------- |
61: for (size_t d = 0; d < D; ++d) |
/beegfs/hackathon/users/eoseret/qaas_runs/170-855-3059/intel/miniqmc/build/miniqmc/src/Numerics/Spline2/MultiBsplineRef.hpp: 226 - 262 |
-------------------------------------------------------------------------------- |
226: for (int i = 0; i < 4; i++) |
227: for (int j = 0; j < 4; j++) |
[...] |
234: const T pre20 = d2a[i] * b[j]; |
235: const T pre10 = da[i] * b[j]; |
236: const T pre00 = a[i] * b[j]; |
237: const T pre11 = da[i] * db[j]; |
238: const T pre01 = a[i] * db[j]; |
239: const T pre02 = a[i] * d2b[j]; |
240: |
241: const int iSplitPoint = num_splines; |
242: for (int n = 0; n < iSplitPoint; n++) |
243: { |
244: T coefsv = coefs[n]; |
245: T coefsvzs = coefszs[n]; |
246: T coefsv2zs = coefs2zs[n]; |
247: T coefsv3zs = coefs3zs[n]; |
248: |
249: T sum0 = c[0] * coefsv + c[1] * coefsvzs + c[2] * coefsv2zs + c[3] * coefsv3zs; |
250: T sum1 = dc[0] * coefsv + dc[1] * coefsvzs + dc[2] * coefsv2zs + dc[3] * coefsv3zs; |
251: T sum2 = d2c[0] * coefsv + d2c[1] * coefsvzs + d2c[2] * coefsv2zs + d2c[3] * coefsv3zs; |
252: |
253: hxx[n] += pre20 * sum0; |
254: hxy[n] += pre11 * sum0; |
255: hxz[n] += pre10 * sum1; |
256: hyy[n] += pre02 * sum0; |
257: hyz[n] += pre01 * sum1; |
258: hzz[n] += pre00 * sum2; |
259: gx[n] += pre10 * sum0; |
260: gy[n] += pre01 * sum0; |
261: gz[n] += pre00 * sum1; |
262: vals[n] += pre00 * sum0; |
Path / |
Metric | Value |
---|---|
CQA speedup if no scalar integer | 4.43 |
CQA speedup if FP arith vectorized | 1.00 |
CQA speedup if fully vectorized | 5.51 |
CQA speedup if no inter-iteration dependency | NA |
CQA speedup if next bottleneck killed | 1.35 |
Bottlenecks | P5, P6, P7, |
Function | _ZN16miniqmcreference17einspline_spo_refIdE8evaluateERKN11qmcplusplus11ParticleSetEiRNS2_6VectorIdSaIdEEERNS6_INS2_10TinyVectorIdLj3EEESaISB_EEES9_ |
Source | MultiBsplineRef.hpp:226-226,MultiBsplineRef.hpp:234-236,MultiBsplineRef.hpp:244-244 |
Source loop unroll info | NA |
Source loop unroll confidence level | NA |
Unroll/vectorization loop type | NA |
Unroll factor | NA |
CQA cycles | 10.33 |
CQA cycles if no scalar integer | 2.33 |
CQA cycles if FP arith vectorized | 10.33 |
CQA cycles if fully vectorized | 1.88 |
Front-end cycles | 7.67 |
DIV/SQRT cycles | 2.75 |
P0 cycles | 2.75 |
P1 cycles | 2.75 |
P2 cycles | 2.75 |
P3 cycles | 1.00 |
P4 cycles | 10.33 |
P5 cycles | 10.33 |
P6 cycles | 10.33 |
P7 cycles | 0.00 |
P8 cycles | 0.00 |
P9 cycles | 0.00 |
P10 cycles | 0.00 |
P11 cycles | 1.50 |
P12 cycles | 1.50 |
P13 cycles | 0.00 |
Inter-iter dependencies cycles | 0 |
FE+BE cycles (UFS) | NA |
Stall cycles (UFS) | NA |
Nb insns | 45.00 |
Nb uops | 46.00 |
Nb loads | 15.00 |
Nb stores | 17.00 |
Nb stack references | 18.00 |
FLOP/cycle | 0.00 |
Nb FLOP add-sub | 0.00 |
Nb FLOP mul | 0.00 |
Nb FLOP fma | 0.00 |
Nb FLOP div | 0.00 |
Nb FLOP rcp | 0.00 |
Nb FLOP sqrt | 0.00 |
Nb FLOP rsqrt | 0.00 |
Bytes/cycle | 24.77 |
Bytes prefetched | 0.00 |
Bytes loaded | 120.00 |
Bytes stored | 136.00 |
Stride 0 | 1.00 |
Stride 1 | 0.00 |
Stride n | 0.00 |
Stride unknown | 1.00 |
Stride indirect | 0.00 |
Vectorization ratio all | 0.00 |
Vectorization ratio load | 0.00 |
Vectorization ratio store | 0.00 |
Vectorization ratio mul | NA |
Vectorization ratio add_sub | 0.00 |
Vectorization ratio fma | NA |
Vectorization ratio div_sqrt | NA |
Vectorization ratio other | 0.00 |
Vector-efficiency ratio all | 12.33 |
Vector-efficiency ratio load | 12.50 |
Vector-efficiency ratio store | 12.50 |
Vector-efficiency ratio mul | NA |
Vector-efficiency ratio add_sub | 12.50 |
Vector-efficiency ratio fma | NA |
Vector-efficiency ratio div_sqrt | NA |
Vector-efficiency ratio other | 9.38 |
Metric | Value |
---|---|
CQA speedup if no scalar integer | 4.43 |
CQA speedup if FP arith vectorized | 1.00 |
CQA speedup if fully vectorized | 5.51 |
CQA speedup if no inter-iteration dependency | NA |
CQA speedup if next bottleneck killed | 1.35 |
Bottlenecks | P5, P6, P7, |
Function | _ZN16miniqmcreference17einspline_spo_refIdE8evaluateERKN11qmcplusplus11ParticleSetEiRNS2_6VectorIdSaIdEEERNS6_INS2_10TinyVectorIdLj3EEESaISB_EEES9_ |
Source | MultiBsplineRef.hpp:226-226,MultiBsplineRef.hpp:234-236,MultiBsplineRef.hpp:244-244 |
Source loop unroll info | NA |
Source loop unroll confidence level | NA |
Unroll/vectorization loop type | NA |
Unroll factor | NA |
CQA cycles | 10.33 |
CQA cycles if no scalar integer | 2.33 |
CQA cycles if FP arith vectorized | 10.33 |
CQA cycles if fully vectorized | 1.88 |
Front-end cycles | 7.67 |
DIV/SQRT cycles | 2.75 |
P0 cycles | 2.75 |
P1 cycles | 2.75 |
P2 cycles | 2.75 |
P3 cycles | 1.00 |
P4 cycles | 10.33 |
P5 cycles | 10.33 |
P6 cycles | 10.33 |
P7 cycles | 0.00 |
P8 cycles | 0.00 |
P9 cycles | 0.00 |
P10 cycles | 0.00 |
P11 cycles | 1.50 |
P12 cycles | 1.50 |
P13 cycles | 0.00 |
Inter-iter dependencies cycles | 0 |
FE+BE cycles (UFS) | NA |
Stall cycles (UFS) | NA |
Nb insns | 45.00 |
Nb uops | 46.00 |
Nb loads | 15.00 |
Nb stores | 17.00 |
Nb stack references | 18.00 |
FLOP/cycle | 0.00 |
Nb FLOP add-sub | 0.00 |
Nb FLOP mul | 0.00 |
Nb FLOP fma | 0.00 |
Nb FLOP div | 0.00 |
Nb FLOP rcp | 0.00 |
Nb FLOP sqrt | 0.00 |
Nb FLOP rsqrt | 0.00 |
Bytes/cycle | 24.77 |
Bytes prefetched | 0.00 |
Bytes loaded | 120.00 |
Bytes stored | 136.00 |
Stride 0 | 1.00 |
Stride 1 | 0.00 |
Stride n | 0.00 |
Stride unknown | 1.00 |
Stride indirect | 0.00 |
Vectorization ratio all | 0.00 |
Vectorization ratio load | 0.00 |
Vectorization ratio store | 0.00 |
Vectorization ratio mul | NA |
Vectorization ratio add_sub | 0.00 |
Vectorization ratio fma | NA |
Vectorization ratio div_sqrt | NA |
Vectorization ratio other | 0.00 |
Vector-efficiency ratio all | 12.33 |
Vector-efficiency ratio load | 12.50 |
Vector-efficiency ratio store | 12.50 |
Vector-efficiency ratio mul | NA |
Vector-efficiency ratio add_sub | 12.50 |
Vector-efficiency ratio fma | NA |
Vector-efficiency ratio div_sqrt | NA |
Vector-efficiency ratio other | 9.38 |
Path / |
Function | _ZN16miniqmcreference17einspline_spo_refIdE8evaluateERKN11qmcplusplus11ParticleSetEiRNS2_6VectorIdSaIdEEERNS6_INS2_10TinyVectorIdLj3EEESaISB_EEES9_ |
Source file and lines | MultiBsplineRef.hpp:226-262 |
Module | exec |
nb instructions | 45 |
nb uops | 46 |
loop length | 260 |
used x86 registers | 10 |
used mmx registers | 0 |
used xmm registers | 1 |
used ymm registers | 0 |
used zmm registers | 0 |
nb stack references | 18 |
micro-operation queue | 7.67 cycles |
front end | 7.67 cycles |
ALU0/BRU0 | ALU1 | ALU2 | ALU3 | BRU1 | AGU0 | AGU1 | AGU2 | FP0 | FP1 | FP2 | FP3 | FP4 | FP5 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 2.75 | 2.75 | 2.75 | 2.75 | 1.00 | 10.33 | 10.33 | 10.33 | 0.00 | 0.00 | 0.00 | 0.00 | 1.50 | 1.50 |
cycles | 2.75 | 2.75 | 2.75 | 2.75 | 1.00 | 10.33 | 10.33 | 10.33 | 0.00 | 0.00 | 0.00 | 0.00 | 1.50 | 1.50 |
Cycles executing div or sqrt instructions | NA |
Longest recurrence chain latency (RecMII) | 0.00 |
Front-end | 7.67 |
Dispatch | 10.33 |
Data deps. | 0.00 |
Overall L1 | 10.33 |
all | 0% |
load | 0% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 0% |
all | 0% |
load | 0% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | NA (no add-sub vectorizable/vectorized instructions) |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | NA (no other vectorizable/vectorized instructions) |
all | 0% |
load | 0% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 0% |
all | 12% |
load | 12% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 12% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 9% |
all | 12% |
load | 12% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | NA (no add-sub vectorizable/vectorized instructions) |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | NA (no other vectorizable/vectorized instructions) |
all | 12% |
load | 12% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 12% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 9% |
Instruction | Nb FU | ALU0/BRU0 | ALU1 | ALU2 | ALU3 | BRU1 | AGU0 | AGU1 | AGU2 | FP0 | FP1 | FP2 | FP3 | FP4 | FP5 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
MOV -0x1b0(%RBP),%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
LEA 0x1(%R12),%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV %RAX,-0x40(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV -0x88(%RBP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
MOV -0x1d8(%RBP),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x1e8(%RBP),%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%R13 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x1e0(%RBP),%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%R15 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
ADD %RAX,-0x68(%RBP) | 2 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
MOV -0x1d0(%RBP),%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%R10 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x1c8(%RBP),%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%R11 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x1c0(%RBP),%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%R8 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x1b8(%RBP),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%RSI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
CMP $0x3,%R12 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x40(%RBP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
JE 443d70 <_ZN16miniqmcreference17einspline_spo_refIdE8evaluateERKN11qmcplusplus11ParticleSetEiRNS2_6VectorIdSaIdEEERNS6_INS2_10TinyVectorIdLj3EEESaISB_EEES9_+0x11c0> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 |
MOVSD -0x280(%RBP,%RAX,8),%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
MOVSD %XMM0,-0x200(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 1 | 1 |
MOVSD -0x2c0(%RBP,%RAX,8),%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
MOVSD %XMM0,-0x1f8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 1 | 1 |
MOV %RAX,-0x1b0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOVSD -0x430(%RBP,%RAX,8),%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
MOVSD %XMM0,-0x1f0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 1 | 1 |
MOV %RSI,-0x1b8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R8,-0x1c0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R8,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV %R11,-0x1c8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R11,-0xa0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R10,-0x1d0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV -0x68(%RBP),%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
MOV %R8,-0x120(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R15,-0x1e0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R15,-0x118(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R13,-0x1e8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %RDI,-0x1d8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %RDI,%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %EDI,%EDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 |
MOV %RDI,-0xa8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
JMP 4437bc <_ZN16miniqmcreference17einspline_spo_refIdE8evaluateERKN11qmcplusplus11ParticleSetEiRNS2_6VectorIdSaIdEEERNS6_INS2_10TinyVectorIdLj3EEESaISB_EEES9_+0xc0c> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 |
Function | _ZN16miniqmcreference17einspline_spo_refIdE8evaluateERKN11qmcplusplus11ParticleSetEiRNS2_6VectorIdSaIdEEERNS6_INS2_10TinyVectorIdLj3EEESaISB_EEES9_ |
Source file and lines | MultiBsplineRef.hpp:226-262 |
Module | exec |
nb instructions | 45 |
nb uops | 46 |
loop length | 260 |
used x86 registers | 10 |
used mmx registers | 0 |
used xmm registers | 1 |
used ymm registers | 0 |
used zmm registers | 0 |
nb stack references | 18 |
micro-operation queue | 7.67 cycles |
front end | 7.67 cycles |
ALU0/BRU0 | ALU1 | ALU2 | ALU3 | BRU1 | AGU0 | AGU1 | AGU2 | FP0 | FP1 | FP2 | FP3 | FP4 | FP5 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 2.75 | 2.75 | 2.75 | 2.75 | 1.00 | 10.33 | 10.33 | 10.33 | 0.00 | 0.00 | 0.00 | 0.00 | 1.50 | 1.50 |
cycles | 2.75 | 2.75 | 2.75 | 2.75 | 1.00 | 10.33 | 10.33 | 10.33 | 0.00 | 0.00 | 0.00 | 0.00 | 1.50 | 1.50 |
Cycles executing div or sqrt instructions | NA |
Longest recurrence chain latency (RecMII) | 0.00 |
Front-end | 7.67 |
Dispatch | 10.33 |
Data deps. | 0.00 |
Overall L1 | 10.33 |
all | 0% |
load | 0% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 0% |
all | 0% |
load | 0% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | NA (no add-sub vectorizable/vectorized instructions) |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | NA (no other vectorizable/vectorized instructions) |
all | 0% |
load | 0% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 0% |
all | 12% |
load | 12% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 12% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 9% |
all | 12% |
load | 12% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | NA (no add-sub vectorizable/vectorized instructions) |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | NA (no other vectorizable/vectorized instructions) |
all | 12% |
load | 12% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 12% |
fma | NA (no fma vectorizable/vectorized instructions) |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 9% |
Instruction | Nb FU | ALU0/BRU0 | ALU1 | ALU2 | ALU3 | BRU1 | AGU0 | AGU1 | AGU2 | FP0 | FP1 | FP2 | FP3 | FP4 | FP5 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
MOV -0x1b0(%RBP),%R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
LEA 0x1(%R12),%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV %RAX,-0x40(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV -0x88(%RBP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
MOV -0x1d8(%RBP),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x1e8(%RBP),%R13 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%R13 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x1e0(%RBP),%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%R15 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
ADD %RAX,-0x68(%RBP) | 2 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
MOV -0x1d0(%RBP),%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%R10 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x1c8(%RBP),%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%R11 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x1c0(%RBP),%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%R8 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x1b8(%RBP),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
ADD %RAX,%RSI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
CMP $0x3,%R12 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 |
MOV -0x40(%RBP),%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
JE 443d70 <_ZN16miniqmcreference17einspline_spo_refIdE8evaluateERKN11qmcplusplus11ParticleSetEiRNS2_6VectorIdSaIdEEERNS6_INS2_10TinyVectorIdLj3EEESaISB_EEES9_+0x11c0> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 |
MOVSD -0x280(%RBP,%RAX,8),%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
MOVSD %XMM0,-0x200(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 1 | 1 |
MOVSD -0x2c0(%RBP,%RAX,8),%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
MOVSD %XMM0,-0x1f8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 1 | 1 |
MOV %RAX,-0x1b0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOVSD -0x430(%RBP,%RAX,8),%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 |
MOVSD %XMM0,-0x1f0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 1 | 1 |
MOV %RSI,-0x1b8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R8,-0x1c0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R8,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
MOV %R11,-0x1c8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R11,-0xa0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R10,-0x1d0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV -0x68(%RBP),%R8 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 |
MOV %R8,-0x120(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R15,-0x1e0(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R15,-0x118(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %R13,-0x1e8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %RDI,-0x1d8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
MOV %RDI,%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %EDI,%EDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 |
MOV %RDI,-0xa8(%RBP) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 4 | 0.50 |
JMP 4437bc <_ZN16miniqmcreference17einspline_spo_refIdE8evaluateERKN11qmcplusplus11ParticleSetEiRNS2_6VectorIdSaIdEEERNS6_INS2_10TinyVectorIdLj3EEESaISB_EEES9_+0xc0c> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 |