Loop Id: 3072 | Module: exec | Source: csr_matvec.c:256-263 | Coverage: 6.56% |
---|
Loop Id: 3072 | Module: exec | Source: csr_matvec.c:256-263 | Coverage: 6.56% |
---|
0x57f5c0 SUB %RDX,%RCX |
0x57f5c3 MOV %RDX,%R15 |
0x57f5c6 LEA -0x1(%RCX),%RSI |
0x57f5ca CMP $0x2,%RSI |
0x57f5ce JBE 58082c |
0x57f5d4 MOV %RCX,%R11 |
0x57f5d7 LEA (,%RDX,8),%RSI |
0x57f5df VXORPD %XMM1,%XMM1,%XMM1 |
0x57f5e3 XOR %EAX,%EAX |
0x57f5e5 SHR $0x2,%R11 |
0x57f5e9 LEA (%R14,%RSI,1),%R10 |
0x57f5ed ADD %R13,%RSI |
0x57f5f0 SAL $0x5,%R11 |
0x57f5f4 LEA -0x20(%R11),%RDI |
0x57f5f8 SHR $0x5,%RDI |
0x57f5fc INC %RDI |
0x57f5ff AND $0x7,%EDI |
0x57f602 JE 57f6f0 |
0x57f608 CMP $0x1,%RDI |
0x57f60c JE 57f6ce |
0x57f612 CMP $0x2,%RDI |
0x57f616 JE 57f6b4 |
0x57f61c CMP $0x3,%RDI |
0x57f620 JE 57f69a |
0x57f622 CMP $0x4,%RDI |
0x57f626 JE 57f680 |
0x57f628 CMP $0x5,%RDI |
0x57f62c JE 57f667 |
0x57f62e CMP $0x6,%RDI |
0x57f632 JE 57f64d |
0x57f634 VMOVDQU (%RSI),%YMM9 |
0x57f638 VMOVAPD %YMM15,%YMM10 |
0x57f63d MOV $0x20,%EAX |
0x57f642 VGATHERQPD %YMM10,(%RBX,%YMM9,8),%YMM8 |
0x57f648 VFMADD231PD (%R10),%YMM8,%YMM1 |
0x57f64d VMOVDQU (%RSI,%RAX,1),%YMM14 |
0x57f652 VMOVAPD %YMM15,%YMM13 |
0x57f657 VGATHERQPD %YMM13,(%RBX,%YMM14,8),%YMM0 |
0x57f65d VFMADD231PD (%R10,%RAX,1),%YMM0,%YMM1 |
0x57f663 ADD $0x20,%RAX |
0x57f667 VMOVDQU (%RSI,%RAX,1),%YMM5 |
0x57f66c VMOVAPD %YMM15,%YMM7 |
0x57f670 VGATHERQPD %YMM7,(%RBX,%YMM5,8),%YMM3 |
0x57f676 VFMADD231PD (%R10,%RAX,1),%YMM3,%YMM1 |
0x57f67c ADD $0x20,%RAX |
0x57f680 VMOVDQU (%RSI,%RAX,1),%YMM4 |
0x57f685 VMOVAPD %YMM15,%YMM11 |
0x57f68a VGATHERQPD %YMM11,(%RBX,%YMM4,8),%YMM6 |
0x57f690 VFMADD231PD (%R10,%RAX,1),%YMM6,%YMM1 |
0x57f696 ADD $0x20,%RAX |
0x57f69a VMOVDQU (%RSI,%RAX,1),%YMM9 |
0x57f69f VMOVAPD %YMM15,%YMM10 |
0x57f6a4 VGATHERQPD %YMM10,(%RBX,%YMM9,8),%YMM8 |
0x57f6aa VFMADD231PD (%R10,%RAX,1),%YMM8,%YMM1 |
0x57f6b0 ADD $0x20,%RAX |
0x57f6b4 VMOVDQU (%RSI,%RAX,1),%YMM14 |
0x57f6b9 VMOVAPD %YMM15,%YMM13 |
0x57f6be VGATHERQPD %YMM13,(%RBX,%YMM14,8),%YMM0 |
0x57f6c4 VFMADD231PD (%R10,%RAX,1),%YMM0,%YMM1 |
0x57f6ca ADD $0x20,%RAX |
0x57f6ce VMOVDQU (%RSI,%RAX,1),%YMM7 |
0x57f6d3 VMOVAPD %YMM15,%YMM5 |
0x57f6d7 VGATHERQPD %YMM5,(%RBX,%YMM7,8),%YMM3 |
0x57f6dd VFMADD231PD (%R10,%RAX,1),%YMM3,%YMM1 |
0x57f6e3 ADD $0x20,%RAX |
0x57f6e7 CMP %RAX,%R11 |
0x57f6ea JE 57f7d1 |
(3073) 0x57f6f0 VMOVDQU (%RSI,%RAX,1),%YMM4 |
(3073) 0x57f6f5 VMOVDQU 0x20(%RSI,%RAX,1),%YMM9 |
(3073) 0x57f6fb VMOVAPD %YMM15,%YMM11 |
(3073) 0x57f700 VMOVAPD %YMM15,%YMM10 |
(3073) 0x57f705 VMOVDQU 0x40(%RSI,%RAX,1),%YMM14 |
(3073) 0x57f70b VMOVAPD %YMM15,%YMM13 |
(3073) 0x57f710 VMOVAPD %YMM15,%YMM7 |
(3073) 0x57f714 VMOVAPD %YMM15,%YMM3 |
(3073) 0x57f718 VGATHERQPD %YMM11,(%RBX,%YMM4,8),%YMM6 |
(3073) 0x57f71e VFMADD231PD (%R10,%RAX,1),%YMM6,%YMM1 |
(3073) 0x57f724 VMOVDQU 0x60(%RSI,%RAX,1),%YMM5 |
(3073) 0x57f72a VGATHERQPD %YMM10,(%RBX,%YMM9,8),%YMM8 |
(3073) 0x57f730 VFMADD231PD 0x20(%R10,%RAX,1),%YMM8,%YMM1 |
(3073) 0x57f737 VMOVAPD %YMM15,%YMM4 |
(3073) 0x57f73b VGATHERQPD %YMM13,(%RBX,%YMM14,8),%YMM0 |
(3073) 0x57f741 VMOVDQU 0x80(%RSI,%RAX,1),%YMM11 |
(3073) 0x57f74a VFMADD132PD 0x40(%R10,%RAX,1),%YMM1,%YMM0 |
(3073) 0x57f751 VMOVAPD %YMM15,%YMM8 |
(3073) 0x57f756 VMOVDQU 0xa0(%RSI,%RAX,1),%YMM10 |
(3073) 0x57f75f VGATHERQPD %YMM7,(%RBX,%YMM5,8),%YMM1 |
(3073) 0x57f765 VMOVDQU 0xc0(%RSI,%RAX,1),%YMM13 |
(3073) 0x57f76e VFMADD132PD 0x60(%R10,%RAX,1),%YMM0,%YMM1 |
(3073) 0x57f775 VGATHERQPD %YMM3,(%RBX,%YMM11,8),%YMM6 |
(3073) 0x57f77b VMOVAPD %YMM15,%YMM0 |
(3073) 0x57f77f VFMADD132PD 0x80(%R10,%RAX,1),%YMM1,%YMM6 |
(3073) 0x57f789 VMOVDQU 0xe0(%RSI,%RAX,1),%YMM7 |
(3073) 0x57f792 VGATHERQPD %YMM4,(%RBX,%YMM10,8),%YMM9 |
(3073) 0x57f798 VFMADD132PD 0xa0(%R10,%RAX,1),%YMM6,%YMM9 |
(3073) 0x57f7a2 VGATHERQPD %YMM8,(%RBX,%YMM13,8),%YMM14 |
(3073) 0x57f7a8 VFMADD132PD 0xc0(%R10,%RAX,1),%YMM9,%YMM14 |
(3073) 0x57f7b2 VGATHERQPD %YMM0,(%RBX,%YMM7,8),%YMM1 |
(3073) 0x57f7b8 VFMADD132PD 0xe0(%R10,%RAX,1),%YMM14,%YMM1 |
(3073) 0x57f7c2 ADD $0x100,%RAX |
(3073) 0x57f7c8 CMP %RAX,%R11 |
(3073) 0x57f7cb JNE 57f6f0 |
0x57f7d1 VEXTRACTF128 $0x1,%YMM1,%XMM3 |
0x57f7d7 VADDPD %XMM1,%XMM3,%XMM5 |
0x57f7db VUNPCKHPD %XMM5,%XMM5,%XMM11 |
0x57f7df VADDPD %XMM5,%XMM11,%XMM8 |
0x57f7e3 TEST $0x3,%CL |
0x57f7e6 JE 57f83f |
0x57f7e8 MOV %RCX,%R10 |
0x57f7eb VADDPD %XMM1,%XMM3,%XMM6 |
0x57f7ef AND $-0x4,%R10 |
0x57f7f3 ADD %R10,%RDX |
0x57f7f6 SUB %R10,%RCX |
0x57f7f9 CMP $0x1,%RCX |
0x57f7fd JE 57f82f |
0x57f7ff ADD %R15,%R10 |
0x57f802 VMOVAPD %XMM12,%XMM4 |
0x57f806 VMOVDQU (%R13,%R10,8),%XMM10 |
0x57f80d VGATHERQPD %XMM4,(%RBX,%XMM10,8),%XMM9 |
0x57f813 VFMADD132PD (%R14,%R10,8),%XMM6,%XMM9 |
0x57f819 VUNPCKHPD %XMM9,%XMM9,%XMM6 |
0x57f81e VADDPD %XMM9,%XMM6,%XMM8 |
0x57f823 TEST $0x1,%CL |
0x57f826 JE 57f83f |
0x57f828 AND $-0x2,%RCX |
0x57f82c ADD %RCX,%RDX |
0x57f82f MOV (%R13,%RDX,8),%RCX |
0x57f834 VMOVSD (%RBX,%RCX,8),%XMM13 |
0x57f839 VFMADD231SD (%R14,%RDX,8),%XMM13,%XMM8 |
0x57f83f MOV 0x38(%RSP),%RDX |
0x57f844 VMOVSD %XMM8,(%RDX,%R12,8) |
0x57f84a INC %R12 |
0x57f84d CMP %R12,%R9 |
0x57f850 JE 57f578 |
0x57f856 MOV (%R8,%R12,8),%RDX |
0x57f85a MOV 0x8(%R8,%R12,8),%RCX |
0x57f85f CMP %RDX,%RCX |
0x57f862 JG 57f5c0 |
0x58082c VMOVSD %XMM2,%XMM2,%XMM8 |
0x580830 VXORPD %XMM6,%XMM6,%XMM6 |
0x580834 XOR %R10D,%R10D |
0x580837 JMP 57f7f6 |
/scratch_na/users/xoserete/qaas_runs/171-172-8217/intel/AMG/build/AMG/AMG/seq_mv/csr_matvec.c: 256 - 263 |
-------------------------------------------------------------------------------- |
256: for (i = iBegin; i < iEnd; i++) |
257: { |
258: tempx = 0.0; |
259: for (jj = A_i[i]; jj < A_i[i+1]; jj++) |
260: { |
261: tempx += A_data[jj] * x_data[A_j[jj]]; |
262: } |
263: y_data[i] = tempx; |
Path / |
Metric | Value |
---|---|
CQA speedup if no scalar integer | 1.35 |
CQA speedup if FP arith vectorized | 1.52 |
CQA speedup if fully vectorized | 3.27 |
CQA speedup if no inter-iteration dependency | NA |
CQA speedup if next bottleneck killed | 1.39 |
Bottlenecks | micro-operation queue, |
Function | hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6 |
Source | csr_matvec.c:256-263 |
Source loop unroll info | NA |
Source loop unroll confidence level | NA |
Unroll/vectorization loop type | NA |
Unroll factor | NA |
CQA cycles | 24.50 |
CQA cycles if no scalar integer | 18.17 |
CQA cycles if FP arith vectorized | 16.17 |
CQA cycles if fully vectorized | 7.50 |
Front-end cycles | 24.50 |
DIV/SQRT cycles | 17.63 |
P0 cycles | 17.63 |
P1 cycles | 17.33 |
P2 cycles | 17.33 |
P3 cycles | 0.50 |
P4 cycles | 17.63 |
P5 cycles | 17.50 |
P6 cycles | 0.50 |
P7 cycles | 0.50 |
P8 cycles | 0.50 |
P9 cycles | 17.60 |
P10 cycles | 17.33 |
P11 cycles | 0.00 |
Inter-iter dependencies cycles | NA |
FE+BE cycles (UFS) | 31.93 - 240.32 |
Stall cycles (UFS) | 6.75 - 215.14 |
Nb insns | 107.00 |
Nb uops | 139.00 |
Nb loads | 30.00 |
Nb stores | 1.00 |
Nb stack references | 1.00 |
FLOP/cycle | 2.86 |
Nb FLOP add-sub | 8.00 |
Nb FLOP mul | 0.00 |
Nb FLOP fma | 31.00 |
Nb FLOP div | 0.00 |
Nb FLOP rcp | 0.00 |
Nb FLOP sqrt | 0.00 |
Nb FLOP rsqrt | 0.00 |
Bytes/cycle | 31.67 |
Bytes prefetched | 0.00 |
Bytes loaded | 768.00 |
Bytes stored | 8.00 |
Stride 0 | NA |
Stride 1 | NA |
Stride n | NA |
Stride unknown | NA |
Stride indirect | NA |
Vectorization ratio all | 66.10 |
Vectorization ratio load | 92.31 |
Vectorization ratio store | 0.00 |
Vectorization ratio mul | NA |
Vectorization ratio add_sub | 80.00 |
Vectorization ratio fma | 88.89 |
Vectorization ratio div_sqrt | NA |
Vectorization ratio other | 54.29 |
Vector-efficiency ratio all | 32.52 |
Vector-efficiency ratio load | 44.23 |
Vector-efficiency ratio store | 12.50 |
Vector-efficiency ratio mul | NA |
Vector-efficiency ratio add_sub | 22.50 |
Vector-efficiency ratio fma | 43.06 |
Vector-efficiency ratio div_sqrt | NA |
Vector-efficiency ratio other | 29.11 |
Metric | Value |
---|---|
CQA speedup if no scalar integer | 1.35 |
CQA speedup if FP arith vectorized | 1.52 |
CQA speedup if fully vectorized | 3.27 |
CQA speedup if no inter-iteration dependency | NA |
CQA speedup if next bottleneck killed | 1.39 |
Bottlenecks | micro-operation queue, |
Function | hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6 |
Source | csr_matvec.c:256-263 |
Source loop unroll info | NA |
Source loop unroll confidence level | NA |
Unroll/vectorization loop type | NA |
Unroll factor | NA |
CQA cycles | 24.50 |
CQA cycles if no scalar integer | 18.17 |
CQA cycles if FP arith vectorized | 16.17 |
CQA cycles if fully vectorized | 7.50 |
Front-end cycles | 24.50 |
DIV/SQRT cycles | 17.63 |
P0 cycles | 17.63 |
P1 cycles | 17.33 |
P2 cycles | 17.33 |
P3 cycles | 0.50 |
P4 cycles | 17.63 |
P5 cycles | 17.50 |
P6 cycles | 0.50 |
P7 cycles | 0.50 |
P8 cycles | 0.50 |
P9 cycles | 17.60 |
P10 cycles | 17.33 |
P11 cycles | 0.00 |
Inter-iter dependencies cycles | NA |
FE+BE cycles (UFS) | 31.93 - 240.32 |
Stall cycles (UFS) | 6.75 - 215.14 |
Nb insns | 107.00 |
Nb uops | 139.00 |
Nb loads | 30.00 |
Nb stores | 1.00 |
Nb stack references | 1.00 |
FLOP/cycle | 2.86 |
Nb FLOP add-sub | 8.00 |
Nb FLOP mul | 0.00 |
Nb FLOP fma | 31.00 |
Nb FLOP div | 0.00 |
Nb FLOP rcp | 0.00 |
Nb FLOP sqrt | 0.00 |
Nb FLOP rsqrt | 0.00 |
Bytes/cycle | 31.67 |
Bytes prefetched | 0.00 |
Bytes loaded | 768.00 |
Bytes stored | 8.00 |
Stride 0 | NA |
Stride 1 | NA |
Stride n | NA |
Stride unknown | NA |
Stride indirect | NA |
Vectorization ratio all | 66.10 |
Vectorization ratio load | 92.31 |
Vectorization ratio store | 0.00 |
Vectorization ratio mul | NA |
Vectorization ratio add_sub | 80.00 |
Vectorization ratio fma | 88.89 |
Vectorization ratio div_sqrt | NA |
Vectorization ratio other | 54.29 |
Vector-efficiency ratio all | 32.52 |
Vector-efficiency ratio load | 44.23 |
Vector-efficiency ratio store | 12.50 |
Vector-efficiency ratio mul | NA |
Vector-efficiency ratio add_sub | 22.50 |
Vector-efficiency ratio fma | 43.06 |
Vector-efficiency ratio div_sqrt | NA |
Vector-efficiency ratio other | 29.11 |
Path / |
Function | hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6 |
Source file and lines | csr_matvec.c:256-263 |
Module | exec |
nb instructions | 107 |
nb uops | 139 |
loop length | 471 |
used x86 registers | 15 |
used mmx registers | 0 |
used xmm registers | 12 |
used ymm registers | 14 |
used zmm registers | 0 |
nb stack references | 1 |
micro-operation queue | 24.50 cycles |
front end | 24.50 cycles |
P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 17.63 | 17.63 | 17.33 | 17.33 | 0.50 | 17.63 | 17.50 | 0.50 | 0.50 | 0.50 | 17.60 | 17.33 |
cycles | 17.63 | 17.63 | 17.33 | 17.33 | 0.50 | 17.63 | 17.50 | 0.50 | 0.50 | 0.50 | 17.60 | 17.33 |
Cycles executing div or sqrt instructions | NA |
FE+BE cycles | 31.93-240.32 |
Stall cycles | 6.75-215.14 |
ROB full (events) | 8.01-222.19 |
Front-end | 24.50 |
Dispatch | 17.63 |
Overall L1 | 24.50 |
all | 36% |
load | 100% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 0% |
all | 83% |
load | 88% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 100% |
fma | 88% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 86% |
all | 66% |
load | 92% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 80% |
fma | 88% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 54% |
all | 24% |
load | 46% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 12% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 12% |
all | 37% |
load | 43% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 25% |
fma | 43% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 39% |
all | 32% |
load | 44% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 22% |
fma | 43% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 29% |
Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
SUB %RDX,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV %RDX,%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
LEA -0x1(%RCX),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
CMP $0x2,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JBE 58082c <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x167c> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RCX,%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
LEA (,%RDX,8),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VXORPD %XMM1,%XMM1,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
SHR $0x2,%R11 | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
LEA (%R14,%RSI,1),%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
ADD %R13,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
SAL $0x5,%R11 | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
LEA -0x20(%R11),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
SHR $0x5,%RDI | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
INC %RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $0x7,%EDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
JE 57f6f0 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x540> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x1,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f6ce <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x51e> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x2,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f6b4 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x504> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x3,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f69a <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x4ea> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x4,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f680 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x4d0> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x5,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f667 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x4b7> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x6,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f64d <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x49d> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
VMOVDQU (%RSI),%YMM9 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
MOV $0x20,%EAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
VGATHERQPD %YMM10,(%RBX,%YMM9,8),%YMM8 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10),%YMM8,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
VMOVDQU (%RSI,%RAX,1),%YMM14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM13,(%RBX,%YMM14,8),%YMM0 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM0,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VMOVDQU (%RSI,%RAX,1),%YMM5 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM7 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM7,(%RBX,%YMM5,8),%YMM3 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM3,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VMOVDQU (%RSI,%RAX,1),%YMM4 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM11,(%RBX,%YMM4,8),%YMM6 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM6,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VMOVDQU (%RSI,%RAX,1),%YMM9 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM10,(%RBX,%YMM9,8),%YMM8 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM8,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VMOVDQU (%RSI,%RAX,1),%YMM14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM13,(%RBX,%YMM14,8),%YMM0 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM0,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VMOVDQU (%RSI,%RAX,1),%YMM7 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM5 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM5,(%RBX,%YMM7,8),%YMM3 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM3,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
CMP %RAX,%R11 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f7d1 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x621> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
VEXTRACTF128 $0x1,%YMM1,%XMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
VADDPD %XMM1,%XMM3,%XMM5 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
VUNPCKHPD %XMM5,%XMM5,%XMM11 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 1 |
VADDPD %XMM5,%XMM11,%XMM8 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
TEST $0x3,%CL | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 2 | 0.20 |
JE 57f83f <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x68f> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RCX,%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VADDPD %XMM1,%XMM3,%XMM6 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
AND $-0x4,%R10 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
ADD %R10,%RDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
SUB %R10,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
CMP $0x1,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f82f <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x67f> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
ADD %R15,%R10 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
VMOVAPD %XMM12,%XMM4 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VMOVDQU (%R13,%R10,8),%XMM10 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VGATHERQPD %XMM4,(%RBX,%XMM10,8),%XMM9 | 5 | 1.33 | 0.83 | 0.67 | 0.67 | 0 | 0.83 | 0 | 0 | 0 | 0 | 0 | 0.67 | 0-29 | 1.25 |
VFMADD132PD (%R14,%R10,8),%XMM6,%XMM9 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
VUNPCKHPD %XMM9,%XMM9,%XMM6 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 1 |
VADDPD %XMM9,%XMM6,%XMM8 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
TEST $0x1,%CL | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 2 | 0.20 |
JE 57f83f <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x68f> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
AND $-0x2,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
ADD %RCX,%RDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV (%R13,%RDX,8),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VMOVSD (%RBX,%RCX,8),%XMM13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VFMADD231SD (%R14,%RDX,8),%XMM13,%XMM8 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
MOV 0x38(%RSP),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VMOVSD %XMM8,(%RDX,%R12,8) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
INC %R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
CMP %R12,%R9 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f578 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x3c8> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV (%R8,%R12,8),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x8(%R8,%R12,8),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
CMP %RDX,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JG 57f5c0 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x410> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
VMOVSD %XMM2,%XMM2,%XMM8 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VXORPD %XMM6,%XMM6,%XMM6 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %R10D,%R10D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP 57f7f6 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x646> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2.08 |
Function | hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6 |
Source file and lines | csr_matvec.c:256-263 |
Module | exec |
nb instructions | 107 |
nb uops | 139 |
loop length | 471 |
used x86 registers | 15 |
used mmx registers | 0 |
used xmm registers | 12 |
used ymm registers | 14 |
used zmm registers | 0 |
nb stack references | 1 |
micro-operation queue | 24.50 cycles |
front end | 24.50 cycles |
P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 17.63 | 17.63 | 17.33 | 17.33 | 0.50 | 17.63 | 17.50 | 0.50 | 0.50 | 0.50 | 17.60 | 17.33 |
cycles | 17.63 | 17.63 | 17.33 | 17.33 | 0.50 | 17.63 | 17.50 | 0.50 | 0.50 | 0.50 | 17.60 | 17.33 |
Cycles executing div or sqrt instructions | NA |
FE+BE cycles | 31.93-240.32 |
Stall cycles | 6.75-215.14 |
ROB full (events) | 8.01-222.19 |
Front-end | 24.50 |
Dispatch | 17.63 |
Overall L1 | 24.50 |
all | 36% |
load | 100% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 0% |
all | 83% |
load | 88% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 100% |
fma | 88% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 86% |
all | 66% |
load | 92% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 80% |
fma | 88% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 54% |
all | 24% |
load | 46% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 12% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 12% |
all | 37% |
load | 43% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 25% |
fma | 43% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 39% |
all | 32% |
load | 44% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 22% |
fma | 43% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 29% |
Instruction | Nb FU | P0 | P1 | P2 | P3 | P4 | P5 | P6 | P7 | P8 | P9 | P10 | P11 | Latency | Recip. throughput |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
SUB %RDX,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV %RDX,%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
LEA -0x1(%RCX),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
CMP $0x2,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JBE 58082c <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x167c> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RCX,%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
LEA (,%RDX,8),%RSI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
VXORPD %XMM1,%XMM1,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
SHR $0x2,%R11 | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
LEA (%R14,%RSI,1),%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
ADD %R13,%RSI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
SAL $0x5,%R11 | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
LEA -0x20(%R11),%RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
SHR $0x5,%RDI | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0-2 | 0.50 |
INC %RDI | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
AND $0x7,%EDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
JE 57f6f0 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x540> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x1,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f6ce <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x51e> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x2,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f6b4 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x504> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x3,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f69a <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x4ea> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x4,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f680 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x4d0> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x5,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f667 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x4b7> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
CMP $0x6,%RDI | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f64d <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x49d> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
VMOVDQU (%RSI),%YMM9 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
MOV $0x20,%EAX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
VGATHERQPD %YMM10,(%RBX,%YMM9,8),%YMM8 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10),%YMM8,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
VMOVDQU (%RSI,%RAX,1),%YMM14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM13,(%RBX,%YMM14,8),%YMM0 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM0,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VMOVDQU (%RSI,%RAX,1),%YMM5 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM7 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM7,(%RBX,%YMM5,8),%YMM3 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM3,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VMOVDQU (%RSI,%RAX,1),%YMM4 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM11,(%RBX,%YMM4,8),%YMM6 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM6,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VMOVDQU (%RSI,%RAX,1),%YMM9 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM10,(%RBX,%YMM9,8),%YMM8 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM8,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VMOVDQU (%RSI,%RAX,1),%YMM14 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM13,(%RBX,%YMM14,8),%YMM0 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM0,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VMOVDQU (%RSI,%RAX,1),%YMM7 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VMOVAPD %YMM15,%YMM5 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VGATHERQPD %YMM5,(%RBX,%YMM7,8),%YMM3 | 5 | 1.33 | 1.33 | 1.33 | 1.33 | 0 | 1.33 | 0 | 0 | 0 | 0 | 0 | 1.33 | 0-29 | 2 |
VFMADD231PD (%R10,%RAX,1),%YMM3,%YMM1 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
ADD $0x20,%RAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
CMP %RAX,%R11 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f7d1 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x621> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
VEXTRACTF128 $0x1,%YMM1,%XMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 1 |
VADDPD %XMM1,%XMM3,%XMM5 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
VUNPCKHPD %XMM5,%XMM5,%XMM11 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 1 |
VADDPD %XMM5,%XMM11,%XMM8 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
TEST $0x3,%CL | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 2 | 0.20 |
JE 57f83f <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x68f> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV %RCX,%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
VADDPD %XMM1,%XMM3,%XMM6 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
AND $-0x4,%R10 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
ADD %R10,%RDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
SUB %R10,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
CMP $0x1,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f82f <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x67f> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
ADD %R15,%R10 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
VMOVAPD %XMM12,%XMM4 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0-1 | 0.17 |
VMOVDQU (%R13,%R10,8),%XMM10 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0-1 | 0.33 |
VGATHERQPD %XMM4,(%RBX,%XMM10,8),%XMM9 | 5 | 1.33 | 0.83 | 0.67 | 0.67 | 0 | 0.83 | 0 | 0 | 0 | 0 | 0 | 0.67 | 0-29 | 1.25 |
VFMADD132PD (%R14,%R10,8),%XMM6,%XMM9 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
VUNPCKHPD %XMM9,%XMM9,%XMM6 | 1 | 0 | 0 | 0 | 0 | 0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 1 |
VADDPD %XMM9,%XMM6,%XMM8 | 1 | 0 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 |
TEST $0x1,%CL | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 2 | 0.20 |
JE 57f83f <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x68f> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
AND $-0x2,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1-2 | 0.20 |
ADD %RCX,%RDX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
MOV (%R13,%RDX,8),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VMOVSD (%RBX,%RCX,8),%XMM13 | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VFMADD231SD (%R14,%RDX,8),%XMM13,%XMM8 | 1 | 0.50 | 0.50 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 4 | 0.50 |
MOV 0x38(%RSP),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
VMOVSD %XMM8,(%RDX,%R12,8) | 1 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0.50 | 0.50 | 0.50 | 0 | 0 | 1 | 0.50 |
INC %R12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.17 |
CMP %R12,%R9 | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JE 57f578 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x3c8> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
MOV (%R8,%R12,8),%RDX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
MOV 0x8(%R8,%R12,8),%RCX | 1 | 0 | 0 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 1 | 0.33 |
CMP %RDX,%RCX | 1 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0.20 | 0 | 0 | 0 | 0.20 | 0 | 1 | 0.20 |
JG 57f5c0 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x410> | 1 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 |
VMOVSD %XMM2,%XMM2,%XMM8 | 1 | 0.33 | 0.33 | 0 | 0 | 0 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.33 |
VXORPD %XMM6,%XMM6,%XMM6 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
XOR %R10D,%R10D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 |
JMP 57f7f6 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x646> | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2.08 |