Loop Id: 139 | Module: libseq_mv.so | Source: csr_matvec.c:334-341 | Coverage: 1.81% |
---|
Loop Id: 139 | Module: libseq_mv.so | Source: csr_matvec.c:334-341 | Coverage: 1.81% |
---|
0xe4a0 MOV 0x30(%RSP),%R15 |
0xe4a5 MOV (%R8,%R12,8),%RDX |
0xe4a9 MOV 0x8(%R8,%R12,8),%RCX |
0xe4ae VMOVSD (%R15,%R12,8),%XMM3 |
0xe4b4 CMP %RCX,%RDX |
0xe4b7 JGE e758 |
0xe4bd SUB %RDX,%RCX |
0xe4c0 MOV %RDX,%R15 |
0xe4c3 LEA -0x1(%RCX),%RSI |
0xe4c7 CMP $0x2,%RSI |
0xe4cb JBE ed1f |
0xe4d1 MOV %RCX,%R11 |
0xe4d4 LEA (,%RDX,8),%RSI |
0xe4dc VXORPD %XMM1,%XMM1,%XMM1 |
0xe4e0 XOR %EAX,%EAX |
0xe4e2 SHR $0x2,%R11 |
0xe4e6 LEA (%R14,%RSI,1),%R10 |
0xe4ea ADD %R13,%RSI |
0xe4ed SAL $0x5,%R11 |
0xe4f1 LEA -0x20(%R11),%RDI |
0xe4f5 SHR $0x5,%RDI |
0xe4f9 INC %RDI |
0xe4fc AND $0x7,%EDI |
0xe4ff JE e5e8 |
0xe505 CMP $0x1,%RDI |
0xe509 JE e5c6 |
0xe50f CMP $0x2,%RDI |
0xe513 JE e5ad |
0xe519 CMP $0x3,%RDI |
0xe51d JE e594 |
0xe51f CMP $0x4,%RDI |
0xe523 JE e57b |
0xe525 CMP $0x5,%RDI |
0xe529 JE e562 |
0xe52b CMP $0x6,%RDI |
0xe52f JE e549 |
0xe531 VMOVDQU (%RSI),%YMM2 |
0xe535 VMOVAPD %YMM5,%YMM9 |
0xe539 MOV $0x20,%EAX |
0xe53e VGATHERQPD %YMM9,(%RBX,%YMM2,8),%YMM0 |
0xe544 VFMADD231PD (%R10),%YMM0,%YMM1 |
0xe549 VMOVDQU (%RSI,%RAX,1),%YMM7 |
0xe54e VMOVAPD %YMM5,%YMM8 |
0xe552 VGATHERQPD %YMM8,(%RBX,%YMM7,8),%YMM11 |
0xe558 VFMADD231PD (%R10,%RAX,1),%YMM11,%YMM1 |
0xe55e ADD $0x20,%RAX |
0xe562 VMOVDQU (%RSI,%RAX,1),%YMM10 |
0xe567 VMOVAPD %YMM5,%YMM13 |
0xe56b VGATHERQPD %YMM13,(%RBX,%YMM10,8),%YMM6 |
0xe571 VFMADD231PD (%R10,%RAX,1),%YMM6,%YMM1 |
0xe577 ADD $0x20,%RAX |
0xe57b VMOVDQU (%RSI,%RAX,1),%YMM4 |
0xe580 VMOVAPD %YMM5,%YMM12 |
0xe584 VGATHERQPD %YMM12,(%RBX,%YMM4,8),%YMM14 |
0xe58a VFMADD231PD (%R10,%RAX,1),%YMM14,%YMM1 |
0xe590 ADD $0x20,%RAX |
0xe594 VMOVDQU (%RSI,%RAX,1),%YMM2 |
0xe599 VMOVAPD %YMM5,%YMM9 |
0xe59d VGATHERQPD %YMM9,(%RBX,%YMM2,8),%YMM0 |
0xe5a3 VFMADD231PD (%R10,%RAX,1),%YMM0,%YMM1 |
0xe5a9 ADD $0x20,%RAX |
0xe5ad VMOVDQU (%RSI,%RAX,1),%YMM7 |
0xe5b2 VMOVAPD %YMM5,%YMM8 |
0xe5b6 VGATHERQPD %YMM8,(%RBX,%YMM7,8),%YMM11 |
0xe5bc VFMADD231PD (%R10,%RAX,1),%YMM11,%YMM1 |
0xe5c2 ADD $0x20,%RAX |
0xe5c6 VMOVDQU (%RSI,%RAX,1),%YMM10 |
0xe5cb VMOVAPD %YMM5,%YMM13 |
0xe5cf VGATHERQPD %YMM13,(%RBX,%YMM10,8),%YMM6 |
0xe5d5 VFMADD231PD (%R10,%RAX,1),%YMM6,%YMM1 |
0xe5db ADD $0x20,%RAX |
0xe5df CMP %R11,%RAX |
0xe5e2 JE e6c5 |
(140) 0xe5e8 VMOVDQU (%RSI,%RAX,1),%YMM4 |
(140) 0xe5ed VMOVDQU 0x20(%RSI,%RAX,1),%YMM2 |
(140) 0xe5f3 VMOVAPD %YMM5,%YMM12 |
(140) 0xe5f7 VMOVAPD %YMM5,%YMM9 |
(140) 0xe5fb VMOVDQU 0x40(%RSI,%RAX,1),%YMM7 |
(140) 0xe601 VMOVAPD %YMM5,%YMM8 |
(140) 0xe605 VMOVAPD %YMM5,%YMM13 |
(140) 0xe609 VMOVAPD %YMM5,%YMM6 |
(140) 0xe60d VGATHERQPD %YMM12,(%RBX,%YMM4,8),%YMM14 |
(140) 0xe613 VFMADD231PD (%R10,%RAX,1),%YMM14,%YMM1 |
(140) 0xe619 VMOVAPD %YMM5,%YMM14 |
(140) 0xe61d VMOVDQU 0x60(%RSI,%RAX,1),%YMM10 |
(140) 0xe623 VGATHERQPD %YMM9,(%RBX,%YMM2,8),%YMM0 |
(140) 0xe629 VFMADD231PD 0x20(%R10,%RAX,1),%YMM0,%YMM1 |
(140) 0xe630 VGATHERQPD %YMM8,(%RBX,%YMM7,8),%YMM11 |
(140) 0xe636 VMOVDQU 0x80(%RSI,%RAX,1),%YMM12 |
(140) 0xe63f VFMADD132PD 0x40(%R10,%RAX,1),%YMM1,%YMM11 |
(140) 0xe646 VMOVAPD %YMM5,%YMM7 |
(140) 0xe64a VMOVDQU 0xa0(%RSI,%RAX,1),%YMM9 |
(140) 0xe653 VGATHERQPD %YMM13,(%RBX,%YMM10,8),%YMM1 |
(140) 0xe659 VMOVDQU 0xc0(%RSI,%RAX,1),%YMM2 |
(140) 0xe662 VFMADD132PD 0x60(%R10,%RAX,1),%YMM11,%YMM1 |
(140) 0xe669 VGATHERQPD %YMM6,(%RBX,%YMM12,8),%YMM4 |
(140) 0xe66f VMOVAPD %YMM5,%YMM11 |
(140) 0xe673 VFMADD132PD 0x80(%R10,%RAX,1),%YMM1,%YMM4 |
(140) 0xe67d VMOVDQU 0xe0(%RSI,%RAX,1),%YMM13 |
(140) 0xe686 VGATHERQPD %YMM14,(%RBX,%YMM9,8),%YMM8 |
(140) 0xe68c VFMADD132PD 0xa0(%R10,%RAX,1),%YMM4,%YMM8 |
(140) 0xe696 VGATHERQPD %YMM7,(%RBX,%YMM2,8),%YMM0 |
(140) 0xe69c VFMADD132PD 0xc0(%R10,%RAX,1),%YMM8,%YMM0 |
(140) 0xe6a6 VGATHERQPD %YMM11,(%RBX,%YMM13,8),%YMM1 |
(140) 0xe6ac VFMADD132PD 0xe0(%R10,%RAX,1),%YMM0,%YMM1 |
(140) 0xe6b6 ADD $0x100,%RAX |
(140) 0xe6bc CMP %R11,%RAX |
(140) 0xe6bf JNE e5e8 |
0xe6c5 VEXTRACTF128 $0x1,%YMM1,%XMM10 |
0xe6cb MOV %RCX,%R10 |
0xe6ce VADDPD %XMM1,%XMM10,%XMM6 |
0xe6d2 AND $-0x4,%R10 |
0xe6d6 VADDPD %XMM10,%XMM1,%XMM14 |
0xe6db ADD %R10,%RDX |
0xe6de VUNPCKHPD %XMM6,%XMM6,%XMM12 |
0xe6e2 VADDPD %XMM6,%XMM12,%XMM4 |
0xe6e6 VADDSD %XMM4,%XMM3,%XMM0 |
0xe6ea TEST $0x3,%CL |
0xe6ed JE e73c |
0xe6ef SUB %R10,%RCX |
0xe6f2 CMP $0x1,%RCX |
0xe6f6 JE e72b |
0xe6f8 ADD %R15,%R10 |
0xe6fb VMOVAPD %XMM15,%XMM9 |
0xe700 VMOVDQU (%R13,%R10,8),%XMM8 |
0xe707 VGATHERQPD %XMM9,(%RBX,%XMM8,8),%XMM7 |
0xe70d VFMADD132PD (%R14,%R10,8),%XMM14,%XMM7 |
0xe713 VUNPCKHPD %XMM7,%XMM7,%XMM14 |
0xe717 VADDPD %XMM7,%XMM14,%XMM2 |
0xe71b VADDSD %XMM2,%XMM3,%XMM0 |
0xe71f TEST $0x1,%CL |
0xe722 JE e73c |
0xe724 AND $-0x2,%RCX |
0xe728 ADD %RCX,%RDX |
0xe72b MOV (%R13,%RDX,8),%RCX |
0xe730 VMOVSD (%R14,%RDX,8),%XMM3 |
0xe736 VFMADD231SD (%RBX,%RCX,8),%XMM3,%XMM0 |
0xe73c MOV 0x38(%RSP),%RDX |
0xe741 VMOVSD %XMM0,(%RDX,%R12,8) |
0xe747 INC %R12 |
0xe74a CMP %R12,%R9 |
0xe74d JNE e4a0 |
0xe758 VMOVSD %XMM3,%XMM3,%XMM0 |
0xe75c JMP e73c |
0xed1f VMOVSD %XMM3,%XMM3,%XMM0 |
0xed23 VXORPD %XMM14,%XMM14,%XMM14 |
0xed28 XOR %R10D,%R10D |
0xed2b JMP e6ef |
/home/eoseret/qaas_runs_CPU_9468/172-019-1763/intel/AMG/build/AMG/AMG/seq_mv/csr_matvec.c: 334 - 341 |
-------------------------------------------------------------------------------- |
334: for (i = iBegin; i < iEnd; i++) |
335: { |
336: tempx = b_data[i]; |
337: for (jj = A_i[i]; jj < A_i[i+1]; jj++) |
338: { |
339: tempx += A_data[jj] * x_data[A_j[jj]]; |
340: } |
341: y_data[i] = tempx; |
Coverage (%) | Name | Source Location | Module |
---|---|---|---|
○95.76 | gomp_thread_start | team.c:130 | libgomp.so.1.0.0 |
○4.24 | GOMP_parallel | libgomp.h:985 | libgomp.so.1.0.0 |
Path / |
Metric | Value |
---|---|
CQA speedup if no scalar integer | 1.20 |
CQA speedup if FP arith vectorized | 2.23 |
CQA speedup if fully vectorized | 4.36 |
CQA speedup if no inter-iteration dependency | NA |
CQA speedup if next bottleneck killed | 2.42 |
Bottlenecks | micro-operation queue, |
Function | hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6 |
Source | csr_matvec.c:334-341 |
Source loop unroll info | NA |
Source loop unroll confidence level | NA |
Unroll/vectorization loop type | NA |
Unroll factor | NA |
CQA cycles | 48.50 |
CQA cycles if no scalar integer | 40.50 |
CQA cycles if FP arith vectorized | 21.70 |
CQA cycles if fully vectorized | 11.13 |
Front-end cycles | 48.50 |
DIV/SQRT cycles | 11.50 |
P0 cycles | 11.50 |
P1 cycles | 11.50 |
P2 cycles | 11.50 |
P3 cycles | 8.00 |
P4 cycles | 8.33 |
P5 cycles | 8.33 |
P6 cycles | 8.33 |
P7 cycles | 20.00 |
P8 cycles | 20.00 |
P9 cycles | 20.00 |
P10 cycles | 20.00 |
P11 cycles | 19.50 |
P12 cycles | 19.50 |
P13 cycles | 0.00 |
Inter-iter dependencies cycles | NA |
FE+BE cycles (UFS) | NA |
Stall cycles (UFS) | NA |
Nb insns | 113.00 |
Nb uops | 291.00 |
Nb loads | 32.00 |
Nb stores | 1.00 |
Nb stack references | 2.00 |
FLOP/cycle | 1.48 |
Nb FLOP add-sub | 10.00 |
Nb FLOP mul | 0.00 |
Nb FLOP fma | 31.00 |
Nb FLOP div | 0.00 |
Nb FLOP rcp | 0.00 |
Nb FLOP sqrt | 0.00 |
Nb FLOP rsqrt | 0.00 |
Bytes/cycle | 16.33 |
Bytes prefetched | 0.00 |
Bytes loaded | 784.00 |
Bytes stored | 8.00 |
Stride 0 | NA |
Stride 1 | NA |
Stride n | NA |
Stride unknown | NA |
Stride indirect | NA |
Vectorization ratio all | 62.90 |
Vectorization ratio load | 88.89 |
Vectorization ratio store | 0.00 |
Vectorization ratio mul | NA |
Vectorization ratio add_sub | 57.14 |
Vectorization ratio fma | 88.89 |
Vectorization ratio div_sqrt | NA |
Vectorization ratio other | 54.29 |
Vector-efficiency ratio all | 31.55 |
Vector-efficiency ratio load | 43.06 |
Vector-efficiency ratio store | 12.50 |
Vector-efficiency ratio mul | NA |
Vector-efficiency ratio add_sub | 19.64 |
Vector-efficiency ratio fma | 43.06 |
Vector-efficiency ratio div_sqrt | NA |
Vector-efficiency ratio other | 29.11 |
Metric | Value |
---|---|
CQA speedup if no scalar integer | 1.20 |
CQA speedup if FP arith vectorized | 2.23 |
CQA speedup if fully vectorized | 4.36 |
CQA speedup if no inter-iteration dependency | NA |
CQA speedup if next bottleneck killed | 2.42 |
Bottlenecks | micro-operation queue, |
Function | hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6 |
Source | csr_matvec.c:334-341 |
Source loop unroll info | NA |
Source loop unroll confidence level | NA |
Unroll/vectorization loop type | NA |
Unroll factor | NA |
CQA cycles | 48.50 |
CQA cycles if no scalar integer | 40.50 |
CQA cycles if FP arith vectorized | 21.70 |
CQA cycles if fully vectorized | 11.13 |
Front-end cycles | 48.50 |
DIV/SQRT cycles | 11.50 |
P0 cycles | 11.50 |
P1 cycles | 11.50 |
P2 cycles | 11.50 |
P3 cycles | 8.00 |
P4 cycles | 8.33 |
P5 cycles | 8.33 |
P6 cycles | 8.33 |
P7 cycles | 20.00 |
P8 cycles | 20.00 |
P9 cycles | 20.00 |
P10 cycles | 20.00 |
P11 cycles | 19.50 |
P12 cycles | 19.50 |
P13 cycles | 0.00 |
Inter-iter dependencies cycles | NA |
FE+BE cycles (UFS) | NA |
Stall cycles (UFS) | NA |
Nb insns | 113.00 |
Nb uops | 291.00 |
Nb loads | 32.00 |
Nb stores | 1.00 |
Nb stack references | 2.00 |
FLOP/cycle | 1.48 |
Nb FLOP add-sub | 10.00 |
Nb FLOP mul | 0.00 |
Nb FLOP fma | 31.00 |
Nb FLOP div | 0.00 |
Nb FLOP rcp | 0.00 |
Nb FLOP sqrt | 0.00 |
Nb FLOP rsqrt | 0.00 |
Bytes/cycle | 16.33 |
Bytes prefetched | 0.00 |
Bytes loaded | 784.00 |
Bytes stored | 8.00 |
Stride 0 | NA |
Stride 1 | NA |
Stride n | NA |
Stride unknown | NA |
Stride indirect | NA |
Vectorization ratio all | 62.90 |
Vectorization ratio load | 88.89 |
Vectorization ratio store | 0.00 |
Vectorization ratio mul | NA |
Vectorization ratio add_sub | 57.14 |
Vectorization ratio fma | 88.89 |
Vectorization ratio div_sqrt | NA |
Vectorization ratio other | 54.29 |
Vector-efficiency ratio all | 31.55 |
Vector-efficiency ratio load | 43.06 |
Vector-efficiency ratio store | 12.50 |
Vector-efficiency ratio mul | NA |
Vector-efficiency ratio add_sub | 19.64 |
Vector-efficiency ratio fma | 43.06 |
Vector-efficiency ratio div_sqrt | NA |
Vector-efficiency ratio other | 29.11 |
Path / |
Function | hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6 |
Source file and lines | csr_matvec.c:334-341 |
Module | libseq_mv.so |
nb instructions | 113 |
nb uops | 291 |
loop length | 493 |
used x86 registers | 15 |
used mmx registers | 0 |
used xmm registers | 13 |
used ymm registers | 14 |
used zmm registers | 0 |
nb stack references | 2 |
micro-operation queue | 48.50 cycles |
front end | 48.50 cycles |
ALU0/BRU0 | ALU1 | ALU2 | ALU3 | BRU1 | AGU0 | AGU1 | AGU2 | FP0 | FP1 | FP2 | FP3 | FP4 | FP5 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 11.50 | 11.50 | 11.50 | 11.50 | 8.00 | 8.33 | 8.33 | 8.33 | 20.00 | 20.00 | 20.00 | 20.00 | 19.50 | 19.50 |
cycles | 11.50 | 11.50 | 11.50 | 11.50 | 8.00 | 8.33 | 8.33 | 8.33 | 20.00 | 20.00 | 20.00 | 20.00 | 19.50 | 19.50 |
Cycles executing div or sqrt instructions | NA |
Front-end | 48.50 |
Dispatch | 20.00 |
Overall L1 | 48.50 |
all | 38% |
load | 100% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 0% |
all | 75% |
load | 84% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 66% |
fma | 88% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 82% |
all | 62% |
load | 88% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 57% |
fma | 88% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 54% |
all | 25% |
load | 46% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 12% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 11% |
all | 34% |
load | 41% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 20% |
fma | 43% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 38% |
all | 31% |
load | 43% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 19% |
fma | 43% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 29% |
Instruction | Nb FU | ALU0/BRU0 | ALU1 | ALU2 | ALU3 | BRU1 | AGU0 | AGU1 | AGU2 | FP0 | FP1 | FP2 | FP3 | FP4 | FP5 | Latency | Recip. throughput | Vectorization |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
MOV 0x30(%RSP),%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
MOV (%R8,%R12,8),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
MOV 0x8(%R8,%R12,8),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
VMOVSD (%R15,%R12,8),%XMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (12.5%) |
CMP %RCX,%RDX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JGE e758 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1f58> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
SUB %RDX,%RCX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
MOV %RDX,%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
LEA -0x1(%RCX),%RSI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
CMP $0x2,%RSI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JBE ed1f <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x251f> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
MOV %RCX,%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
LEA (,%RDX,8),%RSI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VXORPD %XMM1,%XMM1,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
SHR $0x2,%R11 | 1 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
LEA (%R14,%RSI,1),%R10 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
ADD %R13,%RSI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
SAL $0x5,%R11 | 1 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
LEA -0x20(%R11),%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
SHR $0x5,%RDI | 1 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (12.5%) |
INC %RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
AND $0x7,%EDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (6.3%) |
JE e5e8 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1de8> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x1,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e5c6 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1dc6> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x2,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e5ad <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1dad> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x3,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e594 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1d94> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x4,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e57b <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1d7b> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x5,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e562 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1d62> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x6,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e549 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1d49> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
VMOVDQU (%RSI),%YMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
MOV $0x20,%EAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VGATHERQPD %YMM9,(%RBX,%YMM2,8),%YMM0 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10),%YMM0,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
VMOVDQU (%RSI,%RAX,1),%YMM7 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM8,(%RBX,%YMM7,8),%YMM11 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM11,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVDQU (%RSI,%RAX,1),%YMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM13,(%RBX,%YMM10,8),%YMM6 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM6,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVDQU (%RSI,%RAX,1),%YMM4 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM12,(%RBX,%YMM4,8),%YMM14 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM14,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVDQU (%RSI,%RAX,1),%YMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM9,(%RBX,%YMM2,8),%YMM0 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM0,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVDQU (%RSI,%RAX,1),%YMM7 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM8,(%RBX,%YMM7,8),%YMM11 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM11,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVDQU (%RSI,%RAX,1),%YMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM13,(%RBX,%YMM10,8),%YMM6 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM6,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
CMP %R11,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e6c5 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1ec5> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
VEXTRACTF128 $0x1,%YMM1,%XMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
MOV %RCX,%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
VADDPD %XMM1,%XMM10,%XMM6 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
AND $-0x4,%R10 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VADDPD %XMM10,%XMM1,%XMM14 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
ADD %R10,%RDX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VUNPCKHPD %XMM6,%XMM6,%XMM12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 1 | 0.33 | scal (12.5%) |
VADDPD %XMM6,%XMM12,%XMM4 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
VADDSD %XMM4,%XMM3,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | scal (12.5%) |
TEST $0x3,%CL | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
JE e73c <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1f3c> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
SUB %R10,%RCX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
CMP $0x1,%RCX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e72b <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1f2b> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
ADD %R15,%R10 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVAPD %XMM15,%XMM9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
VMOVDQU (%R13,%R10,8),%XMM8 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
VGATHERQPD %XMM9,(%RBX,%XMM8,8),%XMM7 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1.25 | 1.58 | 0.58 | 0.58 | 1.50 | 1.50 | 0-16 | 3 | vect (25.0%) |
VFMADD132PD (%R14,%R10,8),%XMM14,%XMM7 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (25.0%) |
VUNPCKHPD %XMM7,%XMM7,%XMM14 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 1 | 0.33 | scal (12.5%) |
VADDPD %XMM7,%XMM14,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
VADDSD %XMM2,%XMM3,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | scal (12.5%) |
TEST $0x1,%CL | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
JE e73c <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1f3c> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
AND $-0x2,%RCX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
ADD %RCX,%RDX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
MOV (%R13,%RDX,8),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
VMOVSD (%R14,%RDX,8),%XMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (12.5%) |
VFMADD231SD (%RBX,%RCX,8),%XMM3,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
MOV 0x38(%RSP),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
VMOVSD %XMM0,(%RDX,%R12,8) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 1 | 1 | scal (12.5%) |
INC %R12 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
CMP %R12,%R9 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
JNE e4a0 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1ca0> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
VMOVSD %XMM3,%XMM3,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JMP e73c <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1f3c> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | N/A |
VMOVSD %XMM3,%XMM3,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
VXORPD %XMM14,%XMM14,%XMM14 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
XOR %R10D,%R10D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
JMP e6ef <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1eef> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | N/A |
Function | hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6 |
Source file and lines | csr_matvec.c:334-341 |
Module | libseq_mv.so |
nb instructions | 113 |
nb uops | 291 |
loop length | 493 |
used x86 registers | 15 |
used mmx registers | 0 |
used xmm registers | 13 |
used ymm registers | 14 |
used zmm registers | 0 |
nb stack references | 2 |
micro-operation queue | 48.50 cycles |
front end | 48.50 cycles |
ALU0/BRU0 | ALU1 | ALU2 | ALU3 | BRU1 | AGU0 | AGU1 | AGU2 | FP0 | FP1 | FP2 | FP3 | FP4 | FP5 | |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
uops | 11.50 | 11.50 | 11.50 | 11.50 | 8.00 | 8.33 | 8.33 | 8.33 | 20.00 | 20.00 | 20.00 | 20.00 | 19.50 | 19.50 |
cycles | 11.50 | 11.50 | 11.50 | 11.50 | 8.00 | 8.33 | 8.33 | 8.33 | 20.00 | 20.00 | 20.00 | 20.00 | 19.50 | 19.50 |
Cycles executing div or sqrt instructions | NA |
Front-end | 48.50 |
Dispatch | 20.00 |
Overall L1 | 48.50 |
all | 38% |
load | 100% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 0% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 0% |
all | 75% |
load | 84% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 66% |
fma | 88% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 82% |
all | 62% |
load | 88% |
store | 0% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 57% |
fma | 88% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 54% |
all | 25% |
load | 46% |
store | NA (no store vectorizable/vectorized instructions) |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 12% |
fma | NA (no fma vectorizable/vectorized instructions) |
other | 11% |
all | 34% |
load | 41% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 20% |
fma | 43% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 38% |
all | 31% |
load | 43% |
store | 12% |
mul | NA (no mul vectorizable/vectorized instructions) |
add-sub | 19% |
fma | 43% |
div/sqrt | NA (no div/sqrt vectorizable/vectorized instructions) |
other | 29% |
Instruction | Nb FU | ALU0/BRU0 | ALU1 | ALU2 | ALU3 | BRU1 | AGU0 | AGU1 | AGU2 | FP0 | FP1 | FP2 | FP3 | FP4 | FP5 | Latency | Recip. throughput | Vectorization |
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
MOV 0x30(%RSP),%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
MOV (%R8,%R12,8),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
MOV 0x8(%R8,%R12,8),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
VMOVSD (%R15,%R12,8),%XMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (12.5%) |
CMP %RCX,%RDX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JGE e758 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1f58> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
SUB %RDX,%RCX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
MOV %RDX,%R15 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
LEA -0x1(%RCX),%RSI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
CMP $0x2,%RSI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JBE ed1f <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x251f> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
MOV %RCX,%R11 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
LEA (,%RDX,8),%RSI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VXORPD %XMM1,%XMM1,%XMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
XOR %EAX,%EAX | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
SHR $0x2,%R11 | 1 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
LEA (%R14,%RSI,1),%R10 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
ADD %R13,%RSI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
SAL $0x5,%R11 | 1 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | N/A |
LEA -0x20(%R11),%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
SHR $0x5,%RDI | 1 | 0 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (12.5%) |
INC %RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
AND $0x7,%EDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (6.3%) |
JE e5e8 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1de8> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x1,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e5c6 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1dc6> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x2,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e5ad <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1dad> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x3,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e594 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1d94> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x4,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e57b <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1d7b> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x5,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e562 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1d62> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
CMP $0x6,%RDI | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e549 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1d49> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
VMOVDQU (%RSI),%YMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
MOV $0x20,%EAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VGATHERQPD %YMM9,(%RBX,%YMM2,8),%YMM0 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10),%YMM0,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
VMOVDQU (%RSI,%RAX,1),%YMM7 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM8,(%RBX,%YMM7,8),%YMM11 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM11,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVDQU (%RSI,%RAX,1),%YMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM13,(%RBX,%YMM10,8),%YMM6 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM6,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVDQU (%RSI,%RAX,1),%YMM4 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM12,(%RBX,%YMM4,8),%YMM14 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM14,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVDQU (%RSI,%RAX,1),%YMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM9,(%RBX,%YMM2,8),%YMM0 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM0,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVDQU (%RSI,%RAX,1),%YMM7 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM8 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM8,(%RBX,%YMM7,8),%YMM11 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM11,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVDQU (%RSI,%RAX,1),%YMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (50.0%) |
VMOVAPD %YMM5,%YMM13 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (50.0%) |
VGATHERQPD %YMM13,(%RBX,%YMM10,8),%YMM6 | 24 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 2 | 3 | 1.50 | 1.50 | 2.50 | 2.50 | 0-16 | 4 | vect (50.0%) |
VFMADD231PD (%R10,%RAX,1),%YMM6,%YMM1 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (50.0%) |
ADD $0x20,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
CMP %R11,%RAX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e6c5 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1ec5> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
VEXTRACTF128 $0x1,%YMM1,%XMM10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | vect (25.0%) |
MOV %RCX,%R10 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | N/A |
VADDPD %XMM1,%XMM10,%XMM6 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
AND $-0x4,%R10 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VADDPD %XMM10,%XMM1,%XMM14 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
ADD %R10,%RDX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VUNPCKHPD %XMM6,%XMM6,%XMM12 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 1 | 0.33 | scal (12.5%) |
VADDPD %XMM6,%XMM12,%XMM4 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
VADDSD %XMM4,%XMM3,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | scal (12.5%) |
TEST $0x3,%CL | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
JE e73c <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1f3c> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
SUB %R10,%RCX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
CMP $0x1,%RCX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JE e72b <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1f2b> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
ADD %R15,%R10 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
VMOVAPD %XMM15,%XMM9 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
VMOVDQU (%R13,%R10,8),%XMM8 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
VGATHERQPD %XMM9,(%RBX,%XMM8,8),%XMM7 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1.25 | 1.58 | 0.58 | 0.58 | 1.50 | 1.50 | 0-16 | 3 | vect (25.0%) |
VFMADD132PD (%R14,%R10,8),%XMM14,%XMM7 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | vect (25.0%) |
VUNPCKHPD %XMM7,%XMM7,%XMM14 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 1 | 0.33 | scal (12.5%) |
VADDPD %XMM7,%XMM14,%XMM2 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | vect (25.0%) |
VADDSD %XMM2,%XMM3,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 0 | 0 | 3 | 0.50 | scal (12.5%) |
TEST $0x1,%CL | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
JE e73c <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1f3c> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
AND $-0x2,%RCX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
ADD %RCX,%RDX | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
MOV (%R13,%RDX,8),%RCX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
VMOVSD (%R14,%RDX,8),%XMM3 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50 | scal (12.5%) |
VFMADD231SD (%RBX,%RCX,8),%XMM3,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0.50 | 0.50 | 0 | 0 | 0 | 0 | 4 | 0.50 | scal (12.5%) |
MOV 0x38(%RSP),%RDX | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0 | 0 | 3 | 0.33 | N/A |
VMOVSD %XMM0,(%RDX,%R12,8) | 1 | 0 | 0 | 0 | 0 | 0 | 0.33 | 0.33 | 0.33 | 0 | 0 | 0 | 0 | 0.50 | 0.50 | 1 | 1 | scal (12.5%) |
INC %R12 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
CMP %R12,%R9 | 1 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.25 | N/A |
JNE e4a0 <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1ca0> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | 0.50-1 | N/A |
VMOVSD %XMM3,%XMM3,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
JMP e73c <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1f3c> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | N/A |
VMOVSD %XMM3,%XMM3,%XMM0 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | 0.25 | 0.25 | 0.25 | 0 | 0 | 1 | 0.25 | scal (12.5%) |
VXORPD %XMM14,%XMM14,%XMM14 | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.17 | vect (25.0%) |
XOR %R10D,%R10D | 1 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0.25 | N/A |
JMP e6ef <hypre_CSRMatrixMatvecOutOfPlace._omp_fn.6+0x1eef> | 1 | 0.50 | 0 | 0 | 0 | 0.50 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 1 | N/A |