Files
OSACA/testcases/vmulsd-xmmxmmxmm.S
Jan Laukemann a1dc3b639b initial upload
2017-07-17 15:29:56 +02:00

65 lines
1.4 KiB
ArmAsm

#define INSTR vmulsd
#define NINST 24
#define N edi
#define i r8d
.intel_syntax noprefix
.globl ninst
.data
ninst:
.long NINST
.text
.globl latency
.type latency, @function
.align 32
latency:
push rbp
mov rbp, rsp
xor i, i
test N, N
jle done
# create DP 1.0
vpcmpeqw xmm0, xmm0, xmm0 # all ones
vpsllq xmm0, xmm0, 54 # logical left shift: 11111110..0 (54=64-(10-1))
vpsrlq xmm0, xmm0, 2 # logical right shift: 1 bit for sign; leading mantissa bit is zero
# copy DP 1.0
vmovaps xmm0, xmm0
vmovaps xmm1, xmm0
# Create DP 2.0
vaddpd xmm1, xmm1, xmm1
# Create DP 0.5
vdivpd xmm2, xmm0, xmm1
loop:
inc i
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
INSTR xmm0, xmm1, xmm0
INSTR xmm1, xmm0, xmm0
cmp i, N
jl loop
done:
mov rsp, rbp
pop rbp
ret
.size latency, .-latency