Using uiCA I produced a trace table for the following code.
cvtsi2ss xmm0, eax
addss xmm0, xmm0
https://uica.uops.info/tmp/780bce9e56ee4a718d5369deb1326215_trace.html
You can see that each cvtsi2ss has to wait for the previous iteration to finish because it depends on some bits (32:127) of xmm0.
However, changing cvtsi2ss to rsqrtss makes a big difference.
rsqrtss xmm0, xmm1
addss xmm0, xmm0
https://uica.uops.info/tmp/8897a7d45c8348e68279aea4d0b18e15_trace.html
Each rsqrtss executes in parallel with the previous iteration. I don't understand because rsqrtss produces an output with bits 32:127 unchanged, just like cvtsi2ss, so I think it should wait for any operation on the output register to finish, just like cvtsi2ss did.
After reading the answer, I ran a simple test, and it seems sure that uiCA has a bug.
IACA also fails to catch the output dependency.
Please correct me if the test code is wrong.
__asm__ (
R"(.section .text
.balign 16
noXor:
mov eax, 0x3f800000
movd xmm1, eax
rdtscp
shl rdx, 32
or rax, rdx
mov rdi, rax
mov ecx, 1 << 30
jmp noXor_loop
.balign 16
noXor_loop:
rsqrtss xmm0, xmm1
addss xmm0, xmm0
dec ecx
jnz noXor_loop
rdtscp
shl rdx, 32
or rax, rdx
sub rax, rdi
ret
.balign 16
yesXor:
mov eax, 0x3f800000
movd xmm1, eax
rdtscp
shl rdx, 32
or rax, rdx
mov rdi, rax
mov ecx, 1 << 30
jmp yesXor_loop
.balign 16
yesXor_loop:
xorps xmm0, xmm0
rsqrtss xmm0, xmm1
addss xmm0, xmm0
dec ecx
jnz yesXor_loop
rdtscp
shl rdx, 32
or rax, rdx
sub rax, rdi
ret)"
);
unsigned long long noXor(void);
unsigned long long yesXor(void);
#include <stdio.h>
int main() {
for (int i = 0; i < 4; ++i) {
printf("noXor: %llu yesXor: %llu\n", noXor(), yesXor());
}
return 0;
}
noXor: 4978836501 yesXor: 696810039
noXor: 4971780086 yesXor: 690780109
noXor: 4977293771 yesXor: 687404710
noXor: 5499602729 yesXor: 687954399