A
g++ produziert folgenden Code bei konstantem a:
//vom Compiler vorberechnete +-sin/cos(a)-Werte:
movsd .LC1(%rip), %xmm11
movsd .LC2(%rip), %xmm0
movsd .LC3(%rip), %xmm10
movsd .LC4(%rip), %xmm9
//xmm2=halfWidth und xmm3=halfLength
movsd 40(%rsp), %xmm2
movsd 32(%rsp), %xmm3
movapd %xmm2, %xmm4
movapd %xmm3, %xmm7
xorpd %xmm11, %xmm4
xorpd %xmm11, %xmm7
movapd %xmm3, %xmm12
movapd %xmm2, %xmm8
movapd %xmm4, %xmm6
mulsd %xmm10, %xmm12
mulsd %xmm0, %xmm6
movapd %xmm7, %xmm5
mulsd %xmm9, %xmm4
mulsd %xmm0, %xmm3
mulsd %xmm0, %xmm8
mulsd %xmm9, %xmm2
movapd %xmm6, %xmm13
addsd %xmm12, %xmm13
mulsd %xmm10, %xmm5
movapd %xmm4, %xmm15
mulsd %xmm0, %xmm7
addsd %xmm3, %xmm15
addsd %xmm8, %xmm12
addsd %xmm2, %xmm3
addsd %xmm5, %xmm8
addsd %xmm7, %xmm2
addsd %xmm6, %xmm5
addsd %xmm4, %xmm7
Wenn a nicht zur Kompilierzeit bekannt ist, kommt ein call sincos gegen Anfang hinzu.
Wie man sieht, bleibt hier von "doppelter Dereferenzierung" nicht viel übrig und es gibt auch keine Schleife mehr.
Die Moral von der Geschicht' ist (wie immer), dass man Mikrooptimierungen meist besser dem Compiler überlässt.