feat(amd64): floating-point immediates through a synthesised pool

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-21 02:04:44 +02:00
parent bfb7701db1
commit 29ac03468e
10 changed files with 761 additions and 41 deletions
+55
View File
@@ -0,0 +1,55 @@
// Floating-point immediates on the SSE scalar paths: the constant is
// rewritten into a read from a read-only pool symbol ($f64.<hex> or
// $f32.<hex>, the IEEE-754 bits in the name), RIP-relative with the
// displacement left to the relocation. A positive zero on the moves
// collapses to XORPS dst, dst; a negative zero keeps its sign bit and
// takes the pool. The parenthesised $(-1.0) spelling is the one
// math/floor_amd64.s uses. Every result is folded back so no
// instruction is dead.
#include "textflag.h"
// func floatimm(x float64) float64
TEXT ·floatimm(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
MOVQ AX, X0
// The floor kernel's sign fold: the parenthesised negative spelling.
MOVSD $ (-1.0), X2
ANDPD X2, X0
// Positive and fractional constants on the scalar moves.
MOVSD $1.5, X3
MOVSD $0.5, X4
MOVSS $2.5, X5
MOVSS $-0.5, X6
// A positive zero collapses to XORPS; a negative zero does not.
MOVSD $0.0, X7
MOVSS $0.0, X8
MOVSD $-0.0, X9
// The scalar arithmetic reads the pool through r/m (hypot's shape).
ADDSD $1.0, X3
SUBSD $0.5, X4
MULSD $-2.5, X4
DIVSD $2.0, X3
ADDSS $0.25, X5
// Fold everything into one double.
ADDSD X5, X3
ADDSD X6, X3
ADDSD X7, X3
ADDSD X8, X3
ADDSD X9, X3
ADDSD X4, X3
ADDSD X0, X3
MOVSD X3, ret+8(FP)
RET
// func floatimmfloat32() float32
TEXT ·floatimmfloat32(SB), NOSPLIT, $0-4
// The single-width pool constants ride the F3 prefix.
MOVSS $1.0, X0
MOVSS $-1.0, X1
MOVSS $0.0, X2
ADDSS $0.5, X0
ADDSS X1, X0
ADDSS X2, X0
MOVSS X0, ret+0(FP)
RET