56 lines
1.5 KiB
ArmAsm
56 lines
1.5 KiB
ArmAsm
// Floating-point immediates on the SSE scalar paths: the constant is
|
|
// rewritten into a read from a read-only pool symbol ($f64.<hex> or
|
|
// $f32.<hex>, the IEEE-754 bits in the name), RIP-relative with the
|
|
// displacement left to the relocation. A positive zero on the moves
|
|
// collapses to XORPS dst, dst; a negative zero keeps its sign bit and
|
|
// takes the pool. The parenthesised $(-1.0) spelling is the one
|
|
// math/floor_amd64.s uses. Every result is folded back so no
|
|
// instruction is dead.
|
|
|
|
#include "textflag.h"
|
|
|
|
// func floatimm(x float64) float64
|
|
TEXT ·floatimm(SB), NOSPLIT, $0-16
|
|
MOVQ x+0(FP), AX
|
|
MOVQ AX, X0
|
|
// The floor kernel's sign fold: the parenthesised negative spelling.
|
|
MOVSD $ (-1.0), X2
|
|
ANDPD X2, X0
|
|
// Positive and fractional constants on the scalar moves.
|
|
MOVSD $1.5, X3
|
|
MOVSD $0.5, X4
|
|
MOVSS $2.5, X5
|
|
MOVSS $-0.5, X6
|
|
// A positive zero collapses to XORPS; a negative zero does not.
|
|
MOVSD $0.0, X7
|
|
MOVSS $0.0, X8
|
|
MOVSD $-0.0, X9
|
|
// The scalar arithmetic reads the pool through r/m (hypot's shape).
|
|
ADDSD $1.0, X3
|
|
SUBSD $0.5, X4
|
|
MULSD $-2.5, X4
|
|
DIVSD $2.0, X3
|
|
ADDSS $0.25, X5
|
|
// Fold everything into one double.
|
|
ADDSD X5, X3
|
|
ADDSD X6, X3
|
|
ADDSD X7, X3
|
|
ADDSD X8, X3
|
|
ADDSD X9, X3
|
|
ADDSD X4, X3
|
|
ADDSD X0, X3
|
|
MOVSD X3, ret+8(FP)
|
|
RET
|
|
|
|
// func floatimmfloat32() float32
|
|
TEXT ·floatimmfloat32(SB), NOSPLIT, $0-4
|
|
// The single-width pool constants ride the F3 prefix.
|
|
MOVSS $1.0, X0
|
|
MOVSS $-1.0, X1
|
|
MOVSS $0.0, X2
|
|
ADDSS $0.5, X0
|
|
ADDSS X1, X0
|
|
ADDSS X2, X0
|
|
MOVSS X0, ret+0(FP)
|
|
RET
|