feat(amd64): floating-point immediates through a synthesised pool
Assisted-by: GLM 5.3 Flash
This commit is contained in:
Vendored
+32
@@ -0,0 +1,32 @@
|
||||
// The runtime bookkeeping statements: FUNCDATA and PCDATA contribute no
|
||||
// text bytes on any architecture, and amd64 now matches. They sit between
|
||||
// real instructions here, with plain, static and offset symbol references
|
||||
// on the FUNCDATA lines, so the byte counts prove the zero contribution.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func bookkeep(x int64) int64
|
||||
TEXT ·bookkeep(SB), NOSPLIT, $0-16
|
||||
PCDATA $0, $-1
|
||||
MOVQ x+0(FP), AX
|
||||
PCDATA $1, $-2
|
||||
FUNCDATA $0, args_stackmap(SB)
|
||||
ADDQ $1, AX
|
||||
FUNCDATA $5, arginfo0(SB)
|
||||
PCDATA $1, $3
|
||||
MOVQ AX, ret+8(FP)
|
||||
FUNCDATA $1, externalfuncdata(SB)
|
||||
PCDATA $0, $0
|
||||
RET
|
||||
|
||||
// func bookkeepstatic() int64
|
||||
TEXT ·bookkeepstatic(SB), NOSPLIT, $0-8
|
||||
// A static symbol and a defined data symbol as the funcdata target.
|
||||
// (A symbol+offset target the toolchain itself refuses.)
|
||||
FUNCDATA $2, fdtable<>(SB)
|
||||
FUNCDATA $3, undefsym(SB)
|
||||
MOVQ $7, AX
|
||||
MOVQ AX, ret+0(FP)
|
||||
RET
|
||||
|
||||
GLOBL fdtable<>(SB), NOPTR, $16
|
||||
Vendored
+55
@@ -0,0 +1,55 @@
|
||||
// Floating-point immediates on the SSE scalar paths: the constant is
|
||||
// rewritten into a read from a read-only pool symbol ($f64.<hex> or
|
||||
// $f32.<hex>, the IEEE-754 bits in the name), RIP-relative with the
|
||||
// displacement left to the relocation. A positive zero on the moves
|
||||
// collapses to XORPS dst, dst; a negative zero keeps its sign bit and
|
||||
// takes the pool. The parenthesised $(-1.0) spelling is the one
|
||||
// math/floor_amd64.s uses. Every result is folded back so no
|
||||
// instruction is dead.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func floatimm(x float64) float64
|
||||
TEXT ·floatimm(SB), NOSPLIT, $0-16
|
||||
MOVQ x+0(FP), AX
|
||||
MOVQ AX, X0
|
||||
// The floor kernel's sign fold: the parenthesised negative spelling.
|
||||
MOVSD $ (-1.0), X2
|
||||
ANDPD X2, X0
|
||||
// Positive and fractional constants on the scalar moves.
|
||||
MOVSD $1.5, X3
|
||||
MOVSD $0.5, X4
|
||||
MOVSS $2.5, X5
|
||||
MOVSS $-0.5, X6
|
||||
// A positive zero collapses to XORPS; a negative zero does not.
|
||||
MOVSD $0.0, X7
|
||||
MOVSS $0.0, X8
|
||||
MOVSD $-0.0, X9
|
||||
// The scalar arithmetic reads the pool through r/m (hypot's shape).
|
||||
ADDSD $1.0, X3
|
||||
SUBSD $0.5, X4
|
||||
MULSD $-2.5, X4
|
||||
DIVSD $2.0, X3
|
||||
ADDSS $0.25, X5
|
||||
// Fold everything into one double.
|
||||
ADDSD X5, X3
|
||||
ADDSD X6, X3
|
||||
ADDSD X7, X3
|
||||
ADDSD X8, X3
|
||||
ADDSD X9, X3
|
||||
ADDSD X4, X3
|
||||
ADDSD X0, X3
|
||||
MOVSD X3, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func floatimmfloat32() float32
|
||||
TEXT ·floatimmfloat32(SB), NOSPLIT, $0-4
|
||||
// The single-width pool constants ride the F3 prefix.
|
||||
MOVSS $1.0, X0
|
||||
MOVSS $-1.0, X1
|
||||
MOVSS $0.0, X2
|
||||
ADDSS $0.5, X0
|
||||
ADDSS X1, X0
|
||||
ADDSS X2, X0
|
||||
MOVSS X0, ret+0(FP)
|
||||
RET
|
||||
Reference in New Issue
Block a user