test(verify): prove the amd64 sse families through the link-parity kernel
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
b22abf512f
commit
8e3b7f1caa
2 files changed
+121
No files matched your search
@@ -0,0 +1,63 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package verify
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// goobjFoldMainSrc is the Go side of the SSE regression. The kernel
|
||||||
|
// XOR-folds 16-byte blocks and sums the tail bytes through the amd64 SSE
|
||||||
|
// families; main derives the same value in Go and panics on any divergence
|
||||||
|
// before printing the deterministic line the baseline and the gasm-linked
|
||||||
|
// binaries must agree on.
|
||||||
|
const goobjFoldMainSrc = `package main
|
||||||
|
|
||||||
|
import "encoding/binary"
|
||||||
|
|
||||||
|
func fold(buf []byte, seed uint64) uint64
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
buf := []byte("gasm link parity vector kernel! 0123456789abcdef tail")
|
||||||
|
low, high := uint64(0), uint64(0)
|
||||||
|
for i := 0; i+16 <= len(buf); i += 16 {
|
||||||
|
low ^= binary.LittleEndian.Uint64(buf[i:])
|
||||||
|
high ^= binary.LittleEndian.Uint64(buf[i+8:])
|
||||||
|
}
|
||||||
|
tail := uint64(0)
|
||||||
|
for _, b := range buf[len(buf)-len(buf)%16:] {
|
||||||
|
tail += uint64(b)
|
||||||
|
}
|
||||||
|
const k = 0x9e3779b97f4a7c15
|
||||||
|
want := (low + high + tail) * k ^ 0xabcdef
|
||||||
|
if got := fold(buf, 0xabcdef); got != want {
|
||||||
|
panic("fold")
|
||||||
|
}
|
||||||
|
println("ok", fold(buf, 0xabcdef))
|
||||||
|
}
|
||||||
|
`
|
||||||
|
|
||||||
|
// TestGOOBJLinkFoldKernel proves the amd64 SSE families are proven end to
|
||||||
|
// end: the kernel assembles by gasm into a GOOBJ object, substitutes
|
||||||
|
// byte-wise into a real go build archive, relinks with cmd/link and the
|
||||||
|
// linked binary computes the same fold as the toolchain-built one, run
|
||||||
|
// natively on amd64 hosts.
|
||||||
|
func TestGOOBJLinkFoldKernel(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("builds the gasm binary and links Go programs")
|
||||||
|
}
|
||||||
|
if os.Getenv("GASM_LINK_PARITY") == "" {
|
||||||
|
t.Skip("deliberate verification: set GASM_LINK_PARITY=1 (just link-parity)")
|
||||||
|
}
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
gasmBin := buildLinkParityGasm(t, goBin)
|
||||||
|
t.Run("amd64", func(t *testing.T) {
|
||||||
|
goobjLinkKernel(t, goBin, gasmBin, "amd64", "linkfold_amd64.s", goobjFoldMainSrc)
|
||||||
|
})
|
||||||
|
}
|
||||||
Vendored
+58
@@ -0,0 +1,58 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// The link-regression kernel for the amd64 SSE families: fold XOR-folds
|
||||||
|
// every 16-byte block of a buffer into one octa with MOVOU and PXOR,
|
||||||
|
// folds any remaining bytes with a scalar loop, combines the two octa
|
||||||
|
// lanes with PEXTRQ, passes the total to mixq through the ABI0 stack slot
|
||||||
|
// and XORs the seed into the returned mix. mixq multiplies by the odd
|
||||||
|
// golden-ratio constant, the same arithmetic the Go side of the regression
|
||||||
|
// mirrors bit for bit. The run claim is native: the host is amd64.
|
||||||
|
|
||||||
|
// func mixq(x uint64) uint64
|
||||||
|
TEXT ·mixq(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ x+0(FP), AX
|
||||||
|
MOVQ $0x9e3779b97f4a7c15, CX
|
||||||
|
IMULQ CX, AX
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func fold(buf []byte, seed uint64) uint64
|
||||||
|
TEXT ·fold(SB), NOSPLIT, $16-40
|
||||||
|
MOVQ buf+0(FP), SI
|
||||||
|
MOVQ buf_len+8(FP), CX
|
||||||
|
MOVQ seed+24(FP), BX
|
||||||
|
PXOR X1, X1
|
||||||
|
XORQ AX, AX
|
||||||
|
|
||||||
|
blocks:
|
||||||
|
CMPQ CX, $16
|
||||||
|
JB tail
|
||||||
|
MOVOU (SI), X2
|
||||||
|
PXOR X2, X1
|
||||||
|
ADDQ $16, SI
|
||||||
|
SUBQ $16, CX
|
||||||
|
JMP blocks
|
||||||
|
|
||||||
|
tail:
|
||||||
|
TESTQ CX, CX
|
||||||
|
JE gathered
|
||||||
|
MOVBQZX (SI), DX
|
||||||
|
ADDQ DX, AX
|
||||||
|
ADDQ $1, SI
|
||||||
|
SUBQ $1, CX
|
||||||
|
JMP tail
|
||||||
|
|
||||||
|
gathered:
|
||||||
|
MOVQ X1, DX
|
||||||
|
PEXTRQ $1, X1, CX
|
||||||
|
ADDQ CX, DX
|
||||||
|
ADDQ AX, DX
|
||||||
|
MOVQ DX, 0(SP)
|
||||||
|
CALL ·mixq(SB)
|
||||||
|
MOVQ 8(SP), DX
|
||||||
|
XORQ BX, DX
|
||||||
|
MOVQ DX, ret+32(FP)
|
||||||
|
RET
|
||||||
Reference in new issue
Block a user