test(verify): prove the amd64 sse families through the link-parity kernel
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
b22abf512f
commit
8e3b7f1caa
2 files changed
+121
No files matched your search
Vendored
+58
@@ -0,0 +1,58 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// The link-regression kernel for the amd64 SSE families: fold XOR-folds
|
||||
// every 16-byte block of a buffer into one octa with MOVOU and PXOR,
|
||||
// folds any remaining bytes with a scalar loop, combines the two octa
|
||||
// lanes with PEXTRQ, passes the total to mixq through the ABI0 stack slot
|
||||
// and XORs the seed into the returned mix. mixq multiplies by the odd
|
||||
// golden-ratio constant, the same arithmetic the Go side of the regression
|
||||
// mirrors bit for bit. The run claim is native: the host is amd64.
|
||||
|
||||
// func mixq(x uint64) uint64
|
||||
TEXT ·mixq(SB), NOSPLIT, $0-16
|
||||
MOVQ x+0(FP), AX
|
||||
MOVQ $0x9e3779b97f4a7c15, CX
|
||||
IMULQ CX, AX
|
||||
MOVQ AX, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func fold(buf []byte, seed uint64) uint64
|
||||
TEXT ·fold(SB), NOSPLIT, $16-40
|
||||
MOVQ buf+0(FP), SI
|
||||
MOVQ buf_len+8(FP), CX
|
||||
MOVQ seed+24(FP), BX
|
||||
PXOR X1, X1
|
||||
XORQ AX, AX
|
||||
|
||||
blocks:
|
||||
CMPQ CX, $16
|
||||
JB tail
|
||||
MOVOU (SI), X2
|
||||
PXOR X2, X1
|
||||
ADDQ $16, SI
|
||||
SUBQ $16, CX
|
||||
JMP blocks
|
||||
|
||||
tail:
|
||||
TESTQ CX, CX
|
||||
JE gathered
|
||||
MOVBQZX (SI), DX
|
||||
ADDQ DX, AX
|
||||
ADDQ $1, SI
|
||||
SUBQ $1, CX
|
||||
JMP tail
|
||||
|
||||
gathered:
|
||||
MOVQ X1, DX
|
||||
PEXTRQ $1, X1, CX
|
||||
ADDQ CX, DX
|
||||
ADDQ AX, DX
|
||||
MOVQ DX, 0(SP)
|
||||
CALL ·mixq(SB)
|
||||
MOVQ 8(SP), DX
|
||||
XORQ BX, DX
|
||||
MOVQ DX, ret+32(FP)
|
||||
RET
|
||||
Reference in new issue
Block a user