Files
gasm-sdk/verify/testdata/linkfold_amd64.s
T
2026-10-07 01:31:28 +02:00

59 lines
1.4 KiB
ArmAsm

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// The link-regression kernel for the amd64 SSE families: fold XOR-folds
// every 16-byte block of a buffer into one octa with MOVOU and PXOR,
// folds any remaining bytes with a scalar loop, combines the two octa
// lanes with PEXTRQ, passes the total to mixq through the ABI0 stack slot
// and XORs the seed into the returned mix. mixq multiplies by the odd
// golden-ratio constant, the same arithmetic the Go side of the regression
// mirrors bit for bit. The run claim is native: the host is amd64.
// func mixq(x uint64) uint64
TEXT ·mixq(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
MOVQ $0x9e3779b97f4a7c15, CX
IMULQ CX, AX
MOVQ AX, ret+8(FP)
RET
// func fold(buf []byte, seed uint64) uint64
TEXT ·fold(SB), NOSPLIT, $16-40
MOVQ buf+0(FP), SI
MOVQ buf_len+8(FP), CX
MOVQ seed+24(FP), BX
PXOR X1, X1
XORQ AX, AX
blocks:
CMPQ CX, $16
JB tail
MOVOU (SI), X2
PXOR X2, X1
ADDQ $16, SI
SUBQ $16, CX
JMP blocks
tail:
TESTQ CX, CX
JE gathered
MOVBQZX (SI), DX
ADDQ DX, AX
ADDQ $1, SI
SUBQ $1, CX
JMP tail
gathered:
MOVQ X1, DX
PEXTRQ $1, X1, CX
ADDQ CX, DX
ADDQ AX, DX
MOVQ DX, 0(SP)
CALL ·mixq(SB)
MOVQ 8(SP), DX
XORQ BX, DX
MOVQ DX, ret+32(FP)
RET