58 lines
1.4 KiB
ArmAsm
58 lines
1.4 KiB
ArmAsm
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|||
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
||
|
|
|
||
|
|
#include "textflag.h"
|
||
|
|
|
||
|
|
// The link-regression kernel for the amd64 SSE families: fold XOR-folds
|
||
|
|
// every 16-byte block of a buffer into one octa with MOVOU and PXOR,
|
||
|
|
// folds any remaining bytes with a scalar loop, combines the two octa
|
||
|
|
// lanes with PEXTRQ, passes the total to mixq through the ABI0 stack slot
|
||
|
|
// and XORs the seed into the returned mix. mixq multiplies by the odd
|
||
|
|
// golden-ratio constant, the same arithmetic the Go side of the regression
|
||
|
|
// mirrors bit for bit. The run claim is native: the host is amd64.
|
||
|
|
|
||
|
|
// func mixq(x uint64) uint64
|
||
|
|
TEXT ·mixq(SB), NOSPLIT, $0-16
|
||
|
|
MOVQ x+0(FP), AX
|
||
|
|
MOVQ $0x9e3779b97f4a7c15, CX
|
||
|
|
IMULQ CX, AX
|
||
|
|
MOVQ AX, ret+8(FP)
|
||
|
|
RET
|
||
|
|
|
||
|
|
// func fold(buf []byte, seed uint64) uint64
|
||
|
|
TEXT ·fold(SB), NOSPLIT, $16-40
|
||
|
|
MOVQ buf+0(FP), SI
|
||
|
|
MOVQ buf_len+8(FP), CX
|
||
|
|
MOVQ seed+24(FP), BX
|
||
|
|
PXOR X1, X1
|
||
|
|
XORQ AX, AX
|
||
|
|
|
||
|
|
blocks:
|
||
|
|
CMPQ CX, $16
|
||
|
|
JB tail
|
||
|
|
MOVOU (SI), X2
|
||
|
|
PXOR X2, X1
|
||
|
|
ADDQ $16, SI
|
||
|
|
SUBQ $16, CX
|
||
|
|
JMP blocks
|
||
|
|
|
||
|
|
tail:
|
||
|
|
TESTQ CX, CX
|
||
|
|
JE gathered
|
||
|
|
MOVBQZX (SI), DX
|
||
|
|
ADDQ DX, AX
|
||
|
|
ADDQ $1, SI
|
||
|
|
SUBQ $1, CX
|
||
|
|
JMP tail
|
||
|
|
|
||
|
|
gathered:
|
||
|
|
MOVQ X1, DX
|
||
|
|
PEXTRQ $1, X1, CX
|
||
|
|
ADDQ CX, DX
|
||
|
|
ADDQ AX, DX
|
||
|
|
MOVQ DX, 0(SP)
|
||
|
|
CALL ·mixq(SB)
|
||
|
|
MOVQ 8(SP), DX
|
||
|
|
XORQ BX, DX
|
||
|
|
MOVQ DX, ret+32(FP)
|
||
|
|
RET
|