// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause #include "textflag.h" // The link-regression kernel for the amd64 SSE families: fold XOR-folds // every 16-byte block of a buffer into one octa with MOVOU and PXOR, // folds any remaining bytes with a scalar loop, combines the two octa // lanes with PEXTRQ, passes the total to mixq through the ABI0 stack slot // and XORs the seed into the returned mix. mixq multiplies by the odd // golden-ratio constant, the same arithmetic the Go side of the regression // mirrors bit for bit. The run claim is native: the host is amd64. // func mixq(x uint64) uint64 TEXT ·mixq(SB), NOSPLIT, $0-16 MOVQ x+0(FP), AX MOVQ $0x9e3779b97f4a7c15, CX IMULQ CX, AX MOVQ AX, ret+8(FP) RET // func fold(buf []byte, seed uint64) uint64 TEXT ·fold(SB), NOSPLIT, $16-40 MOVQ buf+0(FP), SI MOVQ buf_len+8(FP), CX MOVQ seed+24(FP), BX PXOR X1, X1 XORQ AX, AX blocks: CMPQ CX, $16 JB tail MOVOU (SI), X2 PXOR X2, X1 ADDQ $16, SI SUBQ $16, CX JMP blocks tail: TESTQ CX, CX JE gathered MOVBQZX (SI), DX ADDQ DX, AX ADDQ $1, SI SUBQ $1, CX JMP tail gathered: MOVQ X1, DX PEXTRQ $1, X1, CX ADDQ CX, DX ADDQ AX, DX MOVQ DX, 0(SP) CALL ·mixq(SB) MOVQ 8(SP), DX XORQ BX, DX MOVQ DX, ret+32(FP) RET