66 lines
1.8 KiB
ArmAsm
66 lines
1.8 KiB
ArmAsm
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
#include "textflag.h"
|
|
|
|
// The vector link-regression kernel for riscv64: the RVV load/store and
|
|
// arithmetic families computing a checksum end to end. bytetotal
|
|
// strip-mines a byte buffer with VSETVLI, loads each chunk with VLE8V,
|
|
// widens it to E64 lanes with VSEXTVF8 and reduces the chunk with
|
|
// VREDSUMVS, so the result never depends on the hardware VLEN. checksum
|
|
// passes the slice to bytetotal through the ABI0 stack slots and mixes the
|
|
// reduction with the seed through the intra-file mix call. mix multiplies
|
|
// by the odd golden-ratio constant and XORs the seed, the same arithmetic
|
|
// the Go side of the regression mirrors bit for bit.
|
|
|
|
// func bytetotal(buf []byte) int64
|
|
TEXT ·bytetotal(SB), NOSPLIT, $0-32
|
|
MOV buf+0(FP), X10
|
|
MOV buf_len+8(FP), X12
|
|
MOV $0, X13
|
|
BEQZ X12, empty
|
|
|
|
loop:
|
|
VSETVLI X12, E64, M1, TA, MA, X11
|
|
VLE8V (X10), V16
|
|
VSEXTVF8 V16, V9
|
|
VMVVI $0, V10
|
|
VREDSUMVS V10, V9, V11
|
|
VMVXS V11, X14
|
|
ADD X14, X13, X13
|
|
ADD X11, X10, X10
|
|
SUB X11, X12, X12
|
|
BNE X12, X0, loop
|
|
|
|
empty:
|
|
MOV X13, ret+24(FP)
|
|
RET
|
|
|
|
// func mix(x, seed int64) int64
|
|
TEXT ·mix(SB), NOSPLIT, $0-24
|
|
MOV x+0(FP), X10
|
|
MOV seed+8(FP), X11
|
|
MOV $0x0101010101010101, X12
|
|
MUL X12, X10, X10
|
|
XOR X11, X10, X10
|
|
MOV X10, ret+16(FP)
|
|
RET
|
|
|
|
// func checksum(buf []byte, seed int64) int64
|
|
TEXT ·checksum(SB), NOSPLIT, $40-40
|
|
MOV buf+0(FP), X10
|
|
MOV buf+8(FP), X11
|
|
MOV buf+16(FP), X12
|
|
MOV X10, 8(SP)
|
|
MOV X11, 16(SP)
|
|
MOV X12, 24(SP)
|
|
CALL ·bytetotal(SB)
|
|
MOV 32(SP), X10
|
|
MOV seed+24(FP), X11
|
|
MOV X10, 8(SP)
|
|
MOV X11, 16(SP)
|
|
CALL ·mix(SB)
|
|
MOV 24(SP), X10
|
|
MOV X10, ret+32(FP)
|
|
RET
|