60 lines
1.7 KiB
ArmAsm
60 lines
1.7 KiB
ArmAsm
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
#include "textflag.h"
|
|
|
|
// The link-regression kernel for the arm64 LSE atomics: atomicsum adds a
|
|
// value to a counter with LDADDALD in a loop, accumulating every old value
|
|
// the atomic returns, and finishes with the counter's final contents, so
|
|
// the result is addv times times times times plus one under any schedule.
|
|
// caller passes the pointer and the two scalars to atomicsum and mixes the
|
|
// total through the intra-file mixa call over the ABI0 stack slots. mixa
|
|
// multiplies by the odd golden-ratio constant, the same arithmetic the Go
|
|
// side of the regression mirrors bit for bit. The run claim follows the
|
|
// documented degradation: a qemu-user without LSE fails the baseline and
|
|
// the claim degrades to link-only.
|
|
|
|
// func atomicsum(p *int64, addv int64, times int64) int64
|
|
TEXT ·atomicsum(SB), NOSPLIT, $0-32
|
|
MOVD p+0(FP), R3
|
|
MOVD addv+8(FP), R4
|
|
MOVD times+16(FP), R5
|
|
MOVD $0, R6
|
|
|
|
loop:
|
|
CBZ R5, done
|
|
LDADDALD R4, (R3), R7
|
|
ADD R7, R6, R6
|
|
SUB $1, R5, R5
|
|
JMP loop
|
|
|
|
done:
|
|
MOVD (R3), R8
|
|
ADD R8, R6, R6
|
|
MOVD R6, ret+24(FP)
|
|
RET
|
|
|
|
// func mixa(x int64) int64
|
|
TEXT ·mixa(SB), NOSPLIT, $0-16
|
|
MOVD x+0(FP), R0
|
|
MOVD $0x9e3779b97f4a7c15, R1
|
|
MUL R1, R0, R0
|
|
MOVD R0, ret+8(FP)
|
|
RET
|
|
|
|
// func caller(p *int64, addv int64, times int64) int64
|
|
TEXT ·caller(SB), NOSPLIT, $32-32
|
|
MOVD p+0(FP), R3
|
|
MOVD addv+8(FP), R4
|
|
MOVD times+16(FP), R5
|
|
MOVD R3, 8(RSP)
|
|
MOVD R4, 16(RSP)
|
|
MOVD R5, 24(RSP)
|
|
BL ·atomicsum(SB)
|
|
MOVD 32(RSP), R0
|
|
MOVD R0, 8(RSP)
|
|
BL ·mixa(SB)
|
|
MOVD 16(RSP), R0
|
|
MOVD R0, ret+24(FP)
|
|
RET
|