// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause #include "textflag.h" // The link-regression kernel for the arm64 LSE atomics: atomicsum adds a // value to a counter with LDADDALD in a loop, accumulating every old value // the atomic returns, and finishes with the counter's final contents, so // the result is addv times times times times plus one under any schedule. // caller passes the pointer and the two scalars to atomicsum and mixes the // total through the intra-file mixa call over the ABI0 stack slots. mixa // multiplies by the odd golden-ratio constant, the same arithmetic the Go // side of the regression mirrors bit for bit. The run claim follows the // documented degradation: a qemu-user without LSE fails the baseline and // the claim degrades to link-only. // func atomicsum(p *int64, addv int64, times int64) int64 TEXT ·atomicsum(SB), NOSPLIT, $0-32 MOVD p+0(FP), R3 MOVD addv+8(FP), R4 MOVD times+16(FP), R5 MOVD $0, R6 loop: CBZ R5, done LDADDALD R4, (R3), R7 ADD R7, R6, R6 SUB $1, R5, R5 JMP loop done: MOVD (R3), R8 ADD R8, R6, R6 MOVD R6, ret+24(FP) RET // func mixa(x int64) int64 TEXT ·mixa(SB), NOSPLIT, $0-16 MOVD x+0(FP), R0 MOVD $0x9e3779b97f4a7c15, R1 MUL R1, R0, R0 MOVD R0, ret+8(FP) RET // func caller(p *int64, addv int64, times int64) int64 TEXT ·caller(SB), NOSPLIT, $32-32 MOVD p+0(FP), R3 MOVD addv+8(FP), R4 MOVD times+16(FP), R5 MOVD R3, 8(RSP) MOVD R4, 16(RSP) MOVD R5, 24(RSP) BL ·atomicsum(SB) MOVD 32(RSP), R0 MOVD R0, 8(RSP) BL ·mixa(SB) MOVD 16(RSP), R0 MOVD R0, ret+24(FP) RET