feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 21:17:20 +02:00
parent 81e2673923
commit b0f9071bf5
6 changed files with 318 additions and 30 deletions
+57
View File
@@ -0,0 +1,57 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 whole-vector moves between a general
// register and an arranged vector (VMOV/VDUP Rs, Vd.<T>), the two-operand
// accumulate spellings VADD/VSUB Vm, Vn, and the toolchain-reserved
// R18_PLATFORM register name. Every function is byte-compared against
// go tool asm.
#include "textflag.h"
// func gpIntoVector()
TEXT ·gpIntoVector(SB), NOSPLIT, $0-0
VMOV R1, V2.B8
VMOV R3, V4.B16
VMOV R5, V6.H4
VMOV R7, V8.H8
VMOV R9, V10.S2
VMOV R11, V12.S4
VMOV R13, V14.D2
VDUP R15, V16.B8
VDUP R17, V18.B16
VDUP R19, V20.H8
VDUP R21, V22.S4
VDUP R23, V24.D2
RET
// func simdAccumulate()
TEXT ·simdAccumulate(SB), NOSPLIT, $0-0
VADD V7, V8
VSUB V7, V8
VADD V1, V2
VSUB V30, V31
VADD V0.B16, V1.B16, V2.B16
VSUB V0.S4, V1.S4, V2.S4
RET
// func truncMove()
TEXT ·truncMove(SB), NOSPLIT, $0-0
MOVB R3, R4
MOVH R5, R6
MOVW R9, R10
MOVBU R3, R4
MOVHU R3, R4
MOVWU R3, R4
MOVD R3, R4
RET
// func platformRegister()
TEXT ·platformRegister(SB), NOSPLIT, $0-0
MOVD R18_PLATFORM, R3
MOVW R18_PLATFORM, R4
MOVD R3, R18_PLATFORM
MOVD 0x68(R18_PLATFORM), R5
MOVD R5, 0x68(R18_PLATFORM)
MOVW 8(R18_PLATFORM), R6
RET