Files
gasm-sdk/testdata/verify/simd_arm64.s
T
petrbalvin c66a47973a
Test / test (push) Successful in 2m15s
fix(format): preserve square brackets in SIMD operands
Assisted-by: GLM 5.3 Flash
2026-09-20 09:58:25 +02:00

99 lines
2.7 KiB
ArmAsm

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 NEON slice: the logical and arithmetic
// three-register operations, permutations, comparisons, shifts, the crypto
// four-register group, element moves, table lookups and the structure
// loads and stores. Every function is byte-compared against go tool asm.
#include "textflag.h"
// func simdLogic()
TEXT ·simdLogic(SB), NOSPLIT, $0-0
VADD V1.B16, V2.B16, V3.B16
VADD V1.B8, V2.B8, V3.B8
VSUB V1.S4, V2.S4, V3.S4
VMUL V1.H8, V2.H8, V3.H8
VAND V4.B16, V4.B16, V9.B16
VORR V5.B16, V4.B16, V3.B16
VEOR V0.B16, V1.B16, V0.B16
VADDP V1.H8, V2.H8, V3.H8
VCMEQ V24.S4, V13.S4, V12.S4
VCMEQ $0, V2.H4, V3.H4
RET
// func simdPerm()
TEXT ·simdPerm(SB), NOSPLIT, $0-0
VZIP1 V16.H8, V3.H8, V19.H8
VZIP1 V6.D2, V9.D2, V11.D2
VZIP2 V22.D2, V25.D2, V21.D2
VREV32 V2.H8, V1.H8
VREV64 V2.S4, V3.S4
VUADDLV V31.S4, V11
VEXT $4, V2.B8, V1.B8, V3.B8
VEXT $8, V2.B16, V1.B16, V3.B16
RET
// func simdShift()
TEXT ·simdShift(SB), NOSPLIT, $0-0
VSHL $7, V22.D2, V25.D2
VSHL $24, V1.S4, V2.S4
VUSHR $6, V22.H8, V23.H8
VUSHR $56, V1.D2, V2.D2
VSRI $24, V1.S4, V2.S4
VSRI $56, V1.D2, V2.D2
RET
// func simdCrypto4()
TEXT ·simdCrypto4(SB), NOSPLIT, $0-0
VEOR3 V2.B16, V7.B16, V12.B16, V25.B16
VBCAX V1.B16, V2.B16, V26.B16, V31.B16
VXAR $63, V27.D2, V21.D2, V26.D2
VRAX1 V26.D2, V29.D2, V30.D2
VPMULL V2.D1, V1.D1, V3.Q1
VPMULL V2.B8, V1.B8, V3.H8
VPMULL2 V2.D2, V1.D2, V4.Q1
VPMULL2 V2.B16, V1.B16, V4.H8
RET
// func simdElement()
TEXT ·simdElement(SB), NOSPLIT, $0-0
VDUP V31.B[15], V18
VDUP V19.S[3], V18.S4
VDUP V1.D[1], V2.D2
VMOV V13.S[0], R20
VMOV V11.B[11], V16.B[12]
VMOV R20, V21.B[2]
VMOV V2.B16, V4.B16
RET
// func simdTable()
TEXT ·simdTable(SB), NOSPLIT, $0-0
VTBL V22.B16, [V28.B16], V11.B16
VTBL V18.B8, [V17.B16, V18.B16], V22.B8
VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8
RET
// func simdLoadStore()
TEXT ·simdLoadStore(SB), NOSPLIT, $0-0
VLD1 (R2), [V21.B16]
VLD1 (R24), [V18.D1, V19.D1, V20.D1]
VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]
VLD1.P 32(R1), [V2.B16, V3.B16]
VLD1.P 64(R4), [V5.B16, V6.B16, V7.B16, V8.B16]
VLD1R (R1), [V9.B8]
VLD1R (R0), [V0.B16]
VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]
VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)
VST1 [V14.H4, V15.H4, V16.H4], (R27)
VST1.P [V2.B16], (R1)
VST1.P [V2.B16, V3.B16], 32(R1)
RET
// func simdLiteral()
TEXT ·simdLiteral(SB), NOSPLIT, $0-0
VMOVS $0x80402010, V11
VMOVD $0x8040201008040201, V20
VMOVQ $0x7040201008040201, $0x8040201008040201, V10
RET