feat(arm64): encode pairs, atomics, crypto, system and NEON slices
Assisted-by: GLM 5.3 Flash
This commit is contained in:
Vendored
+72
@@ -0,0 +1,72 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Differential kernel for the arm64 synchronisation instructions: the
|
||||
// acquire/release loads and stores, the exclusive family and the LSE
|
||||
// atomics with acquire and release semantics, plus the register-pair
|
||||
// loads and stores. Every function is byte-compared against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func acquireRelease()
|
||||
TEXT ·acquireRelease(SB), NOSPLIT, $0-0
|
||||
LDAR (R1), R2
|
||||
LDARB (R3), R4
|
||||
LDARH (R5), R6
|
||||
LDARW (R7), R8
|
||||
STLR R2, (R1)
|
||||
STLRB R4, (R3)
|
||||
STLRH R6, (R5)
|
||||
STLRW R8, (R7)
|
||||
RET
|
||||
|
||||
// func exclusive()
|
||||
TEXT ·exclusive(SB), NOSPLIT, $0-0
|
||||
LDAXR (R1), R2
|
||||
LDAXRB (R3), R4
|
||||
LDAXRW (R5), R6
|
||||
STLXR R2, (R1), R8
|
||||
STLXRB R4, (R3), R8
|
||||
STLXRW R6, (R5), R8
|
||||
RET
|
||||
|
||||
// func lseAcquireRelease()
|
||||
TEXT ·lseAcquireRelease(SB), NOSPLIT, $0-0
|
||||
CASALD R1, (R3), R2
|
||||
CASALW R4, (R6), R5
|
||||
LDADDALD R1, (R3), R2
|
||||
LDADDALW R4, (R6), R5
|
||||
LDCLRALB R1, (R3), R2
|
||||
LDCLRALW R4, (R6), R5
|
||||
LDCLRALD R1, (R3), R2
|
||||
LDORALB R1, (R3), R2
|
||||
LDORALW R4, (R6), R5
|
||||
LDORALD R1, (R3), R2
|
||||
SWPALB R1, (R3), R2
|
||||
SWPALW R4, (R6), R5
|
||||
SWPALD R1, (R3), R2
|
||||
RET
|
||||
|
||||
// func lseBase()
|
||||
TEXT ·lseBase(SB), NOSPLIT, $0-0
|
||||
LDADDD R1, (R3), R2
|
||||
LDADDW R4, (R6), R5
|
||||
CASD R1, (R3), R2
|
||||
CASW R4, (R6), R5
|
||||
SWPD R1, (R3), R2
|
||||
SWPW R4, (R6), R5
|
||||
RET
|
||||
|
||||
// func pairs()
|
||||
TEXT ·pairs(SB), NOSPLIT, $0-0
|
||||
LDP (R1), (R2, R3)
|
||||
LDP 8(R4), (R5, R6)
|
||||
LDP -16(R1), (R2, R3)
|
||||
LDPW 4(R4), (R5, R6)
|
||||
STP (R2, R3), 24(R7)
|
||||
STP (R2, R3),-8(R7)
|
||||
STPW (R1, R2), 4(R0)
|
||||
FLDPD (R8), (F1, F2)
|
||||
FLDPD 8(R8), (F3, F4)
|
||||
FSTPD (F3, F4),-8(R9)
|
||||
RET
|
||||
Vendored
+42
@@ -0,0 +1,42 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Differential kernel for the arm64 cryptographic extension: the AES round
|
||||
// instructions and the SHA1, SHA256 and SHA512 families. Every function is
|
||||
// byte-compared against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func aesRound()
|
||||
TEXT ·aesRound(SB), NOSPLIT, $0-0
|
||||
AESE V31.B16, V29.B16
|
||||
AESD V22.B16, V19.B16
|
||||
AESMC V14.B16, V28.B16
|
||||
AESIMC V12.B16, V27.B16
|
||||
RET
|
||||
|
||||
// func sha1Round()
|
||||
TEXT ·sha1Round(SB), NOSPLIT, $0-0
|
||||
SHA1C V8.S4, V8, V2
|
||||
SHA1P V3.S4, V20, V27
|
||||
SHA1M V0.S4, V27, V27
|
||||
SHA1H V17, V25
|
||||
SHA1SU0 V17.S4, V13.S4, V16.S4
|
||||
SHA1SU1 V24.S4, V23.S4
|
||||
RET
|
||||
|
||||
// func sha256Round()
|
||||
TEXT ·sha256Round(SB), NOSPLIT, $0-0
|
||||
SHA256H V4.S4, V2, V11
|
||||
SHA256H2 V6.S4, V16, V11
|
||||
SHA256SU0 V0.S4, V16.S4
|
||||
SHA256SU1 V31.S4, V3.S4, V15.S4
|
||||
RET
|
||||
|
||||
// func sha512Round()
|
||||
TEXT ·sha512Round(SB), NOSPLIT, $0-0
|
||||
SHA512H V2.D2, V1, V0
|
||||
SHA512H2 V4.D2, V3, V2
|
||||
SHA512SU0 V9.D2, V8.D2
|
||||
SHA512SU1 V7.D2, V6.D2, V5.D2
|
||||
RET
|
||||
Vendored
+66
@@ -0,0 +1,66 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Differential kernel for the arm64 integer slice: carry-setting arithmetic,
|
||||
// widening multiplies, bit manipulation, conditional compares, the compare
|
||||
// and test branches, ADR and the wide-constant moves. Every function is
|
||||
// byte-compared against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func carryArith()
|
||||
TEXT ·carryArith(SB), NOSPLIT, $0-0
|
||||
ADC R0, R2, R12
|
||||
ADCS R23, R22, R22
|
||||
ADC $0, R1
|
||||
SBC R25, R10, R26
|
||||
SBCS R5, R9, R5
|
||||
SBCS $0, R1
|
||||
RET
|
||||
|
||||
// func wideningMul()
|
||||
TEXT ·wideningMul(SB), NOSPLIT, $0-0
|
||||
MUL R4, R3, R0
|
||||
MSUB R19, R16, R26, R2
|
||||
SMULH R24, R20, R24
|
||||
UMULH R24, R20, R24
|
||||
RET
|
||||
|
||||
// func bitManip()
|
||||
TEXT ·bitManip(SB), NOSPLIT, $0-0
|
||||
RBIT R11, R4
|
||||
REV R1, R2
|
||||
CLZ R21, R9
|
||||
REVW R1, R2
|
||||
CLSW R1, R2
|
||||
UBFX $33, R17, $25, R5
|
||||
UBFXW $4, R1, $9, R2
|
||||
RET
|
||||
|
||||
// func condCompare()
|
||||
TEXT ·condCompare(SB), NOSPLIT, $0-0
|
||||
CCMP LE, R7, $19, $3
|
||||
CCMP LT, R30, R6, $7
|
||||
CCMN EQ, R1, R2, $3
|
||||
CCMPW LE, R7, $19, $3
|
||||
RET
|
||||
|
||||
// func branchForms()
|
||||
TEXT ·branchForms(SB), NOSPLIT, $0-0
|
||||
CBZ R1, target
|
||||
CBNZ R7, target
|
||||
CBNZW R2, target
|
||||
TBZ $4, R7, target
|
||||
TBNZ $33, R7, target
|
||||
ADR target, R10
|
||||
|
||||
target:
|
||||
RET
|
||||
|
||||
// func wideMoves()
|
||||
TEXT ·wideMoves(SB), NOSPLIT, $0-0
|
||||
MOVK $1234, R5
|
||||
MOVK $305397760, R5
|
||||
MOVKW $1234, R5
|
||||
MOVK $16771847290880, R21
|
||||
RET
|
||||
Vendored
+98
@@ -0,0 +1,98 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Differential kernel for the arm64 NEON slice: the logical and arithmetic
|
||||
// three-register operations, permutations, comparisons, shifts, the crypto
|
||||
// four-register group, element moves, table lookups and the structure
|
||||
// loads and stores. Every function is byte-compared against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func simdLogic()
|
||||
TEXT ·simdLogic(SB), NOSPLIT, $0-0
|
||||
VADD V1.B16, V2.B16, V3.B16
|
||||
VADD V1.B8, V2.B8, V3.B8
|
||||
VSUB V1.S4, V2.S4, V3.S4
|
||||
VMUL V1.H8, V2.H8, V3.H8
|
||||
VAND V4.B16, V4.B16, V9.B16
|
||||
VORR V5.B16, V4.B16, V3.B16
|
||||
VEOR V0.B16, V1.B16, V0.B16
|
||||
VADDP V1.H8, V2.H8, V3.H8
|
||||
VCMEQ V24.S4, V13.S4, V12.S4
|
||||
VCMEQ $0, V2.H4, V3.H4
|
||||
RET
|
||||
|
||||
// func simdPerm()
|
||||
TEXT ·simdPerm(SB), NOSPLIT, $0-0
|
||||
VZIP1 V16.H8, V3.H8, V19.H8
|
||||
VZIP1 V6.D2, V9.D2, V11.D2
|
||||
VZIP2 V22.D2, V25.D2, V21.D2
|
||||
VREV32 V2.H8, V1.H8
|
||||
VREV64 V2.S4, V3.S4
|
||||
VUADDLV V31.S4, V11
|
||||
VEXT $4, V2.B8, V1.B8, V3.B8
|
||||
VEXT $8, V2.B16, V1.B16, V3.B16
|
||||
RET
|
||||
|
||||
// func simdShift()
|
||||
TEXT ·simdShift(SB), NOSPLIT, $0-0
|
||||
VSHL $7, V22.D2, V25.D2
|
||||
VSHL $24, V1.S4, V2.S4
|
||||
VUSHR $6, V22.H8, V23.H8
|
||||
VUSHR $56, V1.D2, V2.D2
|
||||
VSRI $24, V1.S4, V2.S4
|
||||
VSRI $56, V1.D2, V2.D2
|
||||
RET
|
||||
|
||||
// func simdCrypto4()
|
||||
TEXT ·simdCrypto4(SB), NOSPLIT, $0-0
|
||||
VEOR3 V2.B16, V7.B16, V12.B16, V25.B16
|
||||
VBCAX V1.B16, V2.B16, V26.B16, V31.B16
|
||||
VXAR $63, V27.D2, V21.D2, V26.D2
|
||||
VRAX1 V26.D2, V29.D2, V30.D2
|
||||
VPMULL V2.D1, V1.D1, V3.Q1
|
||||
VPMULL V2.B8, V1.B8, V3.H8
|
||||
VPMULL2 V2.D2, V1.D2, V4.Q1
|
||||
VPMULL2 V2.B16, V1.B16, V4.H8
|
||||
RET
|
||||
|
||||
// func simdElement()
|
||||
TEXT ·simdElement(SB), NOSPLIT, $0-0
|
||||
VDUP V31.B[15], V18
|
||||
VDUP V19.S[3], V18.S4
|
||||
VDUP V1.D[1], V2.D2
|
||||
VMOV V13.S[0], R20
|
||||
VMOV V11.B[11], V16.B[12]
|
||||
VMOV R20, V21.B[2]
|
||||
VMOV V2.B16, V4.B16
|
||||
RET
|
||||
|
||||
// func simdTable()
|
||||
TEXT ·simdTable(SB), NOSPLIT, $0-0
|
||||
VTBL V22.B16, [V28.B16], V11.B16
|
||||
VTBL V18.B8, [V17.B16, V18.B16], V22.B8
|
||||
VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8
|
||||
RET
|
||||
|
||||
// func simdLoadStore()
|
||||
TEXT ·simdLoadStore(SB), NOSPLIT, $0-0
|
||||
VLD1 (R2), [V21.B16]
|
||||
VLD1 (R24), [V18.D1, V19.D1, V20.D1]
|
||||
VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]
|
||||
VLD1.P 32(R1), [V2.B16, V3.B16]
|
||||
VLD1.P 64(R4), [V5.B16, V6.B16, V7.B16, V8.B16]
|
||||
VLD1R (R1), [V9.B8]
|
||||
VLD1R (R0), [V0.B16]
|
||||
VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]
|
||||
VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)
|
||||
VST1 [V14.H4, V15.H4, V16.H4], (R27)
|
||||
VST1.P [V2.B16], (R1)
|
||||
VST1.P [V2.B16, V3.B16], 32(R1)
|
||||
RET
|
||||
|
||||
// func simdLiteral()
|
||||
TEXT ·simdLiteral(SB), NOSPLIT, $0-0
|
||||
VMOVS $0x80402010, V11
|
||||
VMOVD $0x8040201008040201, V20
|
||||
VMOVQ $0x7040201008040201, $0x8040201008040201, V10
|
||||
RET
|
||||
Vendored
+61
@@ -0,0 +1,61 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Differential kernel for the arm64 system instructions: barriers,
|
||||
// cache maintenance, the system register accesses, supervisor calls,
|
||||
// breakpoints and prefetches. Every function is byte-compared against
|
||||
// go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func barriers()
|
||||
TEXT ·barriers(SB), NOSPLIT, $0-0
|
||||
DMB $15
|
||||
DMB $1
|
||||
DSB $15
|
||||
DSB $4
|
||||
ISB $15
|
||||
ISB $1
|
||||
RET
|
||||
|
||||
// func cacheOps()
|
||||
TEXT ·cacheOps(SB), NOSPLIT, $0-0
|
||||
DC ZVA, R4
|
||||
DC IVAC, R1
|
||||
DC CVAC, R2
|
||||
DC CVAU, R3
|
||||
DC CIVAC, R7
|
||||
RET
|
||||
|
||||
// func sysRegs()
|
||||
TEXT ·sysRegs(SB), NOSPLIT, $0-0
|
||||
MRS DCZID_EL0, R3
|
||||
MRS CNTVCT_EL0, R0
|
||||
MRS CNTPCT_EL0, R1
|
||||
MRS CNTFRQ_EL0, R2
|
||||
MRS MIDR_EL1, R0
|
||||
MRS ID_AA64PFR0_EL1, R0
|
||||
MRS ID_AA64ISAR0_EL1, R0
|
||||
MRS ID_AA64ISAR1_EL1, R0
|
||||
MRS DIT, R0
|
||||
MSR $3, SPSel
|
||||
MSR $9, DAIFSet
|
||||
MSR $6, DAIFClr
|
||||
MSR $1, DIT
|
||||
RET
|
||||
|
||||
// func exceptions()
|
||||
TEXT ·exceptions(SB), NOSPLIT, $0-0
|
||||
SVC $0
|
||||
SVC $7165
|
||||
BRK
|
||||
BRK $35943
|
||||
RET
|
||||
|
||||
// func prefetch()
|
||||
TEXT ·prefetch(SB), NOSPLIT, $0-0
|
||||
PRFM (R0), PLDL1KEEP
|
||||
PRFM (R3), PLDL3KEEP
|
||||
PRFM (R4), PSTL1KEEP
|
||||
PRFM (R2), $25
|
||||
RET
|
||||
Reference in New Issue
Block a user