feat(arm64): encode pairs, atomics, crypto, system and NEON slices

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 06:44:51 +02:00
parent fc2d92eabd
commit ca3fdce0e0
9 changed files with 2377 additions and 30 deletions
+27
View File
@@ -150,6 +150,33 @@ func arm64Curated() []Instr {
t = append(t, i(op, "Atomic memory operation"))
}
// Register-pair loads and stores.
for _, op := range []string{"LDP", "STP", "LDPW", "STPW", "FLDPD", "FSTPD"} {
t = append(t, ic(op, "Register-pair load or store", 2, 2))
}
// Cache maintenance and prefetch.
t = append(t, i("DC", "Data cache maintenance"))
t = append(t, i("PRFM", "Memory prefetch"))
for _, op := range []string{"LDADDAL", "LDCLRAL", "LDORAL", "SWPAL"} {
t = append(t, i(op, "Atomic memory operation with acquire and release semantics"))
}
// Cryptographic extensions.
for _, op := range []string{"AESE", "AESD", "AESMC", "AESIMC"} {
t = append(t, i(op, "AES round"))
}
for _, op := range []string{
"SHA1C", "SHA1P", "SHA1M", "SHA1H", "SHA1SU0", "SHA1SU1",
"SHA256H", "SHA256H2", "SHA256SU0", "SHA256SU1",
"SHA512H", "SHA512H2", "SHA512SU0", "SHA512SU1",
} {
t = append(t, i(op, "SHA round"))
}
for _, op := range []string{"VEOR3", "VBCAX", "VXAR", "VRAX1"} {
t = append(t, i(op, "Three-way XOR / rotate crypto vector operation"))
}
// Floating-point scalar.
for _, op := range []string{
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
+1139 -19
View File
File diff suppressed because it is too large Load Diff
+423 -6
View File
@@ -27,7 +27,13 @@ package asm
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
import "maps"
import (
"maps"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
@@ -288,7 +294,25 @@ const (
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
a64FCondCmp // conditional compare: CCMP, CCMN
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
a64FVTBL // SIMD table lookup: VTBL
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
// a64Enc is one instruction's encoding: its bit layout (format) and the
@@ -622,10 +646,403 @@ func init() {
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
// ---- SIMD basics ----
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
// ---- SIMD: the arrangement-aware tables in this file carry VADD,
// VSUB, VMUL and every other three-register vector op. ----
// ---- data-processing (1 source): sf 10 11010110 opcode 00000 Rn Rd ----
dp1 := map[string]uint32{
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
}
for m, op := range dp1 {
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
}
// ---- bitfield extract: the UBFM/SBFM bases, immediate operands wrap ----
a64InstrTable["UBFX"] = a64Enc{format: a64FBitfield2, op: 0xd3400000}
a64InstrTable["SBFX"] = a64Enc{format: a64FBitfield2, op: 0x93400000}
a64InstrTable["UBFXW"] = a64Enc{format: a64FBitfield2, op: 0x53000000}
a64InstrTable["SBFXW"] = a64Enc{format: a64FBitfield2, op: 0x13000000}
// ---- conditional compare: sf 1 1 101001 0 imm5/Rm cond op2 Rn nzcv ----
a64InstrTable["CCMP"] = a64Enc{format: a64FCondCmp, op: 0xfa400000}
a64InstrTable["CCMN"] = a64Enc{format: a64FCondCmp, op: 0xba400000}
a64InstrTable["CCMPW"] = a64Enc{format: a64FCondCmp, op: 0x7a400000}
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
// ---- compare/test and branch ----
a64InstrTable["CBZ"] = a64Enc{format: a64FBranch19, op: 0xb4000000}
a64InstrTable["CBZW"] = a64Enc{format: a64FBranch19, op: 0x34000000}
a64InstrTable["CBNZ"] = a64Enc{format: a64FBranch19, op: 0xb5000000}
a64InstrTable["CBNZW"] = a64Enc{format: a64FBranch19, op: 0x35000000}
a64InstrTable["TBZ"] = a64Enc{format: a64FTestBranch, op: 0x36000000}
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
// ---- load/store pair (signed offset) ----
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
// ---- acquire/release loads and stores ----
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
a64InstrTable["LDARB"] = a64Enc{format: a64FAcqRel, op: 0x08dffc00}
a64InstrTable["LDARH"] = a64Enc{format: a64FAcqRel, op: 0x48dffc00}
a64InstrTable["LDARW"] = a64Enc{format: a64FAcqRel, op: 0x88dffc00}
a64InstrTable["STLR"] = a64Enc{format: a64FAcqRel, op: 0xc89ffc00}
a64InstrTable["STLRB"] = a64Enc{format: a64FAcqRel, op: 0x089ffc00}
a64InstrTable["STLRH"] = a64Enc{format: a64FAcqRel, op: 0x489ffc00}
a64InstrTable["STLRW"] = a64Enc{format: a64FAcqRel, op: 0x889ffc00}
// ---- LSE atomics with acquire and release semantics ----
// CAS carries a preset fixed op field and a real Rs; the LDADD/LDCLR/
// LDOR/SWP families leave Rs free for the returned value.
lse := map[string]uint32{
"CASALD": 0xc8e0fc00,
"CASALW": 0x88e0fc00,
"LDADDALD": 0xf8e00000,
"LDADDALW": 0xb8e00000,
"LDCLRALB": 0x38e01000,
"LDCLRALW": 0xb8e01000,
"LDCLRALD": 0xf8e01000,
"LDORALB": 0x38e03000,
"LDORALW": 0xb8e03000,
"LDORALD": 0xf8e03000,
"SWPALB": 0x38e08000,
"SWPALW": 0xb8e08000,
"SWPALD": 0xf8e08000,
}
for m, op := range lse {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
// register preset to ZR (bits 14:10 = 11111).
dpsrExtra := map[string]uint32{
"ADC": 0x9a000000, "ADCW": 0x1a000000,
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
"SBC": 0xda000000, "SBCW": 0x5a000000,
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
}
for m, op := range dpsrExtra {
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
}
// ---- crypto, 2-register (Rn, Rd) and 3-register (Rm, Rn, Rd) forms ----
crypto2 := map[string]uint32{
"AESD": 0x4e285800, "AESE": 0x4e284800,
"AESIMC": 0x4e287800, "AESMC": 0x4e286800,
"SHA1H": 0x5e280800, "SHA1SU1": 0x5e281800,
"SHA256SU0": 0x5e282800, "SHA512SU0": 0xcec08000,
}
for m, op := range crypto2 {
a64InstrTable[m] = a64Enc{format: a64FCrypto2, op: op}
}
crypto3 := map[string]uint32{
"SHA1C": 0x5e000000, "SHA1P": 0x5e001000,
"SHA1M": 0x5e002000, "SHA1SU0": 0x5e003000,
"SHA256H": 0x5e004000, "SHA256H2": 0x5e005000,
"SHA256SU1": 0x5e006000, "SHA512H": 0xce608000,
"SHA512H2": 0xce608400, "SHA512SU1": 0xce608800,
}
for m, op := range crypto3 {
a64InstrTable[m] = a64Enc{format: a64FCrypto3, op: op}
}
// ---- arrangement-aware SIMD, see a64SimdVTable and a64SimdV2Table ----
a64InstrTable["VEOR3"] = a64Enc{format: a64FSIMDV4, op: 0xce000000}
a64InstrTable["VBCAX"] = a64Enc{format: a64FSIMDV4, op: 0xce200000}
a64InstrTable["VXAR"] = a64Enc{format: a64FSIMDV4, op: 0xce800000}
a64InstrTable["VEXT"] = a64Enc{format: a64FSIMDV4, op: 0x2e000000}
a64InstrTable["VTBL"] = a64Enc{format: a64FVTBL}
a64InstrTable["VDUP"] = a64Enc{format: a64FDUP}
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
// for instructions that exist at a single arrangement and carry that
// arrangement's bits inside the base already, the fixed flag.
type a64SimdVSpec struct {
base uint32
arrs uint16
fixed bool
}
// a64Arr names the vector arrangements the encoders deal with, indexed by
// a64Arr. The source spellings put the element letter first: B8, H4, S2,
// D1 and the 128-bit halves B16, H8, S4, D2.
const (
a64Arr8B = iota
a64Arr16B
a64Arr4H
a64Arr8H
a64Arr2S
a64Arr4S
a64Arr2D
a64ArrD1
a64ArrQ1
a64ArrCount
)
// a64ArrNames maps an arrangement to its source spelling (element letter
// first, as the toolchain writes it).
var a64ArrNames = [a64ArrCount]string{
a64Arr8B: "B8", a64Arr16B: "B16", a64Arr4H: "H4", a64Arr8H: "H8",
a64Arr2S: "S2", a64Arr4S: "S4", a64Arr2D: "D2", a64ArrD1: "D1", a64ArrQ1: "Q1",
}
// a64ArrIndex resolves a source spelling to its a64Arr index, -1 when
// unknown.
func a64ArrIndex(s string) int {
for i, n := range a64ArrNames {
if n == s {
return i
}
}
return -1
}
// a64ElemLetter reports whether s is a bare element spelling (B, H, S, D, Q)
// as it appears in element operands such as V13.S[0].
func a64ElemLetter(s string) bool {
switch s {
case "B", "H", "S", "D", "Q":
return true
}
return false
}
// a64ArrBits carries the fixed bits an arrangement contributes to the
// three-same word shape: the element size at bits 23:22 and the 128-bit
// flag at bit 30. Bit 29 belongs to the instruction's own base.
var a64ArrBits = [a64ArrCount]uint32{
a64Arr8B: 0,
a64Arr16B: 1 << 30,
a64Arr4H: 1 << 22,
a64Arr8H: 1<<30 | 1<<22,
a64Arr2S: 1 << 23,
a64Arr4S: 1<<30 | 1<<23,
a64Arr2D: 1<<30 | 1<<23 | 1<<22,
a64ArrD1: 1<<23 | 1<<22,
a64ArrQ1: 0,
}
// a64SimdVTable holds the arrangement-aware three-register SIMD
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
// word and arrangement bit was read off go tool asm.
var a64SimdVTable = map[string]a64SimdVSpec{
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
}
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
// the register pair spelling ORR Vd, Vn, Vm.
var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false},
"VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
}
// a64CryptoArr is the arrangement each crypto instruction's operands must
// carry when they spell one at all; a bare V/F spelling is accepted as is.
var a64CryptoArr = map[string]int{
"AESD": a64Arr16B, "AESE": a64Arr16B, "AESIMC": a64Arr16B, "AESMC": a64Arr16B,
"SHA1H": a64Arr4S, "SHA1SU1": a64Arr4S, "SHA256SU0": a64Arr4S, "SHA512SU0": a64Arr2D,
"SHA1C": a64Arr4S, "SHA1P": a64Arr4S, "SHA1M": a64Arr4S, "SHA1SU0": a64Arr4S,
"SHA256H": a64Arr4S, "SHA256H2": a64Arr4S, "SHA256SU1": a64Arr4S,
"SHA512H": a64Arr2D, "SHA512H2": a64Arr2D, "SHA512SU1": a64Arr2D,
}
// a64DCOps maps the data-cache maintenance operation names to their fixed
// word (the register rides bits 4:0).
var a64DCOps = map[string]uint32{
"IVAC": 0xd5087620, "ZVA": 0xd50b7420,
"CVAC": 0xd50b7a20, "CVAU": 0xd50b7b20, "CIVAC": 0xd50b7e20,
}
// a64MRSOps maps the system register names GOROOT reads to their fixed word
// (the destination register rides bits 4:0).
var a64MRSOps = map[string]uint32{
"ELR_EL1": 0xd5384020, "MIDR_EL1": 0xd5380000,
"ID_AA64PFR0_EL1": 0xd5380400, "ID_AA64ISAR0_EL1": 0xd5380600,
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
}
// a64MSROps maps the system register names GOROOT writes to their fixed
// word; the immediate rides CRm at bits 11:8 and Rt is the fixed 11111.
var a64MSROps = map[string]uint32{
"SPSel": 0xd50040a0, "DAIFSet": 0xd50340c0, "DAIFClr": 0xd50340e0, "DIT": 0xd5034040,
}
// a64PRFOps maps the prefetch operation names to their prfop immediate
// (word = 0xf9800000 | Rn<<5 | prfop).
var a64PRFOps = map[string]int{
"PLDL1KEEP": 0x00, "PLDL1STRM": 0x01, "PLDL2KEEP": 0x02, "PLDL2STRM": 0x03,
"PLDL3KEEP": 0x04, "PLDL3STRM": 0x05,
"PLIL1KEEP": 0x08, "PLIL1STRM": 0x09, "PLIL2KEEP": 0x0a, "PLIL2STRM": 0x0b,
"PLIL3KEEP": 0x0c, "PLIL3STRM": 0x0d,
"PSTL1KEEP": 0x10, "PSTL1STRM": 0x11, "PSTL2KEEP": 0x12, "PSTL2STRM": 0x13,
"PSTL3KEEP": 0x14, "PSTL3STRM": 0x15,
}
// a64VLD1Base holds the fixed words of the multi-register structure
// accesses, indexed by register count 1..4, before the Q and size bits.
// Post-index spellings add 0x9f0000 (post bit and Rm = 11111).
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
// a64Vec is a parsed vector operand: the register number, the arrangement
// ("" when the operand spells none) and, for element forms, the lane index.
type a64Vec struct {
reg int
arr string
idx int
hasIdx bool
}
// a64VecReg parses a vector register operand: V0..V31 (F0..F31 as an alias,
// the same architectural registers the scalar floating-point spellings use),
// optionally with an arrangement suffix such as V0.B16 and, for element
// forms, a lane index such as V13.S[0]. It reports ok=false for anything
// else, including X/W and R spellings, which the toolchain's vector
// operands reject as well.
func a64VecReg(name string) (v a64Vec, ok bool) {
s := strings.TrimSpace(name)
if i := strings.IndexByte(s, '.'); i >= 0 {
v.arr = strings.TrimSpace(s[i+1:])
s = s[:i]
}
if v.arr != "" {
// Element form: B[3], S[2] and friends.
if j := strings.IndexByte(v.arr, '['); j >= 0 {
k := strings.LastIndexByte(v.arr, ']')
if k < j {
return v, false
}
n, err := strconv.Atoi(strings.TrimSpace(v.arr[j+1 : k]))
if err != nil || n < 0 {
return v, false
}
v.idx, v.hasIdx = n, true
v.arr = strings.TrimSpace(v.arr[:j])
}
if a64ArrIndex(v.arr) < 0 && !a64ElemLetter(v.arr) {
return v, false
}
}
if len(s) < 2 || (s[0] != 'V' && s[0] != 'F') {
return v, false
}
n := 0
for i := 1; i < len(s); i++ {
if s[i] < '0' || s[i] > '9' {
return v, false
}
n = n*10 + int(s[i]-'0')
}
if n > 31 {
return v, false
}
v.reg = n
return v, true
}
// a64ElemField encodes a lane index for the copy/insert group: imm5 = the
// index shifted by the element scale, with the scale's own bit set. B gets
// shift 1 (the Q bit rides elsewhere), H shift 2, S shift 3 and D shift 4.
func a64ElemField(arr string, idx int) (uint32, bool) {
var shift, low uint32
switch arr {
case "B8", "B16", "B":
shift, low = 1, 1
case "H4", "H8", "H":
shift, low = 2, 2
case "S2", "S4", "S":
shift, low = 3, 4
case "D1", "D2", "D":
shift, low = 4, 8
default:
return 0, false
}
if idx < 0 || idx >= 1<<(5-shift) {
return 0, false
}
return uint32(idx)<<shift | low, true
}
// a64VecListOf recovers the register list of a VLD1/VST1/VTBL operand run.
// The parser keeps parenthesised groups whole but splits bracketed lists on
// the commas, so a list arrives as one operand run whose first Raw starts
// with "[" and whose last Raw ends with "]". It returns the parsed
// registers with the brackets and spaces removed.
func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool) {
if start >= len(ops) || !strings.HasPrefix(strings.TrimSpace(ops[start].Raw), "[") {
return nil, 0, false
}
end = start
for end < len(ops) {
if strings.HasSuffix(strings.TrimSpace(ops[end].Raw), "]") {
break
}
end++
}
if end >= len(ops) {
return nil, 0, false
}
for i := start; i <= end; i++ {
s := strings.TrimSpace(ops[i].Raw)
s = strings.TrimPrefix(s, "[")
s = strings.TrimSuffix(s, "]")
if s == "" && len(ops) > start+1 {
return nil, 0, false
}
for part := range strings.SplitSeq(s, ",") {
v, ok := a64VecReg(part)
if !ok {
return nil, 0, false
}
vs = append(vs, v)
}
}
return vs, end, true
}
// ---- load/store helper tables ----
+449 -5
View File
@@ -472,12 +472,456 @@ TEXT ·f(SB), NOSPLIT, $0-0
}
}
// TestArm64SIMD tests SIMD encoding (via the instruction table).
// TestArm64SIMD tests SIMD encoding (via the arrangement-aware table).
func TestArm64SIMD(t *testing.T) {
// Verify SIMD instructions are in the table.
for _, mnem := range []string{"VADD", "VSUB", "VMUL"} {
if _, ok := a64InstrTable[mnem]; !ok {
t.Errorf("%s not in instruction table", mnem)
// Verify SIMD instructions are in the arrangement table.
for _, mnem := range []string{"VADD", "VSUB", "VMUL", "VAND", "VEOR", "VORR", "VCMEQ", "VZIP1", "VZIP2"} {
if _, ok := a64SimdVTable[mnem]; !ok {
t.Errorf("%s not in the SIMD arrangement table", mnem)
}
}
}
// TestArm64CarryAndBitOps pins the carry-setting arithmetic, the widening
// multiplies and the data-processing (1 source) group against go tool asm.
func TestArm64CarryAndBitOps(t *testing.T) {
got := arm64Words(t, "\tADC R0, R2, R12\n\tADCS $0, R1\n\tSBCS R5, R9, R5\n\tSBC R25, R10, R26\n"+
"\tMUL R4, R3, R0\n\tUMULH R24, R20, R24\n\tSMULH R1, R2, R3\n\tMSUB R19, R16, R26, R2\n"+
"\tRBIT R11, R4\n\tREV R1, R2\n\tCLZ R21, R9\n\tREVW R1, R2\n\tCLSW R1, R2\n")
want := []uint32{
0x9a00004c, // ADC R12, R2, R0
0xba1f0021, // ADCS R1, R1, ZR
0xfa050125, // SBCS R5, R9, R5
0xda19015a, // SBC R26, R10, R25
0x9b047c60, // MUL R0, R3, R4
0x9bd87e98, // UMULH R24, R20, R24
0x9b417c43, // SMULH R3, R2, R1
0x9b13c342, // MSUB R2, R26, R19, R16
0xdac00164, // RBIT R4, R11
0xdac00c22, // REV R2, R1
0xdac012a9, // CLZ R9, R21
0x5ac00822, // REVW R2, R1
0x5ac01422, // CLSW R2, R1
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64BitfieldExtract pins UBFX/SBFX: immr wraps to the register
// width, an out-of-range imms is an error.
func TestArm64BitfieldExtract(t *testing.T) {
got := arm64Words(t, "\tUBFX $33, R17, $25, R5\n\tUBFXW $4, R1, $9, R2\n")
want := []uint32{
0xd361e625, // UBFX immr=1 (33 wrapped), imms=25
0x53043022, // UBFXW immr=4, imms=9
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
for _, body := range []string{"\tUBFX $33, R17, $70, R5\n", "\tUBFX $-1, R17, $3, R5\n"} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
}
// TestArm64CondCompare pins CCMP/CCMN.
func TestArm64CondCompare(t *testing.T) {
got := arm64Words(t, "\tCCMP LE, R7, $19, $3\n\tCCMP LT, R30, R6, $7\n\tCCMN EQ, R1, R2, $3\n\tCCMPW LE, R7, $19, $3\n")
want := []uint32{
0xfa53d8e3, // CCMP imm form
0xfa46b3c7, // CCMP register form
0xba420023, // CCMN register form
0x7a53d8e3, // CCMPW
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64CompareBranch pins CBZ/CBNZ/TBZ/TBNZ against a label five and
// six words ahead, matching go tool asm's own offsets.
func TestArm64CompareBranch(t *testing.T) {
// Layout: CBZ(0) TBZ(4) TBNZ(8) CBNZ(12) NOP(16) NOP(17th word...) done.
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
"\tCBZ R1, done\n\tTBZ $4, R7, done\n\tTBNZ $33, R7, done\n\tCBNZW R2, done\n" +
"\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
// done sits at word 6 from each branch's own pc: CBZ rel 6, TBZ rel 5,
// TBNZ rel 4, CBNZW rel 3.
want := []uint32{
0xb40000c1, // CBZ R1, +6
0x362000a7, // TBZ $4, R7, +5
0xb7080087, // TBNZ $33, R7, +4
0x35000062, // CBNZW R2, +3
0xd503201f, 0xd503201f, 0xd503201f,
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64ADR pins ADR against a forward label.
func TestArm64ADR(t *testing.T) {
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
"\tADR done, R10\n\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
// rel = 12 bytes: immlo 0, immhi 3.
want := []uint32{0x1000006a, 0xd503201f, 0xd503201f, 0xd503201f, 0xd65f03c0}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64PairLoadStore pins LDP/STP/LDPW/FLDPD/FSTPD.
func TestArm64PairLoadStore(t *testing.T) {
got := arm64Words(t, "\tSTP (R2, R3), 8(R5)\n\tLDP -8(R5), (R2, R3)\n\tLDPW 4(R0), (R1, R2)\n\tSTPW (R1, R2), 4(R0)\n"+
"\tFLDPD 8(R0), (F1, F2)\n\tFSTPD (F3, F4), -8(R5)\n")
want := []uint32{
0xa9008ca2, // STP (R2, R3), 8(R5)
0xa97f8ca2, // LDP -8(R5), (R2, R3)
0x29408801, // LDPW 4(R0), (R1, R2)
0x29008801, // STPW (R1, R2), 4(R0)
0x6d408801, // FLDPD 8(R0), (F1, F2)
0x6d3f90a3, // FSTPD (F3, F4), -8(R5)
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64AcquireRelease pins LDAR/STLR and the acquire/release LSE
// families.
func TestArm64AcquireRelease(t *testing.T) {
got := arm64Words(t, "\tLDAR (R27), R22\n\tLDARB (R25), R2\n\tLDARW (R12), R29\n\tSTLR R3, (R24)\n\tSTLRB R11, (R22)\n"+
"\tCASALD R5, (R6), R7\n\tLDADDALD R5, (R6), R7\n\tLDCLRALB R5, (R6), R7\n\tLDORALD R5, (RSP), R7\n\tSWPALW R5, (R6), R7\n")
want := []uint32{
0xc8dfff76, // LDAR R22, (R27)
0x08dfff22, // LDARB R2, (R25)
0x88dffd9d, // LDARW R29, (R12)
0xc89fff03, // STLR R3, (R24)
0x089ffecb, // STLRB R11, (R22)
0xc8e5fcc7, // CASALD R7, (R6), R5
0xf8e500c7, // LDADDALD R7, (R6), R5
0x38e510c7, // LDCLRALB R7, (R6), R5
0xf8e533e7, // LDORALD R7, (RSP), R5
0xb8e580c7, // SWPALW R7, (R6), R5
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64System pins BRK, SVC, the barriers, cache maintenance and the
// system register accesses.
func TestArm64System(t *testing.T) {
got := arm64Words(t, "\tBRK $35943\n\tBRK\n\tSVC $7165\n\tDMB $1\n\tDSB $1\n\tISB $15\n"+
"\tDC ZVA, R4\n\tDC IVAC, R1\n\tMRS DCZID_EL0, R3\n\tMRS CNTVCT_EL0, R0\n\tMSR $9, DAIFSet\n\tMSR $3, SPSel\n"+
"\tPRFM (R0), PLDL1KEEP\n\tPRFM (R3), PLDL3KEEP\n\tPRFM (R2), $25\n")
want := []uint32{
0xd4318ce0, // BRK $35943
0xd4200000, // BRK
0xd4037fa1, // SVC $7165
0xd50331bf, // DMB $1
0xd503319f, // DSB $1
0xd5033fdf, // ISB $15
0xd50b7424, // DC ZVA, R4
0xd5087621, // DC IVAC, R1
0xd53b00e3, // MRS DCZID_EL0, R3
0xd53be040, // MRS CNTVCT_EL0, R0
0xd50349df, // MSR $9, DAIFSet
0xd50043bf, // MSR $3, SPSel
0xf9800000, // PRFM (R0), PLDL1KEEP
0xf9800064, // PRFM (R3), PLDL3KEEP
0xf9800059, // PRFM (R2), $25
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64Crypto pins the AES and SHA families.
func TestArm64Crypto(t *testing.T) {
got := arm64Words(t, "\tAESE V31.B16, V29.B16\n\tAESD V22.B16, V19.B16\n\tAESIMC V12.B16, V27.B16\n\tAESMC V14.B16, V28.B16\n"+
"\tSHA1C V8.S4, V8, V2\n\tSHA1H V17, V25\n\tSHA1P V3.S4, V20, V27\n\tSHA1SU0 V17.S4, V13.S4, V16.S4\n\tSHA1SU1 V24.S4, V23.S4\n"+
"\tSHA256H V4.S4, V2, V11\n\tSHA256H2 V6.S4, V16, V11\n\tSHA256SU0 V0.S4, V16.S4\n\tSHA256SU1 V31.S4, V3.S4, V15.S4\n"+
"\tSHA512H V2.D2, V1, V0\n\tSHA512H2 V4.D2, V3, V2\n\tSHA512SU0 V9.D2, V8.D2\n\tSHA512SU1 V7.D2, V6.D2, V5.D2\n")
want := []uint32{
0x4e284bfd, // AESE
0x4e285ad3, // AESD
0x4e28799b, // AESIMC
0x4e2869dc, // AESMC
0x5e080102, // SHA1C
0x5e280a39, // SHA1H
0x5e03129b, // SHA1P
0x5e1131b0, // SHA1SU0
0x5e281b17, // SHA1SU1
0x5e04404b, // SHA256H
0x5e06520b, // SHA256H2
0x5e282810, // SHA256SU0
0x5e1f606f, // SHA256SU1
0xce628020, // SHA512H
0xce648462, // SHA512H2
0xcec08128, // SHA512SU0
0xce6788c5, // SHA512SU1
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLogical pins the arrangement-aware three- and two-register
// SIMD paths.
func TestArm64SIMDLogical(t *testing.T) {
got := arm64Words(t, "\tVADD V1.B16, V2.B16, V3.B16\n\tVAND V4.B16, V4.B16, V9.B16\n\tVEOR V0.B16, V1.B16, V0.B16\n"+
"\tVORR V5.B16, V4.B16, V3.B16\n\tVADDP V1.H8, V2.H8, V3.H8\n\tVZIP1 V16.H8, V3.H8, V19.H8\n\tVZIP2 V22.D2, V25.D2, V21.D2\n"+
"\tVCMEQ V24.S4, V13.S4, V12.S4\n\tVCMEQ $0, V2.H4, V3.H4\n\tVREV32 V2.H8, V1.H8\n\tVREV64 V2.S4, V3.S4\n\tVUADDLV V31.S4, V11\n"+
"\tVPMULL V2.D1, V1.D1, V3.Q1\n\tVPMULL2 V2.B16, V1.B16, V4.H8\n\tVRAX1 V26.D2, V29.D2, V30.D2\n\tVMOV V2.B16, V4.B16\n")
want := []uint32{
0x4e218443, // VADD 16B
0x4e241c89, // VAND
0x6e201c20, // VEOR
0x4ea51c83, // VORR
0x4e61bc43, // VADDP 8H
0x4e503873, // VZIP1 8H
0x4ed67b35, // VZIP2 2D
0x6eb88dac, // VCMEQ 4S
0x0e609843, // VCMEQ $0, 4H
0x6e600841, // VREV32 8H
0x4ea00843, // VREV64 4S
0x6eb03beb, // VUADDLV 4S
0x0ee2e023, // VPMULL D1
0x4e22e024, // VPMULL2 16B
0xce7a8fbe, // VRAX1 2D
0x4ea21c44, // VMOV 16B pair
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDWide pins the four-register crypto group, VXAR, VEXT and the
// shift-by-immediate encodings.
func TestArm64SIMDWide(t *testing.T) {
got := arm64Words(t, "\tVEOR3 V2.B16, V7.B16, V12.B16, V25.B16\n\tVBCAX V1.B16, V2.B16, V26.B16, V31.B16\n"+
"\tVXAR $63, V27.D2, V21.D2, V26.D2\n\tVEXT $4, V2.B8, V1.B8, V3.B8\n\tVEXT $8, V2.B16, V1.B16, V3.B16\n"+
"\tVSHL $7, V22.D2, V25.D2\n\tVUSHR $6, V22.H8, V23.H8\n\tVSRI $24, V1.S4, V2.S4\n")
want := []uint32{
0xce070999, // VEOR3
0xce22075f, // VBCAX
0xce9bfeba, // VXAR
0x2e022023, // VEXT B8
0x6e024023, // VEXT B16
0x4f4756d9, // VSHL D2 $7
0x6f1a06d7, // VUSHR H8 $6
0x6f284422, // VSRI S4 $24
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDElement pins VDUP and the VMOV element forms.
func TestArm64SIMDElement(t *testing.T) {
got := arm64Words(t, "\tVDUP V31.B[15], V18\n\tVDUP V19.S[3], V18.S4\n\tVDUP V1.D[1], V2.D2\n"+
"\tVMOV V13.S[0], R20\n\tVMOV V11.B[11], V16.B[12]\n\tVMOV R20, V21.B[2]\n")
want := []uint32{
0x5e1f07f2, // VDUP element to register
0x4e1c0672, // VDUP element across S4
0x4e180422, // VDUP element across D2
0x0e043db4, // VMOV element to register
0x6e195d70, // VMOV element to element
0x4e051e95, // VMOV register into element
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLoadStore pins the structure loads and stores.
func TestArm64SIMDLoadStore(t *testing.T) {
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
"\tVLD1.P 32(R1), [V2.B16, V3.B16]\n\tVST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)\n\tVST1.P [V2.B16], (R1)\n"+
"\tVLD1R (R1), [V9.B8]\n\tVLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]\n")
want := []uint32{
0x4c407055, // VLD1 one register
0x4c40a022, // VLD1 two registers
0x0c402fae, // VLD1 four registers D1
0x4cdfa022, // VLD1.P two registers
0x4c0029c2, // VST1 four registers S4
0x4c9f7022, // VST1.P one register
0x0d40c029, // VLD1R
0x0d60e000, // VLD4R
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64MoviLiteral pins the VMOVS/VMOVD/VMOVQ constant loads: three
// words each (ADRP, ADD, wide load) plus the pooled literal in the data
// section.
func TestArm64MoviLiteral(t *testing.T) {
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
"\tVMOVS $0x80402010, V11\n\tVMOVD $0x8040201008040201, V20\n" +
"\tVMOVQ $0x7040201008040201, $0x8040201008040201, V10\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 12*3+4 {
t.Errorf("func size = %d, want %d", img.Funcs[0].Size, 12*3+4)
}
want := []uint32{
0x9000001b, 0x9100037b, 0xbd40036b, // VMOVS: ADRP, ADD, LDR S
0x9000001b, 0x9100037b, 0xfd400374, // VMOVD: ADRP, ADD, LDR D
0x9000001b, 0x9100037b, 0x3dc0036a, // VMOVQ: ADRP, ADD, LDR Q
0xd65f03c0,
}
got := leWords(img.Code)
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
// The literals sit in the data section.
var found32, found64, found128 bool
for _, d := range img.DataSyms {
switch d.Name {
case "$i32.80402010":
found32 = d.Size == 4
case "$i64.8040201008040201":
found64 = d.Size == 8
case "$i128.80402010080402017040201008040201":
found128 = d.Size == 16
}
}
if !found32 || !found64 || !found128 {
t.Errorf("literals missing: i32=%v i64=%v i128=%v", found32, found64, found128)
}
}
// TestArm64MOVK pins standalone MOVK with the hw field derived from the
// chunk position.
func TestArm64MOVK(t *testing.T) {
got := arm64Words(t, "\tMOVK $1234, R5\n\tMOVK $305397760, R5\n\tMOVKW $1234, R5\n")
want := []uint32{
0xf2809a45, // MOVK hw=0
0xf2a24685, // MOVK hw=1
0x72809a45, // MOVKW hw=0
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
+72
View File
@@ -0,0 +1,72 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 synchronisation instructions: the
// acquire/release loads and stores, the exclusive family and the LSE
// atomics with acquire and release semantics, plus the register-pair
// loads and stores. Every function is byte-compared against go tool asm.
#include "textflag.h"
// func acquireRelease()
TEXT ·acquireRelease(SB), NOSPLIT, $0-0
LDAR (R1), R2
LDARB (R3), R4
LDARH (R5), R6
LDARW (R7), R8
STLR R2, (R1)
STLRB R4, (R3)
STLRH R6, (R5)
STLRW R8, (R7)
RET
// func exclusive()
TEXT ·exclusive(SB), NOSPLIT, $0-0
LDAXR (R1), R2
LDAXRB (R3), R4
LDAXRW (R5), R6
STLXR R2, (R1), R8
STLXRB R4, (R3), R8
STLXRW R6, (R5), R8
RET
// func lseAcquireRelease()
TEXT ·lseAcquireRelease(SB), NOSPLIT, $0-0
CASALD R1, (R3), R2
CASALW R4, (R6), R5
LDADDALD R1, (R3), R2
LDADDALW R4, (R6), R5
LDCLRALB R1, (R3), R2
LDCLRALW R4, (R6), R5
LDCLRALD R1, (R3), R2
LDORALB R1, (R3), R2
LDORALW R4, (R6), R5
LDORALD R1, (R3), R2
SWPALB R1, (R3), R2
SWPALW R4, (R6), R5
SWPALD R1, (R3), R2
RET
// func lseBase()
TEXT ·lseBase(SB), NOSPLIT, $0-0
LDADDD R1, (R3), R2
LDADDW R4, (R6), R5
CASD R1, (R3), R2
CASW R4, (R6), R5
SWPD R1, (R3), R2
SWPW R4, (R6), R5
RET
// func pairs()
TEXT ·pairs(SB), NOSPLIT, $0-0
LDP (R1), (R2, R3)
LDP 8(R4), (R5, R6)
LDP -16(R1), (R2, R3)
LDPW 4(R4), (R5, R6)
STP (R2, R3), 24(R7)
STP (R2, R3),-8(R7)
STPW (R1, R2), 4(R0)
FLDPD (R8), (F1, F2)
FLDPD 8(R8), (F3, F4)
FSTPD (F3, F4),-8(R9)
RET
+42
View File
@@ -0,0 +1,42 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 cryptographic extension: the AES round
// instructions and the SHA1, SHA256 and SHA512 families. Every function is
// byte-compared against go tool asm.
#include "textflag.h"
// func aesRound()
TEXT ·aesRound(SB), NOSPLIT, $0-0
AESE V31.B16, V29.B16
AESD V22.B16, V19.B16
AESMC V14.B16, V28.B16
AESIMC V12.B16, V27.B16
RET
// func sha1Round()
TEXT ·sha1Round(SB), NOSPLIT, $0-0
SHA1C V8.S4, V8, V2
SHA1P V3.S4, V20, V27
SHA1M V0.S4, V27, V27
SHA1H V17, V25
SHA1SU0 V17.S4, V13.S4, V16.S4
SHA1SU1 V24.S4, V23.S4
RET
// func sha256Round()
TEXT ·sha256Round(SB), NOSPLIT, $0-0
SHA256H V4.S4, V2, V11
SHA256H2 V6.S4, V16, V11
SHA256SU0 V0.S4, V16.S4
SHA256SU1 V31.S4, V3.S4, V15.S4
RET
// func sha512Round()
TEXT ·sha512Round(SB), NOSPLIT, $0-0
SHA512H V2.D2, V1, V0
SHA512H2 V4.D2, V3, V2
SHA512SU0 V9.D2, V8.D2
SHA512SU1 V7.D2, V6.D2, V5.D2
RET
+66
View File
@@ -0,0 +1,66 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 integer slice: carry-setting arithmetic,
// widening multiplies, bit manipulation, conditional compares, the compare
// and test branches, ADR and the wide-constant moves. Every function is
// byte-compared against go tool asm.
#include "textflag.h"
// func carryArith()
TEXT ·carryArith(SB), NOSPLIT, $0-0
ADC R0, R2, R12
ADCS R23, R22, R22
ADC $0, R1
SBC R25, R10, R26
SBCS R5, R9, R5
SBCS $0, R1
RET
// func wideningMul()
TEXT ·wideningMul(SB), NOSPLIT, $0-0
MUL R4, R3, R0
MSUB R19, R16, R26, R2
SMULH R24, R20, R24
UMULH R24, R20, R24
RET
// func bitManip()
TEXT ·bitManip(SB), NOSPLIT, $0-0
RBIT R11, R4
REV R1, R2
CLZ R21, R9
REVW R1, R2
CLSW R1, R2
UBFX $33, R17, $25, R5
UBFXW $4, R1, $9, R2
RET
// func condCompare()
TEXT ·condCompare(SB), NOSPLIT, $0-0
CCMP LE, R7, $19, $3
CCMP LT, R30, R6, $7
CCMN EQ, R1, R2, $3
CCMPW LE, R7, $19, $3
RET
// func branchForms()
TEXT ·branchForms(SB), NOSPLIT, $0-0
CBZ R1, target
CBNZ R7, target
CBNZW R2, target
TBZ $4, R7, target
TBNZ $33, R7, target
ADR target, R10
target:
RET
// func wideMoves()
TEXT ·wideMoves(SB), NOSPLIT, $0-0
MOVK $1234, R5
MOVK $305397760, R5
MOVKW $1234, R5
MOVK $16771847290880, R21
RET
+98
View File
@@ -0,0 +1,98 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 NEON slice: the logical and arithmetic
// three-register operations, permutations, comparisons, shifts, the crypto
// four-register group, element moves, table lookups and the structure
// loads and stores. Every function is byte-compared against go tool asm.
#include "textflag.h"
// func simdLogic()
TEXT ·simdLogic(SB), NOSPLIT, $0-0
VADD V1.B16, V2.B16, V3.B16
VADD V1.B8, V2.B8, V3.B8
VSUB V1.S4, V2.S4, V3.S4
VMUL V1.H8, V2.H8, V3.H8
VAND V4.B16, V4.B16, V9.B16
VORR V5.B16, V4.B16, V3.B16
VEOR V0.B16, V1.B16, V0.B16
VADDP V1.H8, V2.H8, V3.H8
VCMEQ V24.S4, V13.S4, V12.S4
VCMEQ $0, V2.H4, V3.H4
RET
// func simdPerm()
TEXT ·simdPerm(SB), NOSPLIT, $0-0
VZIP1 V16.H8, V3.H8, V19.H8
VZIP1 V6.D2, V9.D2, V11.D2
VZIP2 V22.D2, V25.D2, V21.D2
VREV32 V2.H8, V1.H8
VREV64 V2.S4, V3.S4
VUADDLV V31.S4, V11
VEXT $4, V2.B8, V1.B8, V3.B8
VEXT $8, V2.B16, V1.B16, V3.B16
RET
// func simdShift()
TEXT ·simdShift(SB), NOSPLIT, $0-0
VSHL $7, V22.D2, V25.D2
VSHL $24, V1.S4, V2.S4
VUSHR $6, V22.H8, V23.H8
VUSHR $56, V1.D2, V2.D2
VSRI $24, V1.S4, V2.S4
VSRI $56, V1.D2, V2.D2
RET
// func simdCrypto4()
TEXT ·simdCrypto4(SB), NOSPLIT, $0-0
VEOR3 V2.B16, V7.B16, V12.B16, V25.B16
VBCAX V1.B16, V2.B16, V26.B16, V31.B16
VXAR $63, V27.D2, V21.D2, V26.D2
VRAX1 V26.D2, V29.D2, V30.D2
VPMULL V2.D1, V1.D1, V3.Q1
VPMULL V2.B8, V1.B8, V3.H8
VPMULL2 V2.D2, V1.D2, V4.Q1
VPMULL2 V2.B16, V1.B16, V4.H8
RET
// func simdElement()
TEXT ·simdElement(SB), NOSPLIT, $0-0
VDUP V31.B[15], V18
VDUP V19.S[3], V18.S4
VDUP V1.D[1], V2.D2
VMOV V13.S[0], R20
VMOV V11.B[11], V16.B[12]
VMOV R20, V21.B[2]
VMOV V2.B16, V4.B16
RET
// func simdTable()
TEXT ·simdTable(SB), NOSPLIT, $0-0
VTBL V22.B16, [V28.B16], V11.B16
VTBL V18.B8, [V17.B16, V18.B16], V22.B8
VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8
RET
// func simdLoadStore()
TEXT ·simdLoadStore(SB), NOSPLIT, $0-0
VLD1 (R2), [V21.B16]
VLD1 (R24), [V18.D1, V19.D1, V20.D1]
VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]
VLD1.P 32(R1), [V2.B16, V3.B16]
VLD1.P 64(R4), [V5.B16, V6.B16, V7.B16, V8.B16]
VLD1R (R1), [V9.B8]
VLD1R (R0), [V0.B16]
VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]
VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)
VST1 [V14.H4, V15.H4, V16.H4], (R27)
VST1.P [V2.B16], (R1)
VST1.P [V2.B16, V3.B16], 32(R1)
RET
// func simdLiteral()
TEXT ·simdLiteral(SB), NOSPLIT, $0-0
VMOVS $0x80402010, V11
VMOVD $0x8040201008040201, V20
VMOVQ $0x7040201008040201, $0x8040201008040201, V10
RET
+61
View File
@@ -0,0 +1,61 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 system instructions: barriers,
// cache maintenance, the system register accesses, supervisor calls,
// breakpoints and prefetches. Every function is byte-compared against
// go tool asm.
#include "textflag.h"
// func barriers()
TEXT ·barriers(SB), NOSPLIT, $0-0
DMB $15
DMB $1
DSB $15
DSB $4
ISB $15
ISB $1
RET
// func cacheOps()
TEXT ·cacheOps(SB), NOSPLIT, $0-0
DC ZVA, R4
DC IVAC, R1
DC CVAC, R2
DC CVAU, R3
DC CIVAC, R7
RET
// func sysRegs()
TEXT ·sysRegs(SB), NOSPLIT, $0-0
MRS DCZID_EL0, R3
MRS CNTVCT_EL0, R0
MRS CNTPCT_EL0, R1
MRS CNTFRQ_EL0, R2
MRS MIDR_EL1, R0
MRS ID_AA64PFR0_EL1, R0
MRS ID_AA64ISAR0_EL1, R0
MRS ID_AA64ISAR1_EL1, R0
MRS DIT, R0
MSR $3, SPSel
MSR $9, DAIFSet
MSR $6, DAIFClr
MSR $1, DIT
RET
// func exceptions()
TEXT ·exceptions(SB), NOSPLIT, $0-0
SVC $0
SVC $7165
BRK
BRK $35943
RET
// func prefetch()
TEXT ·prefetch(SB), NOSPLIT, $0-0
PRFM (R0), PLDL1KEEP
PRFM (R3), PLDL3KEEP
PRFM (R4), PSTL1KEEP
PRFM (R2), $25
RET