feat(arm64): encode pairs, atomics, crypto, system and NEON slices

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 06:44:51 +02:00
parent fc2d92eabd
commit ca3fdce0e0
9 changed files with 2377 additions and 30 deletions
+423 -6
View File
@@ -27,7 +27,13 @@ package asm
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
import "maps"
import (
"maps"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
@@ -288,7 +294,25 @@ const (
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
a64FCondCmp // conditional compare: CCMP, CCMN
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
a64FVTBL // SIMD table lookup: VTBL
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
// a64Enc is one instruction's encoding: its bit layout (format) and the
@@ -622,10 +646,403 @@ func init() {
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
// ---- SIMD basics ----
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
// ---- SIMD: the arrangement-aware tables in this file carry VADD,
// VSUB, VMUL and every other three-register vector op. ----
// ---- data-processing (1 source): sf 10 11010110 opcode 00000 Rn Rd ----
dp1 := map[string]uint32{
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
}
for m, op := range dp1 {
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
}
// ---- bitfield extract: the UBFM/SBFM bases, immediate operands wrap ----
a64InstrTable["UBFX"] = a64Enc{format: a64FBitfield2, op: 0xd3400000}
a64InstrTable["SBFX"] = a64Enc{format: a64FBitfield2, op: 0x93400000}
a64InstrTable["UBFXW"] = a64Enc{format: a64FBitfield2, op: 0x53000000}
a64InstrTable["SBFXW"] = a64Enc{format: a64FBitfield2, op: 0x13000000}
// ---- conditional compare: sf 1 1 101001 0 imm5/Rm cond op2 Rn nzcv ----
a64InstrTable["CCMP"] = a64Enc{format: a64FCondCmp, op: 0xfa400000}
a64InstrTable["CCMN"] = a64Enc{format: a64FCondCmp, op: 0xba400000}
a64InstrTable["CCMPW"] = a64Enc{format: a64FCondCmp, op: 0x7a400000}
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
// ---- compare/test and branch ----
a64InstrTable["CBZ"] = a64Enc{format: a64FBranch19, op: 0xb4000000}
a64InstrTable["CBZW"] = a64Enc{format: a64FBranch19, op: 0x34000000}
a64InstrTable["CBNZ"] = a64Enc{format: a64FBranch19, op: 0xb5000000}
a64InstrTable["CBNZW"] = a64Enc{format: a64FBranch19, op: 0x35000000}
a64InstrTable["TBZ"] = a64Enc{format: a64FTestBranch, op: 0x36000000}
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
// ---- load/store pair (signed offset) ----
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
// ---- acquire/release loads and stores ----
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
a64InstrTable["LDARB"] = a64Enc{format: a64FAcqRel, op: 0x08dffc00}
a64InstrTable["LDARH"] = a64Enc{format: a64FAcqRel, op: 0x48dffc00}
a64InstrTable["LDARW"] = a64Enc{format: a64FAcqRel, op: 0x88dffc00}
a64InstrTable["STLR"] = a64Enc{format: a64FAcqRel, op: 0xc89ffc00}
a64InstrTable["STLRB"] = a64Enc{format: a64FAcqRel, op: 0x089ffc00}
a64InstrTable["STLRH"] = a64Enc{format: a64FAcqRel, op: 0x489ffc00}
a64InstrTable["STLRW"] = a64Enc{format: a64FAcqRel, op: 0x889ffc00}
// ---- LSE atomics with acquire and release semantics ----
// CAS carries a preset fixed op field and a real Rs; the LDADD/LDCLR/
// LDOR/SWP families leave Rs free for the returned value.
lse := map[string]uint32{
"CASALD": 0xc8e0fc00,
"CASALW": 0x88e0fc00,
"LDADDALD": 0xf8e00000,
"LDADDALW": 0xb8e00000,
"LDCLRALB": 0x38e01000,
"LDCLRALW": 0xb8e01000,
"LDCLRALD": 0xf8e01000,
"LDORALB": 0x38e03000,
"LDORALW": 0xb8e03000,
"LDORALD": 0xf8e03000,
"SWPALB": 0x38e08000,
"SWPALW": 0xb8e08000,
"SWPALD": 0xf8e08000,
}
for m, op := range lse {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
// register preset to ZR (bits 14:10 = 11111).
dpsrExtra := map[string]uint32{
"ADC": 0x9a000000, "ADCW": 0x1a000000,
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
"SBC": 0xda000000, "SBCW": 0x5a000000,
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
}
for m, op := range dpsrExtra {
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
}
// ---- crypto, 2-register (Rn, Rd) and 3-register (Rm, Rn, Rd) forms ----
crypto2 := map[string]uint32{
"AESD": 0x4e285800, "AESE": 0x4e284800,
"AESIMC": 0x4e287800, "AESMC": 0x4e286800,
"SHA1H": 0x5e280800, "SHA1SU1": 0x5e281800,
"SHA256SU0": 0x5e282800, "SHA512SU0": 0xcec08000,
}
for m, op := range crypto2 {
a64InstrTable[m] = a64Enc{format: a64FCrypto2, op: op}
}
crypto3 := map[string]uint32{
"SHA1C": 0x5e000000, "SHA1P": 0x5e001000,
"SHA1M": 0x5e002000, "SHA1SU0": 0x5e003000,
"SHA256H": 0x5e004000, "SHA256H2": 0x5e005000,
"SHA256SU1": 0x5e006000, "SHA512H": 0xce608000,
"SHA512H2": 0xce608400, "SHA512SU1": 0xce608800,
}
for m, op := range crypto3 {
a64InstrTable[m] = a64Enc{format: a64FCrypto3, op: op}
}
// ---- arrangement-aware SIMD, see a64SimdVTable and a64SimdV2Table ----
a64InstrTable["VEOR3"] = a64Enc{format: a64FSIMDV4, op: 0xce000000}
a64InstrTable["VBCAX"] = a64Enc{format: a64FSIMDV4, op: 0xce200000}
a64InstrTable["VXAR"] = a64Enc{format: a64FSIMDV4, op: 0xce800000}
a64InstrTable["VEXT"] = a64Enc{format: a64FSIMDV4, op: 0x2e000000}
a64InstrTable["VTBL"] = a64Enc{format: a64FVTBL}
a64InstrTable["VDUP"] = a64Enc{format: a64FDUP}
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
// for instructions that exist at a single arrangement and carry that
// arrangement's bits inside the base already, the fixed flag.
type a64SimdVSpec struct {
base uint32
arrs uint16
fixed bool
}
// a64Arr names the vector arrangements the encoders deal with, indexed by
// a64Arr. The source spellings put the element letter first: B8, H4, S2,
// D1 and the 128-bit halves B16, H8, S4, D2.
const (
a64Arr8B = iota
a64Arr16B
a64Arr4H
a64Arr8H
a64Arr2S
a64Arr4S
a64Arr2D
a64ArrD1
a64ArrQ1
a64ArrCount
)
// a64ArrNames maps an arrangement to its source spelling (element letter
// first, as the toolchain writes it).
var a64ArrNames = [a64ArrCount]string{
a64Arr8B: "B8", a64Arr16B: "B16", a64Arr4H: "H4", a64Arr8H: "H8",
a64Arr2S: "S2", a64Arr4S: "S4", a64Arr2D: "D2", a64ArrD1: "D1", a64ArrQ1: "Q1",
}
// a64ArrIndex resolves a source spelling to its a64Arr index, -1 when
// unknown.
func a64ArrIndex(s string) int {
for i, n := range a64ArrNames {
if n == s {
return i
}
}
return -1
}
// a64ElemLetter reports whether s is a bare element spelling (B, H, S, D, Q)
// as it appears in element operands such as V13.S[0].
func a64ElemLetter(s string) bool {
switch s {
case "B", "H", "S", "D", "Q":
return true
}
return false
}
// a64ArrBits carries the fixed bits an arrangement contributes to the
// three-same word shape: the element size at bits 23:22 and the 128-bit
// flag at bit 30. Bit 29 belongs to the instruction's own base.
var a64ArrBits = [a64ArrCount]uint32{
a64Arr8B: 0,
a64Arr16B: 1 << 30,
a64Arr4H: 1 << 22,
a64Arr8H: 1<<30 | 1<<22,
a64Arr2S: 1 << 23,
a64Arr4S: 1<<30 | 1<<23,
a64Arr2D: 1<<30 | 1<<23 | 1<<22,
a64ArrD1: 1<<23 | 1<<22,
a64ArrQ1: 0,
}
// a64SimdVTable holds the arrangement-aware three-register SIMD
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
// word and arrangement bit was read off go tool asm.
var a64SimdVTable = map[string]a64SimdVSpec{
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
}
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
// the register pair spelling ORR Vd, Vn, Vm.
var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false},
"VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
}
// a64CryptoArr is the arrangement each crypto instruction's operands must
// carry when they spell one at all; a bare V/F spelling is accepted as is.
var a64CryptoArr = map[string]int{
"AESD": a64Arr16B, "AESE": a64Arr16B, "AESIMC": a64Arr16B, "AESMC": a64Arr16B,
"SHA1H": a64Arr4S, "SHA1SU1": a64Arr4S, "SHA256SU0": a64Arr4S, "SHA512SU0": a64Arr2D,
"SHA1C": a64Arr4S, "SHA1P": a64Arr4S, "SHA1M": a64Arr4S, "SHA1SU0": a64Arr4S,
"SHA256H": a64Arr4S, "SHA256H2": a64Arr4S, "SHA256SU1": a64Arr4S,
"SHA512H": a64Arr2D, "SHA512H2": a64Arr2D, "SHA512SU1": a64Arr2D,
}
// a64DCOps maps the data-cache maintenance operation names to their fixed
// word (the register rides bits 4:0).
var a64DCOps = map[string]uint32{
"IVAC": 0xd5087620, "ZVA": 0xd50b7420,
"CVAC": 0xd50b7a20, "CVAU": 0xd50b7b20, "CIVAC": 0xd50b7e20,
}
// a64MRSOps maps the system register names GOROOT reads to their fixed word
// (the destination register rides bits 4:0).
var a64MRSOps = map[string]uint32{
"ELR_EL1": 0xd5384020, "MIDR_EL1": 0xd5380000,
"ID_AA64PFR0_EL1": 0xd5380400, "ID_AA64ISAR0_EL1": 0xd5380600,
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
}
// a64MSROps maps the system register names GOROOT writes to their fixed
// word; the immediate rides CRm at bits 11:8 and Rt is the fixed 11111.
var a64MSROps = map[string]uint32{
"SPSel": 0xd50040a0, "DAIFSet": 0xd50340c0, "DAIFClr": 0xd50340e0, "DIT": 0xd5034040,
}
// a64PRFOps maps the prefetch operation names to their prfop immediate
// (word = 0xf9800000 | Rn<<5 | prfop).
var a64PRFOps = map[string]int{
"PLDL1KEEP": 0x00, "PLDL1STRM": 0x01, "PLDL2KEEP": 0x02, "PLDL2STRM": 0x03,
"PLDL3KEEP": 0x04, "PLDL3STRM": 0x05,
"PLIL1KEEP": 0x08, "PLIL1STRM": 0x09, "PLIL2KEEP": 0x0a, "PLIL2STRM": 0x0b,
"PLIL3KEEP": 0x0c, "PLIL3STRM": 0x0d,
"PSTL1KEEP": 0x10, "PSTL1STRM": 0x11, "PSTL2KEEP": 0x12, "PSTL2STRM": 0x13,
"PSTL3KEEP": 0x14, "PSTL3STRM": 0x15,
}
// a64VLD1Base holds the fixed words of the multi-register structure
// accesses, indexed by register count 1..4, before the Q and size bits.
// Post-index spellings add 0x9f0000 (post bit and Rm = 11111).
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
// a64Vec is a parsed vector operand: the register number, the arrangement
// ("" when the operand spells none) and, for element forms, the lane index.
type a64Vec struct {
reg int
arr string
idx int
hasIdx bool
}
// a64VecReg parses a vector register operand: V0..V31 (F0..F31 as an alias,
// the same architectural registers the scalar floating-point spellings use),
// optionally with an arrangement suffix such as V0.B16 and, for element
// forms, a lane index such as V13.S[0]. It reports ok=false for anything
// else, including X/W and R spellings, which the toolchain's vector
// operands reject as well.
func a64VecReg(name string) (v a64Vec, ok bool) {
s := strings.TrimSpace(name)
if i := strings.IndexByte(s, '.'); i >= 0 {
v.arr = strings.TrimSpace(s[i+1:])
s = s[:i]
}
if v.arr != "" {
// Element form: B[3], S[2] and friends.
if j := strings.IndexByte(v.arr, '['); j >= 0 {
k := strings.LastIndexByte(v.arr, ']')
if k < j {
return v, false
}
n, err := strconv.Atoi(strings.TrimSpace(v.arr[j+1 : k]))
if err != nil || n < 0 {
return v, false
}
v.idx, v.hasIdx = n, true
v.arr = strings.TrimSpace(v.arr[:j])
}
if a64ArrIndex(v.arr) < 0 && !a64ElemLetter(v.arr) {
return v, false
}
}
if len(s) < 2 || (s[0] != 'V' && s[0] != 'F') {
return v, false
}
n := 0
for i := 1; i < len(s); i++ {
if s[i] < '0' || s[i] > '9' {
return v, false
}
n = n*10 + int(s[i]-'0')
}
if n > 31 {
return v, false
}
v.reg = n
return v, true
}
// a64ElemField encodes a lane index for the copy/insert group: imm5 = the
// index shifted by the element scale, with the scale's own bit set. B gets
// shift 1 (the Q bit rides elsewhere), H shift 2, S shift 3 and D shift 4.
func a64ElemField(arr string, idx int) (uint32, bool) {
var shift, low uint32
switch arr {
case "B8", "B16", "B":
shift, low = 1, 1
case "H4", "H8", "H":
shift, low = 2, 2
case "S2", "S4", "S":
shift, low = 3, 4
case "D1", "D2", "D":
shift, low = 4, 8
default:
return 0, false
}
if idx < 0 || idx >= 1<<(5-shift) {
return 0, false
}
return uint32(idx)<<shift | low, true
}
// a64VecListOf recovers the register list of a VLD1/VST1/VTBL operand run.
// The parser keeps parenthesised groups whole but splits bracketed lists on
// the commas, so a list arrives as one operand run whose first Raw starts
// with "[" and whose last Raw ends with "]". It returns the parsed
// registers with the brackets and spaces removed.
func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool) {
if start >= len(ops) || !strings.HasPrefix(strings.TrimSpace(ops[start].Raw), "[") {
return nil, 0, false
}
end = start
for end < len(ops) {
if strings.HasSuffix(strings.TrimSpace(ops[end].Raw), "]") {
break
}
end++
}
if end >= len(ops) {
return nil, 0, false
}
for i := start; i <= end; i++ {
s := strings.TrimSpace(ops[i].Raw)
s = strings.TrimPrefix(s, "[")
s = strings.TrimSuffix(s, "]")
if s == "" && len(ops) > start+1 {
return nil, 0, false
}
for part := range strings.SplitSeq(s, ",") {
v, ok := a64VecReg(part)
if !ok {
return nil, 0, false
}
vs = append(vs, v)
}
}
return vs, end, true
}
// ---- load/store helper tables ----