feat(asm): encode the arm64 system registers and structure loads

This commit is contained in:
2026-10-02 20:39:26 +02:00
parent e02918c17b
commit 2747fce7d3
4 changed files with 1375 additions and 79 deletions
+146 -14
View File
@@ -374,6 +374,7 @@ const (
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCASP // compare and swap pair: CASP
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
@@ -384,6 +385,7 @@ const (
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FVMoviImm // SIMD move immediate: VMOVI $imm8, Vd.B8/B16
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
@@ -770,7 +772,7 @@ func init() {
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM", "RPRFM", "SYS", "SYSL", "TLBI", "SB", "PACIASP", "PACIBSP"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
@@ -783,12 +785,20 @@ func init() {
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
// ---- load/store pair (signed offset) ----
// The scale column of a64LoadTable does not reach the pair forms, so each
// entry states its own access width through the imm7 divisor the pair
// encoder derives from the opc field (8 for D, 4 for W and SW, 16 for Q).
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
a64InstrTable["LDPSW"] = a64Enc{format: a64FPair, op: 0x69400000}
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
a64InstrTable["FLDPS"] = a64Enc{format: a64FPair, op: 0x2d400000}
a64InstrTable["FSTPS"] = a64Enc{format: a64FPair, op: 0x2d000000}
a64InstrTable["FLDPQ"] = a64Enc{format: a64FPair, op: 0xad400000}
a64InstrTable["FSTPQ"] = a64Enc{format: a64FPair, op: 0xad000000}
// ---- acquire/release loads and stores ----
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
@@ -806,11 +816,23 @@ func init() {
lse := map[string]uint32{
"CASALD": 0xc8e0fc00,
"CASALW": 0x88e0fc00,
"CASB": 0x08a07c00,
"CASAB": 0x08e07c00,
"CASH": 0x48a07c00,
"CASLD": 0xc8a0fc00,
"CASLH": 0x48a0fc00,
"CASAW": 0x88e07c00,
"CASAD": 0xc8e07c00,
"CASALH": 0x48e07c00,
"LDADDALD": 0xf8e00000,
"LDADDALW": 0xb8e00000,
"LDADDAD": 0xf8a00000,
"LDADDAW": 0xb8a00000,
"LDCLRALB": 0x38e01000,
"LDCLRALW": 0xb8e01000,
"LDCLRALD": 0xf8e01000,
"LDCLRAD": 0xf8a01000,
"LDCLRAW": 0xb8a01000,
"LDORALB": 0x38e03000,
"LDORALW": 0xb8e03000,
"LDORALD": 0xf8e03000,
@@ -889,6 +911,12 @@ func init() {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// Compare and swap pair: the second register of each pair is implicit
// (Rs+1 and Rt+1), so the encoding carries Rs and Rt alone over a preset
// fixed field (asm7.go atomicCASP).
a64InstrTable["CASPD"] = a64Enc{format: a64FCASP, op: 1<<30 | 0x41<<21 | 0x1f<<10}
a64InstrTable["CASPW"] = a64Enc{format: a64FCASP, op: 0x41<<21 | 0x1f<<10}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
// register preset to ZR (bits 14:10 = 11111).
@@ -940,11 +968,13 @@ func init() {
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
a64InstrTable["VMOVI"] = a64Enc{format: a64FVMoviImm}
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 7<<10}
a64InstrTable["VUSRA"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 5<<10}
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
@@ -957,6 +987,24 @@ func init() {
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
// Multi-register structure accesses beyond VLD1/VST1: VLD2/VLD3/VLD4 and
// the replicate loads VLD2R/VLD3R, each with the post-index spelling.
a64InstrTable["VLD2"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD2.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD3"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD3.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD2R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD2R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD3R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD3R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST2"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST2.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST3"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST3.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST4"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST4.P"] = a64Enc{format: a64FVLDST, op: 1}
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
@@ -1120,6 +1168,78 @@ var a64SimdVTable = map[string]a64SimdVSpec{
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
// Saturating shifts, register forms (the immediate spellings route to
// a64FShiftImm).
"VSQSHL": {0x0e204c00, 0x7f, false},
"VUQSHL": {0x2e204c00, 0x7f, false},
}
// a64SimdNLForm classifies the narrow/long/wide SIMD families whose
// arrangement does not travel on every operand: the encoding's size and Q
// bits read off one designated operand and the element widths pair up across
// the operands.
type a64SimdNLForm uint8
const (
a64NLTwoNarrow a64SimdNLForm = iota // (Vn.wide, Vd.narrow): size/Q from Vd
a64NLTwoLong // (Vn.narrow, Vd.long): size/Q from Vn
a64NLThreeLongMul // (Vm.narrow, Vn.narrow, Vd.long): size/Q from Vn
a64NLThreeWide // (Vm.narrow, Vn.wide, Vd.wide): size/Q from Vn
a64NLThreeLongShift // ($sh, Vn.narrow, Vd.long): size/Q from Vn, immh = esize+sh
a64NLThreeNarrowShift // ($sh, Vn.wide, Vd.narrow): size/Q from Vd, immh = esize-sh
)
// a64SimdNLSpec is one narrow/long/wide instruction: the base word (U, opcode
// and fixed bits positioned) and the arrangement form. qonly marks the FCVT
// family, whose size field is fixed in the base and only the Q bit follows
// the driving arrangement.
type a64SimdNLSpec struct {
base uint32
form a64SimdNLForm
qonly bool
}
// a64SimdNLTable holds the families the arrangement-driven three-register and
// two-register encoders cannot express. The .2 spellings force the 128-bit
// side of the pair through their operand arrangements, so the base carries no
// arrangement bits of its own.
var a64SimdNLTable = map[string]a64SimdNLSpec{
"VSHRN": {0x0f008400, a64NLThreeNarrowShift, false},
"VSHRN2": {0x0f008400, a64NLThreeNarrowShift, false},
"VSXTL": {0x0f00a400, a64NLTwoLong, false},
"VSXTL2": {0x0f00a400, a64NLTwoLong, false},
"VUXTL": {0x2f00a400, a64NLTwoLong, false},
"VUXTL2": {0x2f00a400, a64NLTwoLong, false},
"VXTN": {0x0e202800, a64NLTwoNarrow, false},
"VXTN2": {0x0e202800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e202800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e202800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e204800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e204800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e217800, a64NLTwoLong, true},
"VFCVTL2": {0x0e217800, a64NLTwoLong, true},
"VSSHLL": {0x0f00a400, a64NLThreeLongShift, false},
"VSSHLL2": {0x0f00a400, a64NLThreeLongShift, false},
"VUSHLL": {0x2f00a400, a64NLThreeLongShift, false},
"VUSHLL2": {0x2f00a400, a64NLThreeLongShift, false},
"VUADDW": {0x2e201000, a64NLThreeWide, false},
"VUADDW2": {0x2e201000, a64NLThreeWide, false},
"VUMULL": {0x2e20c000, a64NLThreeLongMul, false},
"VUMULL2": {0x2e20c000, a64NLThreeLongMul, false},
"VSMULL": {0x0e20c000, a64NLThreeLongMul, false},
"VSMULL2": {0x0e20c000, a64NLThreeLongMul, false},
"VUMLAL": {0x2e208000, a64NLThreeLongMul, false},
"VUMLAL2": {0x2e208000, a64NLThreeLongMul, false},
"VSMLAL": {0x0e208000, a64NLThreeLongMul, false},
"VSMLAL2": {0x0e208000, a64NLThreeLongMul, false},
"VUMLSL": {0x2e20a000, a64NLThreeLongMul, false},
"VUMLSL2": {0x2e20a000, a64NLThreeLongMul, false},
"VSMLSL": {0x0e20a000, a64NLThreeLongMul, false},
"VSMLSL2": {0x0e20a000, a64NLThreeLongMul, false},
}
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
@@ -1243,6 +1363,16 @@ var a64PRFOps = map[string]int{
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
// a64VLDNBase and a64VSTNBase hold the VLD2/VLD3/VLD4 and VST2/VST3/VST4
// fixed words (indexed by register count 2..4): the opcode field at bits
// 15:12 carries the access kind.
var a64VLDNBase = [5]uint32{0, 0, 0x0c408000, 0x0c404000, 0x0c400000}
var a64VSTNBase = [5]uint32{0, 0, 0x0c008000, 0x0c004000, 0x0c000000}
// a64VLDNReplicate holds the VLD2R/VLD3R fixed words beside the existing
// VLD1R (0x0d40c000) and VLD4R (0x0d60e000) bases.
var a64VLDNReplicate = [5]uint32{0, 0x0d40c000, 0x0d60c000, 0x0d40e000, 0x0d60e000}
// a64Vec is a parsed vector operand: the register number, the arrangement
// ("" when the operand spells none) and, for element forms, the lane index.
type a64Vec struct {
@@ -1363,23 +1493,25 @@ func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool)
// a64LSType describes the load/store parameters for a MOV width mnemonic.
type a64LSType struct {
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
scale int // access width in bytes; the unsigned offset divides by it
}
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
var a64LoadTable = map[string]a64LSType{
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
"MOVD": {3, 0, 1, 8}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1, 4}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2, 4}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1, 2}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2, 1}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1, 4}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1, 8}, // LDR D (64-bit FP)
"FMOVQ": {0, 1, 3, 16}, // LDR/STR Q (128-bit FP): opc=11 selects it
}
// a64StoreOpc returns the store opc for a given load type: integer and FP