feat(asm): encode the arm64 system registers and structure loads
This commit is contained in:
+146
-14
@@ -374,6 +374,7 @@ const (
|
||||
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
|
||||
a64FAcqRel // acquire/release: LDAR family, STLR family
|
||||
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
|
||||
a64FCASP // compare and swap pair: CASP
|
||||
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
|
||||
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
|
||||
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
|
||||
@@ -384,6 +385,7 @@ const (
|
||||
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
|
||||
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
|
||||
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
|
||||
a64FVMoviImm // SIMD move immediate: VMOVI $imm8, Vd.B8/B16
|
||||
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
|
||||
)
|
||||
|
||||
@@ -770,7 +772,7 @@ func init() {
|
||||
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
|
||||
|
||||
// ---- system operations ----
|
||||
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
|
||||
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM", "RPRFM", "SYS", "SYSL", "TLBI", "SB", "PACIASP", "PACIBSP"} {
|
||||
a64InstrTable[m] = a64Enc{format: a64FSys}
|
||||
}
|
||||
|
||||
@@ -783,12 +785,20 @@ func init() {
|
||||
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
|
||||
|
||||
// ---- load/store pair (signed offset) ----
|
||||
// The scale column of a64LoadTable does not reach the pair forms, so each
|
||||
// entry states its own access width through the imm7 divisor the pair
|
||||
// encoder derives from the opc field (8 for D, 4 for W and SW, 16 for Q).
|
||||
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
|
||||
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
|
||||
a64InstrTable["LDPSW"] = a64Enc{format: a64FPair, op: 0x69400000}
|
||||
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
|
||||
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
|
||||
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
|
||||
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
|
||||
a64InstrTable["FLDPS"] = a64Enc{format: a64FPair, op: 0x2d400000}
|
||||
a64InstrTable["FSTPS"] = a64Enc{format: a64FPair, op: 0x2d000000}
|
||||
a64InstrTable["FLDPQ"] = a64Enc{format: a64FPair, op: 0xad400000}
|
||||
a64InstrTable["FSTPQ"] = a64Enc{format: a64FPair, op: 0xad000000}
|
||||
|
||||
// ---- acquire/release loads and stores ----
|
||||
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
|
||||
@@ -806,11 +816,23 @@ func init() {
|
||||
lse := map[string]uint32{
|
||||
"CASALD": 0xc8e0fc00,
|
||||
"CASALW": 0x88e0fc00,
|
||||
"CASB": 0x08a07c00,
|
||||
"CASAB": 0x08e07c00,
|
||||
"CASH": 0x48a07c00,
|
||||
"CASLD": 0xc8a0fc00,
|
||||
"CASLH": 0x48a0fc00,
|
||||
"CASAW": 0x88e07c00,
|
||||
"CASAD": 0xc8e07c00,
|
||||
"CASALH": 0x48e07c00,
|
||||
"LDADDALD": 0xf8e00000,
|
||||
"LDADDALW": 0xb8e00000,
|
||||
"LDADDAD": 0xf8a00000,
|
||||
"LDADDAW": 0xb8a00000,
|
||||
"LDCLRALB": 0x38e01000,
|
||||
"LDCLRALW": 0xb8e01000,
|
||||
"LDCLRALD": 0xf8e01000,
|
||||
"LDCLRAD": 0xf8a01000,
|
||||
"LDCLRAW": 0xb8a01000,
|
||||
"LDORALB": 0x38e03000,
|
||||
"LDORALW": 0xb8e03000,
|
||||
"LDORALD": 0xf8e03000,
|
||||
@@ -889,6 +911,12 @@ func init() {
|
||||
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||
}
|
||||
|
||||
// Compare and swap pair: the second register of each pair is implicit
|
||||
// (Rs+1 and Rt+1), so the encoding carries Rs and Rt alone over a preset
|
||||
// fixed field (asm7.go atomicCASP).
|
||||
a64InstrTable["CASPD"] = a64Enc{format: a64FCASP, op: 1<<30 | 0x41<<21 | 0x1f<<10}
|
||||
a64InstrTable["CASPW"] = a64Enc{format: a64FCASP, op: 0x41<<21 | 0x1f<<10}
|
||||
|
||||
// ---- carry-setting/carry-using arithmetic and widening multiply ----
|
||||
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
|
||||
// register preset to ZR (bits 14:10 = 11111).
|
||||
@@ -940,11 +968,13 @@ func init() {
|
||||
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
|
||||
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
|
||||
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
|
||||
a64InstrTable["VMOVI"] = a64Enc{format: a64FVMoviImm}
|
||||
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
|
||||
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
|
||||
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
|
||||
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
|
||||
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
|
||||
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 7<<10}
|
||||
a64InstrTable["VUSRA"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 5<<10}
|
||||
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
|
||||
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
|
||||
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
|
||||
@@ -957,6 +987,24 @@ func init() {
|
||||
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
// Multi-register structure accesses beyond VLD1/VST1: VLD2/VLD3/VLD4 and
|
||||
// the replicate loads VLD2R/VLD3R, each with the post-index spelling.
|
||||
a64InstrTable["VLD2"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD2.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VLD3"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD3.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VLD4"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD4.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VLD2R"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD2R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VLD3R"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD3R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VST2"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VST2.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VST3"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VST3.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VST4"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VST4.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
}
|
||||
|
||||
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
|
||||
@@ -1120,6 +1168,78 @@ var a64SimdVTable = map[string]a64SimdVSpec{
|
||||
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
||||
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
||||
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
||||
// Saturating shifts, register forms (the immediate spellings route to
|
||||
// a64FShiftImm).
|
||||
"VSQSHL": {0x0e204c00, 0x7f, false},
|
||||
"VUQSHL": {0x2e204c00, 0x7f, false},
|
||||
}
|
||||
|
||||
// a64SimdNLForm classifies the narrow/long/wide SIMD families whose
|
||||
// arrangement does not travel on every operand: the encoding's size and Q
|
||||
// bits read off one designated operand and the element widths pair up across
|
||||
// the operands.
|
||||
type a64SimdNLForm uint8
|
||||
|
||||
const (
|
||||
a64NLTwoNarrow a64SimdNLForm = iota // (Vn.wide, Vd.narrow): size/Q from Vd
|
||||
a64NLTwoLong // (Vn.narrow, Vd.long): size/Q from Vn
|
||||
a64NLThreeLongMul // (Vm.narrow, Vn.narrow, Vd.long): size/Q from Vn
|
||||
a64NLThreeWide // (Vm.narrow, Vn.wide, Vd.wide): size/Q from Vn
|
||||
a64NLThreeLongShift // ($sh, Vn.narrow, Vd.long): size/Q from Vn, immh = esize+sh
|
||||
a64NLThreeNarrowShift // ($sh, Vn.wide, Vd.narrow): size/Q from Vd, immh = esize-sh
|
||||
)
|
||||
|
||||
// a64SimdNLSpec is one narrow/long/wide instruction: the base word (U, opcode
|
||||
// and fixed bits positioned) and the arrangement form. qonly marks the FCVT
|
||||
// family, whose size field is fixed in the base and only the Q bit follows
|
||||
// the driving arrangement.
|
||||
type a64SimdNLSpec struct {
|
||||
base uint32
|
||||
form a64SimdNLForm
|
||||
qonly bool
|
||||
}
|
||||
|
||||
// a64SimdNLTable holds the families the arrangement-driven three-register and
|
||||
// two-register encoders cannot express. The .2 spellings force the 128-bit
|
||||
// side of the pair through their operand arrangements, so the base carries no
|
||||
// arrangement bits of its own.
|
||||
var a64SimdNLTable = map[string]a64SimdNLSpec{
|
||||
"VSHRN": {0x0f008400, a64NLThreeNarrowShift, false},
|
||||
"VSHRN2": {0x0f008400, a64NLThreeNarrowShift, false},
|
||||
"VSXTL": {0x0f00a400, a64NLTwoLong, false},
|
||||
"VSXTL2": {0x0f00a400, a64NLTwoLong, false},
|
||||
"VUXTL": {0x2f00a400, a64NLTwoLong, false},
|
||||
"VUXTL2": {0x2f00a400, a64NLTwoLong, false},
|
||||
"VXTN": {0x0e202800, a64NLTwoNarrow, false},
|
||||
"VXTN2": {0x0e202800, a64NLTwoNarrow, false},
|
||||
"VSQXTN": {0x0e204800, a64NLTwoNarrow, false},
|
||||
"VSQXTN2": {0x0e204800, a64NLTwoNarrow, false},
|
||||
"VSQXTUN": {0x2e202800, a64NLTwoNarrow, false},
|
||||
"VSQXTUN2": {0x2e202800, a64NLTwoNarrow, false},
|
||||
"VUQXTN": {0x2e204800, a64NLTwoNarrow, false},
|
||||
"VUQXTN2": {0x2e204800, a64NLTwoNarrow, false},
|
||||
"VFCVTN": {0x0e206800, a64NLTwoNarrow, true},
|
||||
"VFCVTN2": {0x0e206800, a64NLTwoNarrow, true},
|
||||
"VFCVTL": {0x0e217800, a64NLTwoLong, true},
|
||||
"VFCVTL2": {0x0e217800, a64NLTwoLong, true},
|
||||
"VSSHLL": {0x0f00a400, a64NLThreeLongShift, false},
|
||||
"VSSHLL2": {0x0f00a400, a64NLThreeLongShift, false},
|
||||
"VUSHLL": {0x2f00a400, a64NLThreeLongShift, false},
|
||||
"VUSHLL2": {0x2f00a400, a64NLThreeLongShift, false},
|
||||
"VUADDW": {0x2e201000, a64NLThreeWide, false},
|
||||
"VUADDW2": {0x2e201000, a64NLThreeWide, false},
|
||||
"VUMULL": {0x2e20c000, a64NLThreeLongMul, false},
|
||||
"VUMULL2": {0x2e20c000, a64NLThreeLongMul, false},
|
||||
"VSMULL": {0x0e20c000, a64NLThreeLongMul, false},
|
||||
"VSMULL2": {0x0e20c000, a64NLThreeLongMul, false},
|
||||
"VUMLAL": {0x2e208000, a64NLThreeLongMul, false},
|
||||
"VUMLAL2": {0x2e208000, a64NLThreeLongMul, false},
|
||||
"VSMLAL": {0x0e208000, a64NLThreeLongMul, false},
|
||||
"VSMLAL2": {0x0e208000, a64NLThreeLongMul, false},
|
||||
"VUMLSL": {0x2e20a000, a64NLThreeLongMul, false},
|
||||
"VUMLSL2": {0x2e20a000, a64NLThreeLongMul, false},
|
||||
"VSMLSL": {0x0e20a000, a64NLThreeLongMul, false},
|
||||
"VSMLSL2": {0x0e20a000, a64NLThreeLongMul, false},
|
||||
}
|
||||
|
||||
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
|
||||
@@ -1243,6 +1363,16 @@ var a64PRFOps = map[string]int{
|
||||
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
|
||||
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
|
||||
|
||||
// a64VLDNBase and a64VSTNBase hold the VLD2/VLD3/VLD4 and VST2/VST3/VST4
|
||||
// fixed words (indexed by register count 2..4): the opcode field at bits
|
||||
// 15:12 carries the access kind.
|
||||
var a64VLDNBase = [5]uint32{0, 0, 0x0c408000, 0x0c404000, 0x0c400000}
|
||||
var a64VSTNBase = [5]uint32{0, 0, 0x0c008000, 0x0c004000, 0x0c000000}
|
||||
|
||||
// a64VLDNReplicate holds the VLD2R/VLD3R fixed words beside the existing
|
||||
// VLD1R (0x0d40c000) and VLD4R (0x0d60e000) bases.
|
||||
var a64VLDNReplicate = [5]uint32{0, 0x0d40c000, 0x0d60c000, 0x0d40e000, 0x0d60e000}
|
||||
|
||||
// a64Vec is a parsed vector operand: the register number, the arrangement
|
||||
// ("" when the operand spells none) and, for element forms, the lane index.
|
||||
type a64Vec struct {
|
||||
@@ -1363,23 +1493,25 @@ func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool)
|
||||
|
||||
// a64LSType describes the load/store parameters for a MOV width mnemonic.
|
||||
type a64LSType struct {
|
||||
size int // 0=byte, 1=half, 2=word, 3=dword
|
||||
V int // 0=integer, 1=FP
|
||||
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
|
||||
size int // 0=byte, 1=half, 2=word, 3=dword
|
||||
V int // 0=integer, 1=FP
|
||||
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
|
||||
scale int // access width in bytes; the unsigned offset divides by it
|
||||
}
|
||||
|
||||
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
|
||||
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
|
||||
var a64LoadTable = map[string]a64LSType{
|
||||
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
|
||||
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
|
||||
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
|
||||
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
|
||||
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
|
||||
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
|
||||
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
|
||||
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
|
||||
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
|
||||
"MOVD": {3, 0, 1, 8}, // LDR X (64-bit, unsigned offset)
|
||||
"MOVWU": {2, 0, 1, 4}, // LDR W (32-bit unsigned)
|
||||
"MOVW": {2, 0, 2, 4}, // LDRSW (32-bit signed → 64-bit)
|
||||
"MOVHU": {1, 0, 1, 2}, // LDRH (16-bit unsigned)
|
||||
"MOVH": {1, 0, 2, 2}, // LDRSH (16-bit signed)
|
||||
"MOVBU": {0, 0, 1, 1}, // LDRB (8-bit unsigned)
|
||||
"MOVB": {0, 0, 2, 1}, // LDRSB (8-bit signed)
|
||||
"FMOVS": {2, 1, 1, 4}, // LDR S (32-bit FP)
|
||||
"FMOVD": {3, 1, 1, 8}, // LDR D (64-bit FP)
|
||||
"FMOVQ": {0, 1, 3, 16}, // LDR/STR Q (128-bit FP): opc=11 selects it
|
||||
}
|
||||
|
||||
// a64StoreOpc returns the store opc for a given load type: integer and FP
|
||||
|
||||
Reference in New Issue
Block a user