fix(arm64): encode shifts, divides and multiplies and align sizes with emission
Assisted-by: GLM 5.3
This commit is contained in:
+54
-68
@@ -27,6 +27,8 @@ package asm
|
||||
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
|
||||
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
|
||||
|
||||
import "maps"
|
||||
|
||||
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
|
||||
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
|
||||
// runtime's assembly uses. Returns -1 for an unrecognised name.
|
||||
@@ -262,19 +264,15 @@ type a64Format uint8
|
||||
|
||||
const (
|
||||
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
||||
a64FDPIR // data-processing (immediate): ADD/SUB $imm
|
||||
a64FLogImm // logical (immediate): AND/ORR/EOR $imm
|
||||
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
||||
a64FLSU // load/store (unsigned immediate, scaled)
|
||||
a64FLSUnscaled // load/store (unscaled immediate)
|
||||
a64FLSPair // load/store pair
|
||||
a64FBranch // unconditional branch (B/BL)
|
||||
a64FBranchCond // conditional branch (B.cond)
|
||||
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
||||
a64FADR // ADR/ADRP
|
||||
a64FEXTR // EXTR
|
||||
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
||||
a64FSystem // system: NOP, BRK, etc.
|
||||
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
||||
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
||||
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
||||
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
||||
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
||||
@@ -282,7 +280,6 @@ const (
|
||||
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
||||
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
||||
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
||||
a64FFMovGR // FMOV between GP and FP registers
|
||||
a64FCRC32 // CRC32
|
||||
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
||||
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR
|
||||
@@ -332,30 +329,14 @@ func init() {
|
||||
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
|
||||
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
||||
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
||||
// Shift
|
||||
"LSL": 1<<31 | 0<<29 | 0x0a<<24, // alias of UBFM
|
||||
"LSLW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"LSR": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"LSRW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"ASR": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"ASRW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
"ROR": 1<<31 | 0<<29 | 0x0a<<24,
|
||||
"RORW": 0<<31 | 0<<29 | 0x0a<<24,
|
||||
// Multiply
|
||||
"MADD": 1<<31 | 0<<29 | 0x1b<<24 | 0<<21,
|
||||
"MADDW": 0<<31 | 0<<29 | 0x1b<<24 | 0<<21,
|
||||
"MSUB": 1<<31 | 0<<29 | 0x1b<<24 | 1<<21,
|
||||
"MSUBW": 0<<31 | 0<<29 | 0x1b<<24 | 1<<21,
|
||||
// Divide
|
||||
"SDIV": 1<<31 | 0<<29 | 0x0d<<24,
|
||||
"SDIVW": 0<<31 | 0<<29 | 0x0d<<24,
|
||||
"UDIV": 1<<31 | 0<<29 | 0x0d<<24 | 1<<10,
|
||||
"UDIVW": 0<<31 | 0<<29 | 0x0d<<24 | 1<<10,
|
||||
// CRC
|
||||
"CRC32B": 0<<31 | 0<<29 | 0x1b<<24 | 4<<10,
|
||||
"CRC32H": 0<<31 | 0<<29 | 0x1b<<24 | 5<<10,
|
||||
"CRC32W": 0<<31 | 0<<29 | 0x1b<<24 | 6<<10,
|
||||
"CRC32X": 1<<31 | 0<<29 | 0x1b<<24 | 7<<10,
|
||||
// Divide (data-processing 2 source): the opcode occupies bits 15:10
|
||||
// of the 0xd6<<21 fixed field, UDIV=0b0010 and SDIV=0b0011 (ARM ARM
|
||||
// "Data-processing (2 source)"; the toolchain spells them OPDP2(2)
|
||||
// and OPDP2(3)). sf=1 selects the X forms.
|
||||
"SDIV": 1<<31 | 0xd6<<21 | 3<<10,
|
||||
"SDIVW": 0<<31 | 0xd6<<21 | 3<<10,
|
||||
"UDIV": 1<<31 | 0xd6<<21 | 2<<10,
|
||||
"UDIVW": 0<<31 | 0xd6<<21 | 2<<10,
|
||||
// Conditional select
|
||||
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
||||
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
||||
@@ -385,14 +366,37 @@ func init() {
|
||||
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
|
||||
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
|
||||
|
||||
// ---- data-processing (immediate) ----
|
||||
// ADD/SUB $imm, Rn, Rd
|
||||
a64InstrTable["ADDImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["ADDWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 0<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["SUBImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["SUBWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 1<<30 | 0<<29 | 0x11<<24}
|
||||
a64InstrTable["ADDSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 1<<29 | 0x11<<24}
|
||||
a64InstrTable["SUBSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 1<<29 | 0x11<<24}
|
||||
// ---- shifts ----
|
||||
// The mnemonic serves both forms: with an immediate the aliases of the
|
||||
// data-processing (immediate) group apply (ARM ARM "Shifts"), with a
|
||||
// register the data-processing (2 source) LSLV/LSRV/ASRV/RORV. The op
|
||||
// field carries the immediate-alias base; encodeARM64Shift derives both
|
||||
// it and the two-source opcode. Identities, W = 64 (X) or 32 (W):
|
||||
//
|
||||
// LSL $sh, Rn, Rd = UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh
|
||||
// LSR $sh, Rn, Rd = UBFM Rd, Rn, #sh, #(W-1)
|
||||
// ASR $sh, Rn, Rd = SBFM Rd, Rn, #sh, #(W-1)
|
||||
// ROR $sh, Rn, Rd = EXTR Rd, Rn, Rn, #sh
|
||||
shifts := map[string]a64Enc{
|
||||
"LSL": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X
|
||||
"LSLW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W
|
||||
"LSR": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X
|
||||
"LSRW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W
|
||||
"ASR": {format: a64FShift, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}, // SBFM X
|
||||
"ASRW": {format: a64FShift, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}, // SBFM W
|
||||
"ROR": {format: a64FShift, op: 1<<31 | 0x27<<23 | 1<<22}, // EXTR X
|
||||
"RORW": {format: a64FShift, op: 0<<31 | 0x27<<23 | 0<<22}, // EXTR W
|
||||
}
|
||||
maps.Copy(a64InstrTable, shifts)
|
||||
|
||||
// ---- multiply accumulate ----
|
||||
// MADD/MSUB Rm, Ra, Rn, Rd: sf 00 11011 o0(15) Rm Ra Rn Rd. The
|
||||
// toolchain's optab has no shorter row, so all four operands are
|
||||
// mandatory, and Ra is the SECOND operand.
|
||||
a64InstrTable["MADD"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24}
|
||||
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
|
||||
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
|
||||
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
|
||||
|
||||
// ---- move wide ----
|
||||
// MOVZ/MOVN/MOVK
|
||||
@@ -407,22 +411,9 @@ func init() {
|
||||
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
|
||||
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
|
||||
|
||||
// ---- load/store (unsigned immediate) ----
|
||||
a64InstrTable["MOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<22} // LDR 64-bit
|
||||
a64InstrTable["MOVWU"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<22} // LDR 32-bit unsigned
|
||||
a64InstrTable["MOVHU"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 1<<22} // LDRH unsigned
|
||||
a64InstrTable["MOVBU"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 1<<22} // LDRB unsigned
|
||||
a64InstrTable["MOVW"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 2<<22} // LDRSW (signed 32→64)
|
||||
a64InstrTable["MOVH"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 2<<22} // LDRSH (signed half)
|
||||
a64InstrTable["MOVB"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 2<<22} // LDRSB (signed byte)
|
||||
a64InstrTable["FMOVS"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 32-bit FP
|
||||
a64InstrTable["FMOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 64-bit FP
|
||||
|
||||
// Store opcodes (load ^ (1<<22)):
|
||||
// STR 64-bit: size=3, V=0, opc=00 → 3<<30 | 7<<27 | 0<<22
|
||||
// STR 32-bit: size=2, V=0, opc=00 → 2<<30 | 7<<27 | 0<<22
|
||||
// STRH: size=1, V=0, opc=00 → 1<<30 | 7<<27 | 0<<22
|
||||
// STRB: size=0, V=0, opc=00 → 0<<30 | 7<<27 | 0<<22
|
||||
// Load/store mnemonics never enter this table: the MOV pseudo-instruction
|
||||
// dispatch handles them through a64LoadTable, which also carries the store
|
||||
// opcode (integer and FP stores both use opc=00, differing only in V).
|
||||
|
||||
// ---- branches ----
|
||||
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
|
||||
@@ -445,10 +436,8 @@ func init() {
|
||||
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
|
||||
|
||||
// ---- system ----
|
||||
a64InstrTable["NOP"] = a64Enc{format: a64FSystem, op: a64NOP}
|
||||
a64InstrTable["NOOP"] = a64Enc{format: a64FSystem, op: a64NOP}
|
||||
a64InstrTable["BRK"] = a64Enc{format: a64FSystem, op: 0xd4200000}
|
||||
a64InstrTable["UNDEF"] = a64Enc{format: a64FSystem, op: a64BRK(0)}
|
||||
// NOP/NOOP/UNDEF are spelled out in encodeARM64Instr's pseudo switch,
|
||||
// so they carry no table entry; a64NOP and a64BRK are the encoders.
|
||||
|
||||
// ---- EXTR ----
|
||||
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
|
||||
@@ -549,8 +538,8 @@ func init() {
|
||||
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
|
||||
}
|
||||
|
||||
// ---- FMOV between GP and FP registers ----
|
||||
a64InstrTable["FMOVGR"] = a64Enc{format: a64FFMovGR, op: 0x1e260000} // placeholder, actual encoding depends on direction
|
||||
// FMOV between GP and FP registers needs no table entry: the MOV
|
||||
// pseudo-instruction dispatches it by operand class (encodeARM64RegMove).
|
||||
|
||||
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
|
||||
csel := map[string]uint32{
|
||||
@@ -642,14 +631,11 @@ var a64LoadTable = map[string]a64LSType{
|
||||
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
|
||||
}
|
||||
|
||||
// a64StoreOpc returns the store opc for a given load type.
|
||||
// For integer: store opc = 00 (the load opc bits cleared).
|
||||
// For FP: store opc = 00 (same pattern).
|
||||
// a64StoreOpc returns the store opc for a given load type: integer and FP
|
||||
// stores both encode opc=00 (the load's signedness bit sits in opc[1], which
|
||||
// the store form clears; FP registers are selected by V, not opc).
|
||||
func a64StoreOpc(t a64LSType) int {
|
||||
if t.V == 1 {
|
||||
return 0 // FP store
|
||||
}
|
||||
return 0 // integer store
|
||||
return 0
|
||||
}
|
||||
|
||||
// arm64RegClass discriminates integer (R), floating-point (F) registers for
|
||||
|
||||
Reference in New Issue
Block a user