Files
gasm-sdk/asm/arm64_encode.go
T

765 lines
29 KiB
Go
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// arm64 (AArch64) instruction encoding.
//
// The encoder is data-driven: each mnemonic maps to an instruction format and
// an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own arm64
// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite.
//
// All AArch64 instructions are 32 bits, little-endian. The formats used here
// (per the ARM Architecture Reference Manual):
//
// DP-shifted-reg sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
// DP-immediate sf<<31 | op<<30 | S<<29 | 0x11<<24 | imm12<<10 | Rn<<5 | Rd
// Logical-imm sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd
// Move-wide sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd
// Load/store size<<30 | 0x7<<27 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt
// LDST-unscaled size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<5 | Rt (actually imm9<<12 | Rn<<5 | Rt)
// LDST-pair opc<<30 | 0x5<<27 | V<<26 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt
// Branch-imm 0<<31 | 0x5<<26 | imm26 (B)
// Branch-imm 1<<31 | 0x5<<26 | imm26 (BL)
// Branch-cond 0x2A<<25 | imm19<<5 | cond (B.cond)
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0–R30 (integer), F0–F31 (floating point), and the ABI aliases the
// runtime's assembly uses. Returns -1 for an unrecognised name.
func arm64RegNum(name string) int {
switch name {
case "R0":
return 0
case "R1":
return 1
case "R2":
return 2
case "R3":
return 3
case "R4":
return 4
case "R5":
return 5
case "R6":
return 6
case "R7":
return 7
case "R8":
return 8
case "R9":
return 9
case "R10":
return 10
case "R11":
return 11
case "R12":
return 12
case "R13":
return 13
case "R14":
return 14
case "R15":
return 15
case "R16":
return 16
case "R17":
return 17
case "R18":
return 18
case "R19":
return 19
case "R20":
return 20
case "R21":
return 21
case "R22":
return 22
case "R23":
return 23
case "R24":
return 24
case "R25":
return 25
case "R26", "REGCTXT", "CTXT":
return 26
case "R27", "REGTMP", "TMP":
return 27
case "R28", "REGG", "g":
return 28
case "R29", "FP":
return 29
case "R30", "LR", "LINK":
return 30
case "R31", "ZR":
return 31
case "SP":
return 31 // SP and ZR share encoding 31; context determines meaning
}
// F0–F31.
if len(name) >= 1 && name[0] == 'F' {
n := 0
for i := 1; i < len(name); i++ {
if name[i] < '0' || name[i] > '9' {
return -1
}
n = n*10 + int(name[i]-'0')
}
if n <= 31 {
return n
}
}
return -1
}
// arm64IsSP reports whether a register operand is the stack pointer (R31/SP),
// which uses a different encoding path for some instructions.
func arm64IsSP(name string) bool {
return name == "SP"
}
// ---- format helpers ----
// a64wordLE encodes a uint32 as 4 little-endian bytes.
func a64wordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// a64WordsLE concatenates one or more instruction words as little-endian bytes.
func a64WordsLE(ws ...uint32) []byte {
var out []byte
for _, w := range ws {
out = append(out, a64wordLE(w)...)
}
return out
}
// ---- data-processing (shifted register) ----
// a64DPSR encodes a data-processing (shifted register) instruction:
// sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd.
func a64DPSR(sf, op, S, shift, rm, imm6, rn, rd uint32) uint32 {
return sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | rm<<16 | imm6<<10 | rn<<5 | rd
}
// ---- data-processing (immediate) ----
// a64AddSub encodes an ADD/SUB (immediate) instruction:
// sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | Rn<<5 | Rd.
func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 {
return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd
}
// ---- logical (immediate) ----
// a64LogicalImm encodes a logical (immediate) instruction:
// sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd.
func a64LogicalImm(sf, opc, N, immr, imms, rn, rd uint32) uint32 {
return sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | rn<<5 | rd
}
// ---- move wide ----
// a64MoveWide encodes a MOVZ/MOVK/MOVN instruction:
// sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd.
func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
}
// ---- load/store (unsigned immediate, scaled) ----
// a64LSU encodes a load/store register (unsigned immediate, scaled):
// size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt.
// (0x39<<24 encodes bits 29:24 = 111001, the scaled unsigned offset form.)
func a64LSU(size, V, opc, imm12, rn, rt uint32) uint32 {
return size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | rn<<5 | rt
}
// ---- load/store (unscaled immediate) ----
// a64LSUnscaled encodes a load/store register (unscaled immediate, 9-bit signed):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<12 | Rn<<5 | Rt.
// Note: the 0<<24 distinguishes unscaled from the pre/post-index forms.
func a64LSUnscaled(size, V, opc int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | uint32(opc)<<22 |
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// ---- load/store pair ----
// a64LSP encodes a load/store pair instruction (signed offset):
// opc<<30 | 0x5<<27 | V<<26 | 2<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
// opc: 0=32-bit, 1=reserved, 2=64-bit. V: 0=integer, 1=FP/SIMD.
// L: 0=store, 1=load. imm7 is the signed scaled offset (÷8 for 64-bit pairs).
func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- load/store pair (pre-index) ----
// a64LSPPre encodes a load/store pair (pre-index):
// opc<<30 | 0x5<<27 | V<<26 | 0b11<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
func a64LSPPre(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 3<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- load/store pair (post-index) ----
// a64LSPPost encodes a load/store pair (post-index):
// opc<<30 | 0x5<<27 | V<<26 | 0b01<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
func a64LSPPost(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 1<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- pre-index load/store ----
// a64LSPreIndex encodes a load/store register (pre-index):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func a64LSPreIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 {
return size<<30 | 7<<27 | V<<26 | opc<<22 | 3<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt
}
// ---- post-index load/store ----
// a64LSPostIndex encodes a load/store register (post-index):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func a64LSPostIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 {
return size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt
}
// ---- branches ----
// a64Branch encodes an unconditional branch (B/BL):
// op<<31 | 0x5<<26 | imm26.
func a64Branch(op uint32, imm26 int32) uint32 {
return op<<31 | 5<<26 | (uint32(imm26) & 0x03FFFFFF)
}
// a64BranchCond encodes a conditional branch (B.cond):
// 0x2A<<25 | imm19<<5 | cond.
func a64BranchCond(imm19 int32, cond uint32) uint32 {
return 0x2A<<25 | (uint32(imm19)&0x7FFFF)<<5 | cond&0xF
}
// a64UncondBranch encodes an unconditional branch register (BR/BLR/RET):
// 0x6B<<25 | opc<<21 | 0x1F<<16 | Rn<<5 | Rd.
// opc: 0=BR, 1=BLR, 2=RET. For RET, Rn defaults to LR(30).
func a64UncondBranch(opc, rn, rd uint32) uint32 {
return 0x6B<<25 | opc<<21 | 0x1F<<16 | rn<<5 | rd
}
// ---- ADR/ADRP ----
// a64ADR encodes an ADR instruction (p=0) or ADRP instruction (p=1):
// p<<31 | immlo<<29 | 0x10<<24 | immhi<<5 | Rd.
func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 {
return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd
}
// ---- EXTR ----
// a64EXTR encodes an EXTR instruction:
// sf<<31 | 0<<29 | 0x27<<23 | N<<22 | 0<<21 | Rm<<16 | imms<<10 | Rn<<5 | Rd.
func a64EXTR(sf, N, rm, imms, rn, rd uint32) uint32 {
return sf<<31 | 0x27<<23 | N<<22 | rm<<16 | imms<<10 | rn<<5 | rd
}
// ---- system ----
// a64NOP encodes a NOP: 0xd503201f.
const a64NOP uint32 = 0xd503201f
// a64BRK encodes a BRK instruction: 0xd4200000 | imm16<<5.
func a64BRK(imm16 uint32) uint32 {
return 0xd4200000 | imm16<<5
}
// ---- condition codes ----
const (
a64CondEQ = 0x0
a64CondNE = 0x1
a64CondCS = 0x2
a64CondHS = 0x2
a64CondCC = 0x3
a64CondLO = 0x3
a64CondMI = 0x4
a64CondPL = 0x5
a64CondVS = 0x6
a64CondVC = 0x7
a64CondHI = 0x8
a64CondLS = 0x9
a64CondGE = 0xa
a64CondLT = 0xb
a64CondGT = 0xc
a64CondLE = 0xd
a64CondAL = 0xe
a64CondNV = 0xf
)
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
var arm64CondMap = map[string]uint32{
"EQ": a64CondEQ,
"NE": a64CondNE,
"CS": a64CondCS,
"HS": a64CondHS,
"CC": a64CondCC,
"LO": a64CondLO,
"MI": a64CondMI,
"PL": a64CondPL,
"VS": a64CondVS,
"VC": a64CondVC,
"HI": a64CondHI,
"LS": a64CondLS,
"GE": a64CondGE,
"LT": a64CondLT,
"GT": a64CondGT,
"LE": a64CondLE,
}
// ---- instruction format tags ----
type a64Format uint8
const (
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FDPIR // data-processing (immediate): ADD/SUB $imm
a64FLogImm // logical (immediate): AND/ORR/EOR $imm
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FLSU // load/store (unsigned immediate, scaled)
a64FLSUnscaled // load/store (unscaled immediate)
a64FLSPair // load/store pair
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FSystem // system: NOP, BRK, etc.
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FFMovGR // FMOV between GP and FP registers
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
)
// a64Enc is one instruction's encoding: its bit layout (format) and the
// opcode constant, positioned at its exact bit range.
type a64Enc struct {
format a64Format
op uint32 // the pre-positioned opcode bits
size int // 4 for most, 8 for DP-imm with shift, etc.
}
// a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to
// their encoding. The base integer, memory, floating-point and SIMD
// instruction sets are covered.
var a64InstrTable = map[string]a64Enc{}
func init() {
// ---- data-processing (shifted register) ----
// Format: sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
dpsr := map[string]uint32{
// Add/Sub
"ADD": 1<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=1, op=0, S=0 (64-bit default)
"ADDW": 0<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=0
"ADDS": 1<<31 | 0<<30 | 1<<29 | 0x0b<<24,
"ADDSW": 0<<31 | 0<<30 | 1<<29 | 0x0b<<24,
"SUB": 1<<31 | 1<<30 | 0<<29 | 0x0b<<24,
"SUBW": 0<<31 | 1<<30 | 0<<29 | 0x0b<<24,
"SUBS": 1<<31 | 1<<30 | 1<<29 | 0x0b<<24,
"SUBSW": 0<<31 | 1<<30 | 1<<29 | 0x0b<<24,
// Logical (shifted register)
"AND": 1<<31 | 0<<29 | 0x0a<<24,
"ANDW": 0<<31 | 0<<29 | 0x0a<<24,
"BIC": 1<<31 | 0<<29 | 0x0a<<24 | 1<<21,
"BICW": 0<<31 | 0<<29 | 0x0a<<24 | 1<<21,
"ORR": 1<<31 | 1<<29 | 0x0a<<24,
"ORRW": 0<<31 | 1<<29 | 0x0a<<24,
"ORN": 1<<31 | 1<<29 | 0x0a<<24 | 1<<21,
"ORNW": 0<<31 | 1<<29 | 0x0a<<24 | 1<<21,
"EOR": 1<<31 | 2<<29 | 0x0a<<24,
"EORW": 0<<31 | 2<<29 | 0x0a<<24,
"EON": 1<<31 | 2<<29 | 0x0a<<24 | 1<<21,
"EONW": 0<<31 | 2<<29 | 0x0a<<24 | 1<<21,
"ANDS": 1<<31 | 3<<29 | 0x0a<<24,
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
// Shift
"LSL": 1<<31 | 0<<29 | 0x0a<<24, // alias of UBFM
"LSLW": 0<<31 | 0<<29 | 0x0a<<24,
"LSR": 1<<31 | 0<<29 | 0x0a<<24,
"LSRW": 0<<31 | 0<<29 | 0x0a<<24,
"ASR": 1<<31 | 0<<29 | 0x0a<<24,
"ASRW": 0<<31 | 0<<29 | 0x0a<<24,
"ROR": 1<<31 | 0<<29 | 0x0a<<24,
"RORW": 0<<31 | 0<<29 | 0x0a<<24,
// Multiply
"MADD": 1<<31 | 0<<29 | 0x1b<<24 | 0<<21,
"MADDW": 0<<31 | 0<<29 | 0x1b<<24 | 0<<21,
"MSUB": 1<<31 | 0<<29 | 0x1b<<24 | 1<<21,
"MSUBW": 0<<31 | 0<<29 | 0x1b<<24 | 1<<21,
// Divide
"SDIV": 1<<31 | 0<<29 | 0x0d<<24,
"SDIVW": 0<<31 | 0<<29 | 0x0d<<24,
"UDIV": 1<<31 | 0<<29 | 0x0d<<24 | 1<<10,
"UDIVW": 0<<31 | 0<<29 | 0x0d<<24 | 1<<10,
// CRC
"CRC32B": 0<<31 | 0<<29 | 0x1b<<24 | 4<<10,
"CRC32H": 0<<31 | 0<<29 | 0x1b<<24 | 5<<10,
"CRC32W": 0<<31 | 0<<29 | 0x1b<<24 | 6<<10,
"CRC32X": 1<<31 | 0<<29 | 0x1b<<24 | 7<<10,
// Conditional select
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
"CSINC": 1<<31 | 0<<29 | 0x1d<<24 | 1<<10,
"CSINCW": 0<<31 | 0<<29 | 0x1d<<24 | 1<<10,
"CSINV": 1<<31 | 0<<29 | 0x1d<<24 | 2<<10,
"CSINVW": 0<<31 | 0<<29 | 0x1d<<24 | 2<<10,
"CSNEG": 1<<31 | 0<<29 | 0x1d<<24 | 3<<10,
"CSNEGW": 0<<31 | 0<<29 | 0x1d<<24 | 3<<10,
}
for m, op := range dpsr {
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
}
// Aliases that map to the same encoding as their target.
a64InstrTable["CMP"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
a64InstrTable["CMPW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBSW"]}
a64InstrTable["CMN"] = a64Enc{format: a64FDPSR, op: dpsr["ADDS"]}
a64InstrTable["CMNW"] = a64Enc{format: a64FDPSR, op: dpsr["ADDSW"]}
a64InstrTable["TST"] = a64Enc{format: a64FDPSR, op: dpsr["ANDS"]}
a64InstrTable["TSTW"] = a64Enc{format: a64FDPSR, op: dpsr["ANDSW"]}
a64InstrTable["NEG"] = a64Enc{format: a64FDPSR, op: dpsr["SUB"]}
a64InstrTable["NEGW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBW"]}
a64InstrTable["NEGS"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
a64InstrTable["MVN"] = a64Enc{format: a64FDPSR, op: dpsr["ORN"]}
a64InstrTable["MVNW"] = a64Enc{format: a64FDPSR, op: dpsr["ORNW"]}
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
// ---- data-processing (immediate) ----
// ADD/SUB $imm, Rn, Rd
a64InstrTable["ADDImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 0<<29 | 0x11<<24}
a64InstrTable["ADDWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 0<<30 | 0<<29 | 0x11<<24}
a64InstrTable["SUBImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 0<<29 | 0x11<<24}
a64InstrTable["SUBWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 1<<30 | 0<<29 | 0x11<<24}
a64InstrTable["ADDSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 1<<29 | 0x11<<24}
a64InstrTable["SUBSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 1<<29 | 0x11<<24}
// ---- move wide ----
// MOVZ/MOVN/MOVK
a64InstrTable["MOVZ"] = a64Enc{format: a64FMovWide, op: 1<<31 | 2<<29 | 0x25<<23}
a64InstrTable["MOVZW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 2<<29 | 0x25<<23}
a64InstrTable["MOVN"] = a64Enc{format: a64FMovWide, op: 1<<31 | 0<<29 | 0x25<<23}
a64InstrTable["MOVNW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 0<<29 | 0x25<<23}
a64InstrTable["MOVK"] = a64Enc{format: a64FMovWide, op: 1<<31 | 3<<29 | 0x25<<23}
a64InstrTable["MOVKW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 3<<29 | 0x25<<23}
// ---- ADR/ADRP ----
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
// ---- load/store (unsigned immediate) ----
a64InstrTable["MOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<22} // LDR 64-bit
a64InstrTable["MOVWU"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<22} // LDR 32-bit unsigned
a64InstrTable["MOVHU"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 1<<22} // LDRH unsigned
a64InstrTable["MOVBU"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 1<<22} // LDRB unsigned
a64InstrTable["MOVW"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 2<<22} // LDRSW (signed 32→64)
a64InstrTable["MOVH"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 2<<22} // LDRSH (signed half)
a64InstrTable["MOVB"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 2<<22} // LDRSB (signed byte)
a64InstrTable["FMOVS"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 32-bit FP
a64InstrTable["FMOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 64-bit FP
// Store opcodes (load ^ (1<<22)):
// STR 64-bit: size=3, V=0, opc=00 → 3<<30 | 7<<27 | 0<<22
// STR 32-bit: size=2, V=0, opc=00 → 2<<30 | 7<<27 | 0<<22
// STRH: size=1, V=0, opc=00 → 1<<30 | 7<<27 | 0<<22
// STRB: size=0, V=0, opc=00 → 0<<30 | 7<<27 | 0<<22
// ---- branches ----
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
a64InstrTable["BL"] = a64Enc{format: a64FBranch, op: 1<<31 | 5<<26}
// Conditional branches.
condBranches := map[string]uint32{
"BEQ": 0x0, "BNE": 0x1, "BCS": 0x2, "BHS": 0x2,
"BCC": 0x3, "BLO": 0x3, "BMI": 0x4, "BPL": 0x5,
"BVS": 0x6, "BVC": 0x7, "BHI": 0x8, "BLS": 0x9,
"BGE": 0xa, "BLT": 0xb, "BGT": 0xc, "BLE": 0xd,
}
for name, cond := range condBranches {
a64InstrTable[name] = a64Enc{format: a64FBranchCond, op: 0x2A<<25 | cond}
}
// Unconditional branch register (BR/BLR/RET).
a64InstrTable["BR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 0<<21}
a64InstrTable["BLR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 1<<21}
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
// ---- system ----
a64InstrTable["NOP"] = a64Enc{format: a64FSystem, op: a64NOP}
a64InstrTable["NOOP"] = a64Enc{format: a64FSystem, op: a64NOP}
a64InstrTable["BRK"] = a64Enc{format: a64FSystem, op: 0xd4200000}
a64InstrTable["UNDEF"] = a64Enc{format: a64FSystem, op: a64BRK(0)}
// ---- EXTR ----
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
a64InstrTable["EXTRW"] = a64Enc{format: a64FEXTR, op: 0<<31 | 0x27<<23 | 0<<22}
// ---- bitfield ----
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
// ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ----
fp3 := map[string]uint32{
"FADDS": 0x1e202800, "FADDD": 0x1e602800,
"FSUBS": 0x1e203800, "FSUBD": 0x1e603800,
"FMULS": 0x1e200800, "FMULD": 0x1e600800,
"FDIVS": 0x1e201800, "FDIVD": 0x1e601800,
"FMAXS": 0x1e204800, "FMAXD": 0x1e604800,
"FMINS": 0x1e205800, "FMIND": 0x1e605800,
"FMAXNMS": 0x1e206800, "FMAXNMD": 0x1e606800,
"FMINNMS": 0x1e207800, "FMINNMD": 0x1e607800,
"FNMULS": 0x1e208800, "FNMULD": 0x1e608800,
}
for m, op := range fp3 {
a64InstrTable[m] = a64Enc{format: a64FFP3, op: op}
}
// ---- FP unary (Rn, Rd): FMOV reg-reg, FABS, FNEG, FSQRT, FCVT, FRINT* ----
fp1 := map[string]uint32{
"FMOVS": 0x1e204000, "FMOVD": 0x1e604000,
"FABSS": 0x1e20c000, "FABSD": 0x1e60c000,
"FNEGS": 0x1e214000, "FNEGD": 0x1e614000,
"FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000,
"FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000,
"FRINTNS": 0x1e244000, "FRINTND": 0x1e644000,
"FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000,
"FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000,
"FRINTZS": 0x1e25c000, "FRINTZD": 0x1e65c000,
"FRINTAS": 0x1e264000, "FRINTAD": 0x1e664000,
"FRINTXS": 0x1e274000, "FRINTXD": 0x1e674000,
"FRINTIS": 0x1e27c000, "FRINTID": 0x1e67c000,
}
for m, op := range fp1 {
a64InstrTable[m] = a64Enc{format: a64FFPUnary, op: op}
}
// ---- FP 4-operand FMA (Ra, Rm, Rn, Rd) ----
fp4 := map[string]uint32{
"FMADDS": 0x1f000000, "FMADDD": 0x1f400000,
"FMSUBS": 0x1f008000, "FMSUBD": 0x1f408000,
"FNMADDS": 0x1f200000, "FNMADDD": 0x1f600000,
"FNMSUBS": 0x1f208000, "FNMSUBD": 0x1f608000,
}
for m, op := range fp4 {
a64InstrTable[m] = a64Enc{format: a64FFP4, op: op}
}
// ---- FP compare (Rm, Rn or #0, Rn) ----
fpcmp := map[string]uint32{
"FCMPS": 0x1e202000, "FCMPD": 0x1e602000,
"FCMPES": 0x1e202010, "FCMPED": 0x1e602010,
}
for m, op := range fpcmp {
a64InstrTable[m] = a64Enc{format: a64FFPCmp, op: op}
}
// ---- FP conditional compare (Rm, Rn, #nzcv, cond) ----
fpccmp := map[string]uint32{
"FCCMPS": 0x1e200400, "FCCMPD": 0x1e600400,
"FCCMPES": 0x1e200410, "FCCMPED": 0x1e600410,
}
for m, op := range fpccmp {
a64InstrTable[m] = a64Enc{format: a64FFPCCmp, op: op}
}
// ---- FP conditional select (Rm, Rn, Rd, cond) ----
a64InstrTable["FCSELS"] = a64Enc{format: a64FFPSel, op: 0x1e200c00}
a64InstrTable["FCSELD"] = a64Enc{format: a64FFPSel, op: 0x1e600c00}
// ---- FP ↔ integer conversion ----
fpcvt := map[string]uint32{
"FCVTZSD": 0x9e780000, "FCVTZSDW": 0x1e780000,
"FCVTZSS": 0x9e380000, "FCVTZSSW": 0x1e380000,
"FCVTZUD": 0x9e790000, "FCVTZUDW": 0x1e790000,
"FCVTZUS": 0x9e390000, "FCVTZUSW": 0x1e390000,
"SCVTFD": 0x9e620000, "SCVTFS": 0x9e220000,
"SCVTFWD": 0x1e620000, "SCVTFWS": 0x1e220000,
"UCVTFD": 0x9e630000, "UCVTFS": 0x9e230000,
"UCVTFWD": 0x1e630000, "UCVTFWS": 0x1e230000,
}
for m, op := range fpcvt {
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
}
// ---- FMOV between GP and FP registers ----
a64InstrTable["FMOVGR"] = a64Enc{format: a64FFMovGR, op: 0x1e260000} // placeholder, actual encoding depends on direction
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
csel := map[string]uint32{
"CSEL": 0x9a800000, "CSELW": 0x1a800000,
"CSINC": 0x9a800400, "CSINCW": 0x1a800400,
"CSINV": 0xda800000, "CSINVW": 0x5a800000,
"CSNEG": 0xda800400, "CSNEGW": 0x5a800400,
}
for m, op := range csel {
a64InstrTable[m] = a64Enc{format: a64FCSEL, op: op}
}
// Aliases
a64InstrTable["CSET"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
a64InstrTable["CSETW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
a64InstrTable["CSETM"] = a64Enc{format: a64FCSEL, op: 0xda800000}
a64InstrTable["CSETMW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
a64InstrTable["CINC"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
a64InstrTable["CINCW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
a64InstrTable["CINV"] = a64Enc{format: a64FCSEL, op: 0xda800000}
a64InstrTable["CINVW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
a64InstrTable["CNEG"] = a64Enc{format: a64FCSEL, op: 0xda800400}
a64InstrTable["CNEGW"] = a64Enc{format: a64FCSEL, op: 0x5a800400}
// ---- CRC32 ----
crc32 := map[string]uint32{
"CRC32B": 0x1ac04000, "CRC32H": 0x1ac04400,
"CRC32W": 0x1ac04800, "CRC32X": 0x9ac04c00,
"CRC32CB": 0x1ac05000, "CRC32CH": 0x1ac05400,
"CRC32CW": 0x1ac05800, "CRC32CX": 0x9ac05c00,
}
for m, op := range crc32 {
a64InstrTable[m] = a64Enc{format: a64FCRC32, op: op}
}
// ---- exclusive load/store ----
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
a64InstrTable["LDXRW"] = a64Enc{format: a64FExcl, op: 0x885f7c00}
a64InstrTable["LDAXR"] = a64Enc{format: a64FExcl, op: 0xc85ffc00}
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
a64InstrTable["STXRW"] = a64Enc{format: a64FExcl, op: 0x88007c00}
a64InstrTable["STLXR"] = a64Enc{format: a64FExcl, op: 0xc800fc00}
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
// ---- LSE atomics ----
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDB"] = a64Enc{format: a64FLSE, op: 0<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDH"] = a64Enc{format: a64FLSE, op: 1<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["CASD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x45<<21 | 0x1f<<10}
a64InstrTable["CASW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x45<<21 | 0x1f<<10}
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
// ---- SIMD basics ----
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
}
// ---- load/store helper tables ----
// a64LSType describes the load/store parameters for a MOV width mnemonic.
type a64LSType struct {
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
}
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
var a64LoadTable = map[string]a64LSType{
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
}
// a64StoreOpc returns the store opc for a given load type.
// For integer: store opc = 00 (the load opc bits cleared).
// For FP: store opc = 00 (same pattern).
func a64StoreOpc(t a64LSType) int {
if t.V == 1 {
return 0 // FP store
}
return 0 // integer store
}
// a64MovRegTable maps register-to-register MOV mnemonic expansions.
// The Go toolchain encodes MOV Rn, Rd as ORR Rn, ZR, Rd.
var a64MovRegTable = map[string]uint32{
"MOVD": 1<<31 | 1<<29 | 0x0a<<24, // ORR 64-bit
"MOVW": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVB": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit (byte move)
"MOVBU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVH": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVHU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
"MOVWU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit
}
// arm64RegClass discriminates integer (R), floating-point (F) registers for
// the MOV pseudo-instruction.
type arm64RegClass int
const (
arm64ClsNone arm64RegClass = iota
arm64ClsGR
arm64ClsFP
)
// arm64RegClassOf reports the register class of a register operand name.
func arm64RegClassOf(name string) arm64RegClass {
switch {
case name == "":
return arm64ClsNone
case len(name) >= 1 && name[0] == 'F':
return arm64ClsFP
default:
return arm64ClsGR
}
}
// arm64Movcon returns the shift (in units of 16 bits) at which a non-zero
// 16-bit chunk of v sits, or -1 if v cannot be represented as a single
// MOVZ/MOVN immediate. This is the Go toolchain's movcon function.
func arm64Movcon(v int64) int {
for s := 0; s < 64; s += 16 {
if (uint64(v) &^ (uint64(0xFFFF) << uint(s))) == 0 {
return s
}
}
return -1
}