Files
gasm-sdk/asm/arm64_assemble.go
T

1334 lines
43 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine
// code. Every instruction is 4 bytes; the MOV pseudo-instruction and the
// immediate-arithmetic forms expand to 2–4 instructions when the immediate
// does not fit, so the layout is computed in two passes (sizes, then encoding
// with resolved branch targets).
//
// The emitted bytes match the Go toolchain's arm64 assembler, which is the
// ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch
// encodings and the MOV immediate expansions all follow cmd/internal/obj/
// arm64's asmout cases.
func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := arm64ComputeFrame(t)
prologue := arm64Prologue(fi)
chain := arm64JumpChain(t)
resolve := func(name string) string {
if r, ok := chain[name]; ok {
return r
}
return name
}
var relocs []Reloc
var spadj []SpadjStep
// The prologue (3 instructions when a small frame, 4 for large)
// raises the SP delta by autosize.
if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: arm64PrologueSpadjPC(fi), Value: fi.autosize})
}
// Pass 1: label offsets from the instruction sizes.
offsets := map[string]int{}
pos := len(prologue)
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
pos += arm64InstrSize(s, fi)
}
}
// Pass 2: encode. Relocation offsets are recorded function-relative.
out := append([]byte(nil), prologue...)
pc := len(prologue)
preCount := len(relocs)
var lines []LineEntry
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve)
if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
}
for j := preCount; j < len(relocs); j++ {
relocs[j].Off += pc - len(prologue)
}
preCount = len(relocs)
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
// The RET's epilogue closes the frame: the SP delta returns to zero.
if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 {
epi := arm64ReturnEpilogueLen(fi)
spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0})
}
out = append(out, code...)
pc += len(code)
}
return out, offsets, relocs, lines, spadj, nil
}
// arm64JumpChain precomputes jump-to-jump folding: a label whose first
// instruction is an unconditional local jump redirects its own jumpers to
// the ultimate target. The Go toolchain chases these chains before it
// encodes branches, so matching its bytes requires the same redirection.
func arm64JumpChain(t *ast.Text) map[string]string {
leadsTo := map[string]string{}
for i, stmt := range t.Body {
l, ok := stmt.(*ast.Label)
if !ok {
continue
}
j := i + 1
for j < len(t.Body) {
if _, isLabel := t.Body[j].(*ast.Label); !isLabel {
break
}
j++
}
if j >= len(t.Body) {
continue
}
in, ok := t.Body[j].(*ast.Instr)
if !ok {
continue
}
mnem := strings.ToUpper(in.Mnemonic.Text)
if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 {
continue
}
if name, ok := arm64LabelOK(in.Operands[0]); ok {
leadsTo[l.Name.Text] = name
}
}
chain := map[string]string{}
for name := range leadsTo {
visited := map[string]bool{name: true}
cur := name
for {
next, ok := leadsTo[cur]
if !ok || visited[next] {
break
}
visited[next] = true
cur = next
}
if cur != name {
chain[name] = cur
}
}
return chain
}
// arm64LabelOK returns the local label name of a jump operand.
func arm64LabelOK(op *ast.Operand) (string, bool) {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
return "", false
}
// arm64InstrSize returns the encoded size of an instruction: 4 bytes for
// most, more for the multi-instruction expansions.
func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo) int {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
if mnem == "RET" {
return len(arm64Return(fi))
}
switch mnem {
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
"FMOVS", "FMOVD":
return arm64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "SUB", "SUBW", "AND", "ANDW", "ORR", "ORRW", "EOR", "EORW":
if len(ops) >= 2 && isImmOperand(ops[0]) {
v := immFromOperand(ops[0])
// Small immediate (0..4095 or -2048..-1) fits in one instruction.
if v >= 0 && v <= 0xFFF {
return 4
}
if v >= -2048 && v < 0 {
return 4
}
// Larger immediates need MOV materialisation + op.
return 8
}
}
return 4
}
// encodeARM64Instr encodes a single AArch64 instruction.
func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
// Pseudo-instructions and special cases first.
switch mnem {
case "RET":
return arm64Return(fi), nil
case "NOP", "NOOP":
return a64wordLE(a64NOP), nil
case "UNDEF":
return a64wordLE(a64BRK(0)), nil
case "WORD":
if len(ops) != 1 {
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return a64wordLE(uint32(immFromOperand(ops[0]))), nil
case "B":
return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve)
case "BL", "CALL":
return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve)
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
"FMOVS", "FMOVD":
return encodeARM64Mov(instr, mnem, fi, relocs)
}
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBranchCond {
return encodeARM64BranchCond(mnem, enc.op, ops, pc, offsets, resolve)
}
// ADD/SUB immediate.
if mnem == "ADD" || mnem == "ADDW" || mnem == "SUB" || mnem == "SUBW" ||
mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
if len(ops) >= 2 && isImmOperand(ops[0]) {
return encodeARM64AddSubImm(mnem, ops)
}
}
// Register-register data processing.
// ASR/LSL/LSR/ROR with immediate operands use bitfield encoding (SBFM/UBFM).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPSR {
isShift := mnem == "ASR" || mnem == "ASRW" || mnem == "LSL" || mnem == "LSLW" ||
mnem == "LSR" || mnem == "LSRW" || mnem == "ROR" || mnem == "RORW"
if isShift && len(ops) >= 2 && isImmOperand(ops[0]) {
return encodeARM64Bitfield(mnem, enc.op, ops)
}
return encodeARM64DPSR(mnem, enc.op, ops)
}
// FP 3-operand (Rm, Rn, Rd).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFP3 {
return encodeARM64FP3(mnem, enc.op, ops)
}
// FP unary (Rn, Rd).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPUnary {
return encodeARM64FPUnary(mnem, enc.op, ops)
}
// FP 4-operand FMA (Ra, Rm, Rn, Rd).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFP4 {
return encodeARM64FP4(mnem, enc.op, ops)
}
// FP compare (Rm, Rn or #0, Rn).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCmp {
return encodeARM64FPCmp(mnem, enc.op, ops)
}
// FP conditional compare (Rm, Rn, #nzcv, cond).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCCmp {
return encodeARM64FPCCmp(mnem, enc.op, ops)
}
// FP conditional select (Rm, Rn, Rd, cond).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPSel {
return encodeARM64FPSel(mnem, enc.op, ops)
}
// FP ↔ integer conversion.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCvt {
return encodeARM64FPCvt(mnem, enc.op, ops)
}
// Conditional select (CSEL, CSINC, CSINV, CSNEG, CSET, CSETM, CINC, CINV, CNEG).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCSEL {
return encodeARM64CSEL(mnem, enc.op, ops)
}
// CRC32.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCRC32 {
return encodeARM64CRC32(mnem, enc.op, ops)
}
// Exclusive load/store (LDXR, STXR, LDAXR, STLXR).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl {
return encodeARM64Excl(mnem, enc.op, ops)
}
// LSE atomics (LDADD, CAS, SWP).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FLSE {
return encodeARM64LSEAtom(mnem, enc.op, ops)
}
// Bitfield/shift (ASR, LSL, LSR, ROR, BFI, BFXIL, SBFM, UBFM).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield {
return encodeARM64Bitfield(mnem, enc.op, ops)
}
// EXTR.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
return encodeARM64Extr(mnem, enc.op, ops)
}
// SIMD 3-operand (VADD, VSUB, VMUL).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMD3 {
return encodeARM64SIMD3(mnem, enc.op, ops)
}
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
}
// ---- branch encoding ----
// encodeARM64Branch encodes an unconditional branch (B/BL) to a label.
func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[string]int, link bool, relocs *[]Reloc, resolve func(string) string) ([]byte, error) {
if len(ops) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
}
op := ops[0]
// External symbol reference: BL sym(SB).
if link && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
if relocs != nil {
*relocs = append(*relocs, Reloc{
Off: 0,
After: 4,
Name: op.Addr.Sym.Name,
Addend: op.Addr.Sym.Offset,
Kind: RelArm64Branch,
})
}
// Emit BL with zero offset; the linker fills in the target.
return a64wordLE(a64Branch(1, 0)), nil
}
target := resolve(arm64Label(op))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q", target)
}
rel := (targetOff - pc) >> 2
if rel < -(1<<25) || rel >= (1<<25) {
return nil, fmt.Errorf("branch to %q too far (26-bit range)", target)
}
bop := uint32(0) // B
if link {
bop = 1 // BL
}
return a64wordLE(a64Branch(bop, int32(rel))), nil
}
// encodeARM64BranchCond encodes a conditional branch (B.cond) to a label.
func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
}
target := resolve(arm64Label(ops[0]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q", target)
}
rel := (targetOff - pc) >> 2
if rel < -(1<<18) || rel >= (1<<18) {
return nil, fmt.Errorf("branch to %q too far (19-bit range)", target)
}
// The condition code is in the low 4 bits of baseOp.
cond := baseOp & 0xF
return a64wordLE(a64BranchCond(int32(rel), cond)), nil
}
// ---- data-processing (shifted register) ----
// encodeARM64DPSR encodes a data-processing (shifted register) instruction.
// For most instructions: OP Rm, Rn, Rd (3 operands) or OP Rm, Rd (2 operands, Rn=Rd).
// For CMP/CMN/TST: CMP Rm, Rn (Rd=ZR).
// For NEG: NEG Rm, Rd (Rn=ZR).
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW"
switch len(ops) {
case 3:
// OP Rm, Rn, Rd
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rm < 0 || rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
case 2:
if isCmp {
// CMP Rm, Rn → SUBS XZR, Rn, Rm
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
if rm < 0 || rn < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | 31), nil
}
if isNeg {
// NEG Rm, Rd → SUB Rd, ZR, Rm
rm := arm64RegNum(operandRegName(ops[0]))
rd := arm64RegNum(operandRegName(ops[1]))
if rm < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | 31<<5 | uint32(rd)), nil
}
// OP Rm, Rd → OP Rm, Rd, Rd
rm := arm64RegNum(operandRegName(ops[0]))
rd := arm64RegNum(operandRegName(ops[1]))
if rm < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil
}
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
// ---- ADD/SUB immediate ----
// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction.
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
v := int32(immFromOperand(ops[0]))
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
rn := rd
if len(ops) == 3 {
rn = arm64RegNum(operandRegName(ops[1]))
}
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW"
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW"
sf := uint32(1) // 64-bit
if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" {
sf = 0 // 32-bit
}
if mnem == "CMP" || mnem == "CMPW" {
rd = 31 // ZR
}
if mnem == "CMN" || mnem == "CMNW" {
rd = 31 // ZR
}
op := uint32(0) // ADD
S := uint32(0)
if isSub {
op = 1
}
if isS {
S = 1
}
if v >= 0 && v <= 0xFFF {
return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil
}
if v >= -2048 && v < 0 {
// Encode as the opposite operation with positive immediate.
opp := op ^ 1
return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil
}
// Try with shift by 12.
if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 {
return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil
}
return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v)
}
// ---- MOV pseudo-instruction ----
// encodeARM64Mov encodes the MOV family — the load/store/immediate workhorse
// of Go's arm64 assembly. MOV is an alias of MOVD (the width mnemonics
// select the access width). The forms, mirroring the toolchain:
//
// MOVx $imm, rd load immediate (MOVZ/MOVN/MOVK)
// MOVx mem, rd load from memory
// MOVx rd, mem store to memory
// MOVx rs, rd register move (ORR Rd, ZR, Rs)
// MOVx $sym(SB), rd address of a static symbol (ADRP+ADD)
// MOVx sym(SB), rd load from a static symbol (ADRP+LDR)
// MOVx rd, sym(SB) store to a static symbol (ADRP+STR)
func encodeARM64Mov(instr *ast.Instr, mnem string, fi arm64FrameInfo, relocs *[]Reloc) ([]byte, error) {
ops := instr.Operands
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
src, dst := ops[0], ops[1]
// Immediate → register (including $sym(SB)).
if isImmOperand(src) && !isMemOperand(src) {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
rd := arm64RegNum(operandRegName(dst))
if rd < 0 {
return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem)
}
return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil
}
rd := arm64RegNum(operandRegName(dst))
if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
}
return encodeARM64LoadImm(rd, arm64Imm64(src), mnem)
}
// Static symbol load/store via ADRP.
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) {
rd := arm64RegNum(operandRegName(dst))
if rd < 0 {
return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem)
}
return encodeARM64SBLoad(src.Addr.Sym, rd, mnem, relocs)
}
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) {
rs := arm64RegNum(operandRegName(src))
if rs < 0 {
return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem)
}
return encodeARM64SBStore(dst.Addr.Sym, rs, mnem, relocs)
}
// Memory load/store with offset.
if isMemOperand(src) && !isMemOperand(dst) {
rd := arm64RegNum(operandRegName(dst))
if rd < 0 {
return nil, fmt.Errorf("%s: invalid destination register", mnem)
}
return encodeARM64MemOp(mnem, src, rd, true, fi)
}
if !isMemOperand(src) && isMemOperand(dst) {
rs := arm64RegNum(operandRegName(src))
if rs < 0 {
return nil, fmt.Errorf("%s: invalid source register", mnem)
}
return encodeARM64MemOp(mnem, dst, rs, false, fi)
}
// Register → register.
return encodeARM64RegMove(mnem, src, dst)
}
// arm64MovSize returns the encoded size of a MOV instruction.
func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
if len(ops) != 2 {
return 4
}
src, dst := ops[0], ops[1]
switch {
case isImmOperand(src):
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // ADRP + ADD
}
v := arm64Imm64(src)
if v == 0 {
return 4
}
if arm64Movcon(v) >= 0 || arm64Movcon(^v) >= 0 {
return 4
}
return 8 // MOVZ + MOVK
case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB":
return 8 // ADRP + LDR
case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB":
return 8 // ADRP + STR
case isMemOperand(src) || isMemOperand(dst):
mem := src
if !isMemOperand(src) {
mem = dst
}
_, off := arm64MemWithFrame(mem, fi)
// Scaled unsigned offset fits if aligned and in range.
lt := a64LoadTable[mnem]
if lt.size == 0 {
lt.size = 3 // default to64-bit for MOV
}
scale := int32(1) << uint(lt.size)
if off >= 0 && off%scale == 0 && off/scale < 4096 {
return 4
}
if off >= -256 && off <= 255 {
return 4 // unscaled
}
return 12 // materialise offset + LDR/STR
default:
return 4 // register move
}
}
// encodeARM64LoadImm loads an immediate into a register, matching the
// toolchain's MOVZ/MOVN/MOVK sequence.
func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
d := v
// For 32-bit MOVW, zero-extend.
if mnem == "MOVW" || mnem == "MOVWU" {
d = int64(uint32(v))
}
if d == 0 {
// ORR Rd, ZR, ZR (MOV $0, Rd)
op := uint32(1<<31 | 1<<29 | 0x0a<<24) // ORR 64-bit
if mnem == "MOVW" || mnem == "MOVWU" {
op = 0<<31 | 1<<29 | 0x0a<<24 // ORR 32-bit
}
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
}
sf := uint32(1) // 64-bit
if mnem == "MOVW" || mnem == "MOVWU" {
sf = 0
}
// The Go toolchain classifies immediates:
// - C_ABCON0 (0 < v ≤ 4095): bitmask first for positive values
// - Negative values: MOVN first, then bitmask
// - C_MOVCON (movcon-eligible, outside ABCON range): MOVZ/MOVN first
tryBitmaskFirst := (d > 0 && d <= 0xFFF)
if tryBitmaskFirst {
// Small immediate: try bitmask first (Go uses ORR for values like $1, $256).
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
if ok {
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
}
}
// Try MOVZ (single non-zero 16-bit chunk).
s := arm64Movcon(d)
if s >= 0 {
return a64wordLE(a64MoveWide(sf, 2, uint32(s>>4), uint32((d>>uint(s))&0xFFFF), uint32(rd))), nil
}
// Try MOVN (single non-0xFFFF 16-bit chunk of ^d).
sn := arm64Movcon(^d)
if sn >= 0 {
return a64wordLE(a64MoveWide(sf, 0, uint32(sn>>4), uint32((^d>>uint(sn))&0xFFFF), uint32(rd))), nil
}
// For values outside the bitmask-first range that are not movcon: try bitmask.
if !tryBitmaskFirst {
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
if ok {
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
}
}
// Multi-instruction: MOVZ + MOVK for each non-zero16-bit chunk.
var ws []uint32
first := true
for i := range 4 {
chunk := (d >> uint(i*16)) & 0xFFFF
if chunk == 0 {
continue
}
if first {
ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ
first = false
} else {
ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK
}
}
if len(ws) == 0 {
op := uint32(1<<31 | 1<<29 | 0x0a<<24)
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
}
return a64WordsLE(ws...), nil
}
// arm64Bitmask checks whether a value can be encoded as an AArch64 logical
// immediate (bitmask). Returns the N, immr, imms fields and true if
// representable. sf is 0 for 32-bit or 1 for 64-bit.
func arm64Bitmask(v uint64, sf int) (N, immr, imms uint32, ok bool) {
if v == 0 {
return
}
maxElem := uint(6) // 2^6 = 64
if sf == 0 {
maxElem = 5 // 2^5 = 32
v &= 0xFFFFFFFF
}
for e := uint(0); e < maxElem; e++ {
esize := uint(1) << (e + 1) // 2, 4, 8, 16, 32, 64
emask := uint64(1<<esize) - 1
pattern := v & emask
if pattern == 0 {
continue
}
// Check each rotation: is the rotated pattern a contiguous block of 1s at the LSB?
for r := range esize {
rotated := (pattern >> r) | ((pattern << (esize - r)) & emask)
if rotated == 0 {
continue
}
// Count trailing 1s (contiguous block of 1s from bit 0).
tz := uint(0)
tmp := ^rotated
for tmp&1 == 0 && tz < esize {
tz++
tmp >>= 1
}
if tz == 0 || tz >= esize {
continue
}
mask := uint64(1<<tz) - 1
if rotated != mask {
continue
}
ones := tz
// Verify the pattern repeats to fill the register.
full := uint64(0)
for i := uint(0); i < 64/esize; i++ {
full |= pattern << (i * esize)
}
if sf == 0 {
full &= 0xFFFFFFFF
}
if full != v {
continue
}
// Encode N, immr, imms.
if esize == 64 && sf == 1 {
N = 1
} else {
N = 0
}
imms = uint32((^(esize - 1))&0x3F) | uint32(ones-1)
immr = uint32((esize - r) % esize)
return N, immr, imms, true
}
}
return
}
// encodeARM64RegMove encodes a register-to-register move.
// Integer → integer: ORR Rd, ZR, Rs.
// FP → FP: FMOV Fd, Fn (FP data processing).
// FP ↔ GP: FMOV general (FPCVTI encoding).
// Go Plan 9 syntax: MOV dst, src (first operand = destination).
func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
rs := arm64RegNum(operandRegName(src))
rd := arm64RegNum(operandRegName(dst))
if rs < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
sc := arm64RegClassOf(operandRegName(src))
dc := arm64RegClassOf(operandRegName(dst))
// FP → FP: FMOV Fd, Fn (FP data processing unary form).
if sc == arm64ClsFP && dc == arm64ClsFP {
typ := uint32(1) // 64-bit double
if mnem == "FMOVS" {
typ = 0 // 32-bit float
}
// FPOP1S encoding: 0x1E204000 | type<<22 | Rn<<5 | Rd
return a64wordLE(0x1E<<24 | typ<<22 | 1<<21 | 0x10<<10 | uint32(rs)<<5 | uint32(rd)), nil
}
// GP ↔ FP: FMOV general (FPCVTI encoding).
// Go syntax: FMOV FPdst, GPsrc or FMOV GPdst, FPsrc.
// First operand = destination, second = source.
if sc == arm64ClsFP && dc == arm64ClsGR {
// FP → GP: FMOV Wd/Xd, Sn/Dn. opcode bits[20:16]=6.
sf, typ := uint32(0), uint32(0)
if mnem == "FMOVD" {
sf, typ = 1, 1
}
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 6<<16 | uint32(rs)<<5 | uint32(rd)), nil
}
if sc == arm64ClsGR && dc == arm64ClsFP {
// GP → FP: FMOV Vd, Wn/Xn. opcode bits[20:16]=7.
sf, typ := uint32(0), uint32(0)
if mnem == "FMOVD" {
sf, typ = 1, 1
}
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
}
// Integer → integer: ORR Rd, ZR, Rs.
sf := uint32(1) // 64-bit
if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" ||
mnem == "MOVH" || mnem == "MOVHU" {
sf = 0
}
op := uint32(1<<29 | 0x0a<<24) // ORR
return a64wordLE(sf<<31 | op | uint32(rs)<<16 | 31<<5 | uint32(rd)), nil
}
// encodeARM64MemOp encodes a memory load or store with offset.
func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm64FrameInfo) ([]byte, error) {
rn, off := arm64MemWithFrame(mem, fi)
if rn < 0 {
return nil, fmt.Errorf("invalid memory operand")
}
lt, ok := a64LoadTable[mnem]
if !ok {
// MOV defaults to MOVD (64-bit load/store).
lt = a64LoadTable["MOVD"]
}
scale := int32(1) << uint(lt.size)
if load {
// Try scaled unsigned offset first.
if off >= 0 && off%scale == 0 {
imm12 := uint32(off / scale)
if imm12 < 4096 {
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), imm12, uint32(rn), uint32(reg))), nil
}
}
// Try unscaled (9-bit signed).
if off >= -256 && off <= 255 {
return a64wordLE(a64LSUnscaled(lt.size, lt.V, lt.opc, off, rn, reg)), nil
}
// Large offset: materialise in R20 (TMP) and use register-offset.
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
}
// Store: same encoding but opc bits indicate store.
storeOpc := a64StoreOpc(lt)
if off >= 0 && off%scale == 0 {
imm12 := uint32(off / scale)
if imm12 < 4096 {
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), imm12, uint32(rn), uint32(reg))), nil
}
}
if off >= -256 && off <= 255 {
return a64wordLE(a64LSUnscaled(lt.size, lt.V, storeOpc, off, rn, reg)), nil
}
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
}
// ---- static symbol references (ADRP + offset) ----
// encodeARM64SBAddr emits ADRP Rd, 0; ADD Rd, Rd, 0 with the
// R_ADDRARM64 relocation pair, loading a symbol's address.
func encodeARM64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
)
}
return a64WordsLE(
a64ADR(1, 0, 0, uint32(rd)), // ADRP Rd, 0
a64AddSub(1, 0, 0, 0, 0, uint32(rd), uint32(rd)), // ADD $0, Rd, Rd
)
}
// encodeARM64SBLoad emits ADRP R20, 0; LDR Rd, [R20, 0] with relocations.
func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([]byte, error) {
lt, ok := a64LoadTable[mnem]
if !ok {
lt = a64LoadTable["MOVD"]
}
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
)
}
return a64WordsLE(
a64ADR(1, 0, 0, 20), // ADRP R20, 0
a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 20, uint32(rd)), // LDR Rd, [R20, #0]
), nil
}
// encodeARM64SBStore emits ADRP R20, 0; STR Rs, [R20, 0] with relocations.
func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ([]byte, error) {
lt, ok := a64LoadTable[mnem]
if !ok {
lt = a64LoadTable["MOVD"]
}
storeOpc := a64StoreOpc(lt)
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
)
}
return a64WordsLE(
a64ADR(1, 0, 0, 20), // ADRP R20, 0
a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 20, uint32(rs)), // STR Rs, [R20, #0]
), nil
}
// ---- operand helpers ----
// arm64Imm64 returns the full 64-bit immediate value of an operand.
func arm64Imm64(op *ast.Operand) int64 {
if op.Imm.HasVal {
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return v
}
return 0
}
// arm64MemWithFrame resolves a memory operand, translating FP/SP pseudo-
// registers via the frame mapping.
func arm64MemWithFrame(op *ast.Operand, fi arm64FrameInfo) (rn int, off int32) {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
return arm64ResolvePseudo(op.Addr.Sym, fi)
}
return arm64RegNum(op.Addr.Base), int32(op.Addr.Offset)
}
// arm64Label returns the label name of an operand.
func arm64Label(op *ast.Operand) string {
if op.Addr.Sym != nil {
return op.Addr.Sym.Name
}
return op.Raw
}
// ---- FP instruction encoding ----
// encodeARM64FP3 encodes a FP 3-operand instruction (Rm, Rn, Rd).
// FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL.
func encodeARM64FP3(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rm < 0 || rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64FPUnary encodes a FP unary instruction (Rn, Rd).
// FMOV reg-reg, FABS, FNEG, FSQRT, FCVT cross-precision, FRINT*.
func encodeARM64FPUnary(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rn := arm64RegNum(operandRegName(ops[0]))
rd := arm64RegNum(operandRegName(ops[1]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64FP4 encodes a FP 4-operand FMA instruction (Ra, Rm, Rn, Rd).
// FMADD, FMSUB, FNMADD, FNMSUB.
func encodeARM64FP4(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
var ra, rm, rn, rd int
switch len(ops) {
case 4:
ra = arm64RegNum(operandRegName(ops[0]))
rm = arm64RegNum(operandRegName(ops[1]))
rn = arm64RegNum(operandRegName(ops[2]))
rd = arm64RegNum(operandRegName(ops[3]))
case 3:
// 3-operand form: Fa, Fm, Fd → Fd = Fa ± Fd*Fm (Rn = Rd)
ra = arm64RegNum(operandRegName(ops[0]))
rm = arm64RegNum(operandRegName(ops[1]))
rd = arm64RegNum(operandRegName(ops[2]))
rn = rd
default:
return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops))
}
if ra < 0 || rm < 0 || rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(ra)<<16 | uint32(rm)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64FPCmp encodes a FP compare instruction.
// Go assembler syntax: FCMP Fn, Fm (register) or FCMP $0.0, Fn (compare with zero).
// ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5].
// Go puts first operand → Rm, second → Rn.
func encodeARM64FPCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
// Check if first operand is #0 (compare with zero): FCMP $0.0, Fn.
if isImmOperand(ops[0]) && immFromOperand(ops[0]) == 0 {
rn := arm64RegNum(operandRegName(ops[1]))
if rn < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
// For compare with zero: Rm=0, op2 bit 3 set (|= 8).
return a64wordLE((baseOp | 8) | 0<<16 | uint32(rn)<<5), nil
}
// Register compare: FCMP Fn, Fm.
// Go puts first operand in Rm field, second in Rn field.
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
if rm < 0 || rn < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5), nil
}
// encodeARM64FPCCmp encodes a FP conditional compare.
// Go assembler syntax: FCCMP cond, Fn, Fm, $nzcv
// ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5].
// Go puts ops[1] in Rm field, ops[2] in Rn field.
func encodeARM64FPCCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
condName := operandRegName(ops[0])
cond, ok := arm64CondMap[condName]
if !ok {
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
}
// Go puts ops[1] in Rm (bits 20:16), ops[2] in Rn (bits 9:5).
rm := arm64RegNum(operandRegName(ops[1]))
rn := arm64RegNum(operandRegName(ops[2]))
if rm < 0 || rn < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
nzcv := uint32(immFromOperand(ops[3]))
return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | nzcv&0xF), nil
}
// encodeARM64FPSel encodes a FP conditional select.
// Go assembler syntax: FCSEL cond, Fn, Fm, Fd
func encodeARM64FPSel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
// Operand order: cond, Fn, Fm, Fd
condName := operandRegName(ops[0])
cond, ok := arm64CondMap[condName]
if !ok {
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
}
rn := arm64RegNum(operandRegName(ops[1]))
rm := arm64RegNum(operandRegName(ops[2]))
rd := arm64RegNum(operandRegName(ops[3]))
if rn < 0 || rm < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64FPCvt encodes a FP ↔ integer conversion instruction.
// The operand order depends on direction: FCVTZS Fd, Rn (FP→int) or SCVTF Rd, Fn (int→FP).
func encodeARM64FPCvt(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
src := arm64RegNum(operandRegName(ops[0]))
dst := arm64RegNum(operandRegName(ops[1]))
if src < 0 || dst < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(src)<<5 | uint32(dst)), nil
}
// encodeARM64CSEL encodes a conditional select instruction.
// CSEL Rm, Rn, Rd, cond (4 operands) or CSET Rd, cond (2 operands).
func encodeARM64CSEL(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
isAlias := mnem == "CSET" || mnem == "CSETW" || mnem == "CSETM" || mnem == "CSETMW" ||
mnem == "CINC" || mnem == "CINCW" || mnem == "CINV" || mnem == "CINVW" ||
mnem == "CNEG" || mnem == "CNEGW"
if isAlias {
is2op := mnem == "CSET" || mnem == "CSETW" || mnem == "CSETM" || mnem == "CSETMW"
if is2op {
// CSET cond, Rd → CSEL XZR, XZR, Rd, inverted_cond
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
condName := operandRegName(ops[0])
cond, ok := arm64CondMap[condName]
if !ok {
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
}
rd := arm64RegNum(operandRegName(ops[1]))
if rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
invCond := cond ^ 1
return a64wordLE(baseOp | 31<<16 | invCond<<12 | 31<<5 | uint32(rd)), nil
}
// CINC cond, Rn, Rd → CSINC Rn, Rn, Rd, inverted_cond
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
condName := operandRegName(ops[0])
cond, ok := arm64CondMap[condName]
if !ok {
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
}
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
invCond := cond ^ 1
return a64wordLE(baseOp | uint32(rn)<<16 | invCond<<12 | uint32(rn)<<5 | uint32(rd)), nil
}
// CSEL cond, Rn, Rm, Rd (4 operands) — condition first.
// Go assembler syntax: CSEL cond, Rn, Rm, Rd
// ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5], Rd in bits[4:0].
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
condName := operandRegName(ops[0])
cond, ok := arm64CondMap[condName]
if !ok {
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
}
rn := arm64RegNum(operandRegName(ops[1]))
rm := arm64RegNum(operandRegName(ops[2]))
rd := arm64RegNum(operandRegName(ops[3]))
if rn < 0 || rm < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64CRC32 encodes a CRC32 instruction.
// Go assembler syntax: CRC32B Rm, Rd (2 operands, Rn=Rd).
func encodeARM64CRC32(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) == 3 {
// 3-operand form: CRC32B Rm, Rn, Rd → use Rm and Rd, Rn=Rd.
rm := arm64RegNum(operandRegName(ops[0]))
rd := arm64RegNum(operandRegName(ops[2]))
if rm < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil
}
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
rm := arm64RegNum(operandRegName(ops[0]))
rd := arm64RegNum(operandRegName(ops[1]))
if rm < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil
}
// ---- Atomics encoding ----
// encodeARM64Excl encodes an exclusive load/store instruction.
// LDXR (Rn), Rt → LDXR Rt, [Rn] (2 operands: mem, reg or reg, mem)
// STXR Rs, (Rn), Rt → STXR Rs, Rt, [Rn] (3 operands: Rs, mem, Rt-status)
func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
// LDXR/STXR have different operand forms.
isLoad := strings.HasPrefix(mnem, "LD")
if isLoad {
// LDXR (Rn), Rt → 2 operands: mem, reg
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rn, _ := arm64MemWithFrame(ops[0], arm64FrameInfo{})
rt := arm64RegNum(operandRegName(ops[1]))
if rn < 0 || rt < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil
}
// STXR Rs, (Rn), Rt → 3 operands: Rs, mem, Rt
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rs := arm64RegNum(operandRegName(ops[0]))
rn, _ := arm64MemWithFrame(ops[1], arm64FrameInfo{})
rt := arm64RegNum(operandRegName(ops[2]))
if rs < 0 || rn < 0 || rt < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
}
// encodeARM64LSEAtom encodes an LSE atomic instruction (LDADD, CAS, SWP).
// LDADD Rs, (Rn), Rt → 3 operands: Rs, mem, Rt
// CAS Rs, (Rn), Rt → 3 operands: Rs, mem, Rt
func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rs := arm64RegNum(operandRegName(ops[0]))
rn, _ := arm64MemWithFrame(ops[1], arm64FrameInfo{})
rt := arm64RegNum(operandRegName(ops[2]))
if rs < 0 || rn < 0 || rt < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
}
// ---- Bitfield/EXTR encoding ----
// encodeARM64Bitfield encodes a bitfield instruction.
// ASR/LSL/LSR/ROR $shamt, Rn, Rd → 3 operands: $imm, Rn, Rd
// BFI/BFXIL/SBFM/UBFM $immr, Rn, $imms, Rd → 4 operands
func encodeARM64Bitfield(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
isShift := mnem == "ASR" || mnem == "ASRW" || mnem == "LSL" || mnem == "LSLW" ||
mnem == "LSR" || mnem == "LSRW" || mnem == "ROR" || mnem == "RORW"
if isShift {
// ASR $shamt, Rn, Rd → SBFM with immr=shamt, imms=31/63
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
shamt := int(immFromOperand(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
// ASR: SBFM with immr=shamt, imms=31(32-bit) or 63(64-bit)
is64 := mnem == "ASR"
imms := 31
if is64 {
imms = 63
}
return a64wordLE(baseOp | uint32(shamt)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// BFI/BFXIL/SBFM/UBFM: 4 operands ($immr, Rn, $imms, Rd)
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
immr := int(immFromOperand(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
imms := int(immFromOperand(ops[2]))
rd := arm64RegNum(operandRegName(ops[3]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64Extr encodes an EXTR instruction.
// EXTR $lsb, Rm, Rn, Rd → 4 operands
func encodeARM64Extr(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
lsb := int(immFromOperand(ops[0]))
rm := arm64RegNum(operandRegName(ops[1]))
rn := arm64RegNum(operandRegName(ops[2]))
rd := arm64RegNum(operandRegName(ops[3]))
if rm < 0 || rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(lsb)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// ---- SIMD/NEON encoding ----
// encodeARM64SIMD3 encodes a SIMD 3-operand instruction.
// VADD Vm, Vn, Vd → base | Rm<<16 | Rn<<5 | Rd (Q and size bits in base)
func encodeARM64SIMD3(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rm < 0 || rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
// AssembleFileARM64 assembles every TEXT function of a parsed arm64 file
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
// code. SB references in the code are encoded as ADRP pairs with zero
// immediates; the object-file emitters record R_ADDRARM64 relocations for
// the linker.
func AssembleFileARM64(f *ast.File) (*Image, error) {
dataSyms, err := collectData(f)
if err != nil {
return nil, err
}
img := &Image{Symbols: map[string]int{}}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
continue
}
code, labels, relocs, lines, spadj, err := assembleARM64(t)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
fl := FuncLayout{
Name: t.Name.Name,
Pkg: t.Name.Pkg,
Static: t.Name.Static,
Offset: len(img.Code),
Size: len(code),
Frame: frameSize(t),
Args: argsSize(t),
Line: t.Pos().Line,
Labels: labels,
Lines: lines,
Spadj: spadj,
Relocs: relocs,
}
for _, f := range t.Flags {
switch f {
case "NOSPLIT":
fl.NoSplit = true
case "SPWRITE":
fl.SPWrite = true
}
}
img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...)
}
// Lay out the data section behind the code, 16-aligned.
dataStart := len(img.Code)
for _, d := range dataSyms {
pos := dataStart + len(img.Data)
for pos%16 != 0 {
img.Data = append(img.Data, 0)
pos++
}
img.Symbols[d.name] = pos
img.Data = append(img.Data, d.buf...)
img.DataSyms = append(img.DataSyms, DataSymbol{
Name: d.name,
Pkg: d.pkg,
Offset: len(img.Data) - len(d.buf),
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Dupok: d.dupok,
})
}
markExternals(img, dataSyms)
return img, nil
}