// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "fmt" "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" ) // assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine // code. Every instruction is 4 bytes; the MOV pseudo-instruction and the // immediate-arithmetic forms expand to 2–4 instructions when the immediate // does not fit, so the layout is computed in two passes (sizes, then encoding // with resolved branch targets). // // The emitted bytes match the Go toolchain's arm64 assembler, which is the // ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch // encodings and the MOV immediate expansions all follow cmd/internal/obj/ // arm64's asmout cases. func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { fi := arm64ComputeFrame(t) prologue := arm64Prologue(fi) chain := arm64JumpChain(t) resolve := func(name string) string { if r, ok := chain[name]; ok { return r } return name } var relocs []Reloc var spadj []SpadjStep // The prologue (3 instructions when a small frame, 4 for large) // raises the SP delta by autosize. if fi.autosize != 0 { spadj = append(spadj, SpadjStep{PC: arm64PrologueSpadjPC(fi), Value: fi.autosize}) } // Pass 1: label offsets from the instruction sizes. offsets := map[string]int{} pos := len(prologue) for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: offsets[s.Name.Text] = pos case *ast.Instr: pos += arm64InstrSize(s, fi) } } // Pass 2: encode. Relocation offsets are recorded function-relative. out := append([]byte(nil), prologue...) pc := len(prologue) preCount := len(relocs) var lines []LineEntry for _, stmt := range t.Body { in, ok := stmt.(*ast.Instr) if !ok { continue } code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve) if err != nil { return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) } for j := preCount; j < len(relocs); j++ { relocs[j].Off += pc - len(prologue) } preCount = len(relocs) lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line}) // The RET's epilogue closes the frame: the SP delta returns to zero. if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 { epi := arm64ReturnEpilogueLen(fi) spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0}) } out = append(out, code...) pc += len(code) } return out, offsets, relocs, lines, spadj, nil } // arm64JumpChain precomputes jump-to-jump folding: a label whose first // instruction is an unconditional local jump redirects its own jumpers to // the ultimate target. The Go toolchain chases these chains before it // encodes branches, so matching its bytes requires the same redirection. func arm64JumpChain(t *ast.Text) map[string]string { leadsTo := map[string]string{} for i, stmt := range t.Body { l, ok := stmt.(*ast.Label) if !ok { continue } j := i + 1 for j < len(t.Body) { if _, isLabel := t.Body[j].(*ast.Label); !isLabel { break } j++ } if j >= len(t.Body) { continue } in, ok := t.Body[j].(*ast.Instr) if !ok { continue } mnem := strings.ToUpper(in.Mnemonic.Text) if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 { continue } if name, ok := arm64LabelOK(in.Operands[0]); ok { leadsTo[l.Name.Text] = name } } chain := map[string]string{} for name := range leadsTo { visited := map[string]bool{name: true} cur := name for { next, ok := leadsTo[cur] if !ok || visited[next] { break } visited[next] = true cur = next } if cur != name { chain[name] = cur } } return chain } // arm64LabelOK returns the local label name of a jump operand. func arm64LabelOK(op *ast.Operand) (string, bool) { if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Sym.Name != "" { return op.Addr.Sym.Name, true } return "", false } // arm64InstrSize returns the encoded size of an instruction: 4 bytes for // most, more for the multi-instruction expansions. func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo) int { mnem := strings.ToUpper(instr.Mnemonic.Text) ops := instr.Operands if mnem == "RET" { return len(arm64Return(fi)) } switch mnem { case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", "FMOVS", "FMOVD": return arm64MovSize(mnem, ops, fi) case "ADD", "ADDW", "SUB", "SUBW", "AND", "ANDW", "ORR", "ORRW", "EOR", "EORW": if len(ops) >= 2 && isImmOperand(ops[0]) { v := immFromOperand(ops[0]) // Small immediate (0..4095 or -2048..-1) fits in one instruction. if v >= 0 && v <= 0xFFF { return 4 } if v >= -2048 && v < 0 { return 4 } // Larger immediates need MOV materialisation + op. return 8 } } return 4 } // encodeARM64Instr encodes a single AArch64 instruction. func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) { mnem := strings.ToUpper(instr.Mnemonic.Text) ops := instr.Operands // Pseudo-instructions and special cases first. switch mnem { case "RET": return arm64Return(fi), nil case "NOP", "NOOP": return a64wordLE(a64NOP), nil case "UNDEF": return a64wordLE(a64BRK(0)), nil case "WORD": if len(ops) != 1 { return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) } return a64wordLE(uint32(immFromOperand(ops[0]))), nil case "B": return encodeARM64Branch(mnem, ops, pc, offsets, false, resolve) case "BL", "CALL": return encodeARM64Branch(mnem, ops, pc, offsets, true, resolve) case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", "FMOVS", "FMOVD": return encodeARM64Mov(instr, mnem, fi, relocs) } // Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBranchCond { return encodeARM64BranchCond(mnem, enc.op, ops, pc, offsets, resolve) } // ADD/SUB immediate. if mnem == "ADD" || mnem == "ADDW" || mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" { if len(ops) >= 2 && isImmOperand(ops[0]) { return encodeARM64AddSubImm(mnem, ops) } } // Register-register data processing. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPSR { return encodeARM64DPSR(mnem, enc.op, ops) } // FP 3-operand (Rm, Rn, Rd). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFP3 { return encodeARM64FP3(mnem, enc.op, ops) } // FP unary (Rn, Rd). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPUnary { return encodeARM64FPUnary(mnem, enc.op, ops) } // FP 4-operand FMA (Ra, Rm, Rn, Rd). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFP4 { return encodeARM64FP4(mnem, enc.op, ops) } // FP compare (Rm, Rn or #0, Rn). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCmp { return encodeARM64FPCmp(mnem, enc.op, ops) } // FP conditional compare (Rm, Rn, #nzcv, cond). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCCmp { return encodeARM64FPCCmp(mnem, enc.op, ops) } // FP conditional select (Rm, Rn, Rd, cond). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPSel { return encodeARM64FPSel(mnem, enc.op, ops) } // FP ↔ integer conversion. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCvt { return encodeARM64FPCvt(mnem, enc.op, ops) } // Conditional select (CSEL, CSINC, CSINV, CSNEG, CSET, CSETM, CINC, CINV, CNEG). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCSEL { return encodeARM64CSEL(mnem, enc.op, ops) } // CRC32. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCRC32 { return encodeARM64CRC32(mnem, enc.op, ops) } // Exclusive load/store (LDXR, STXR, LDAXR, STLXR). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl { return encodeARM64Excl(mnem, enc.op, ops) } // LSE atomics (LDADD, CAS, SWP). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FLSE { return encodeARM64LSEAtom(mnem, enc.op, ops) } // Bitfield/shift (ASR, LSL, LSR, ROR, BFI, BFXIL, SBFM, UBFM). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield { return encodeARM64Bitfield(mnem, enc.op, ops) } // EXTR. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR { return encodeARM64Extr(mnem, enc.op, ops) } // SIMD 3-operand (VADD, VSUB, VMUL). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMD3 { return encodeARM64SIMD3(mnem, enc.op, ops) } return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) } // ---- branch encoding ---- // encodeARM64Branch encodes an unconditional branch (B/BL) to a label. func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[string]int, link bool, resolve func(string) string) ([]byte, error) { if len(ops) != 1 { return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) } op := ops[0] // External symbol reference: BL sym(SB). if link && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" { // Emit BL with zero offset; the linker fills in the target. return a64wordLE(a64Branch(1, 0)), nil } target := resolve(arm64Label(op)) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q", target) } rel := (targetOff - pc) >> 2 if rel < -(1<<25) || rel >= (1<<25) { return nil, fmt.Errorf("branch to %q too far (26-bit range)", target) } bop := uint32(0) // B if link { bop = 1 // BL } return a64wordLE(a64Branch(bop, int32(rel))), nil } // encodeARM64BranchCond encodes a conditional branch (B.cond) to a label. func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { if len(ops) != 1 { return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) } target := resolve(arm64Label(ops[0])) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q", target) } rel := (targetOff - pc) >> 2 if rel < -(1<<18) || rel >= (1<<18) { return nil, fmt.Errorf("branch to %q too far (19-bit range)", target) } // The condition code is in the low 4 bits of baseOp. cond := baseOp & 0xF return a64wordLE(a64BranchCond(int32(rel), cond)), nil } // ---- data-processing (shifted register) ---- // encodeARM64DPSR encodes a data-processing (shifted register) instruction. // For most instructions: OP Rm, Rn, Rd (3 operands) or OP Rm, Rd (2 operands, Rn=Rd). // For CMP/CMN/TST: CMP Rm, Rn (Rd=ZR). // For NEG: NEG Rm, Rd (Rn=ZR). func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW" isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW" switch len(ops) { case 3: // OP Rm, Rn, Rd rm := arm64RegNum(operandRegName(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[2])) if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil case 2: if isCmp { // CMP Rm, Rn → SUBS XZR, Rn, Rm rm := arm64RegNum(operandRegName(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) if rm < 0 || rn < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | 31), nil } if isNeg { // NEG Rm, Rd → SUB Rd, ZR, Rm rm := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[1])) if rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | 31<<5 | uint32(rd)), nil } // OP Rm, Rd → OP Rm, Rd, Rd rm := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[1])) if rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil } return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } // ---- ADD/SUB immediate ---- // encodeARM64AddSubImm encodes an ADD/SUB immediate instruction. func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } v := int32(immFromOperand(ops[0])) rd := arm64RegNum(operandRegName(ops[len(ops)-1])) rn := rd if len(ops) == 3 { rn = arm64RegNum(operandRegName(ops[1])) } if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" sf := uint32(1) // 64-bit if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" { sf = 0 // 32-bit } if mnem == "CMP" || mnem == "CMPW" { rd = 31 // ZR } if mnem == "CMN" || mnem == "CMNW" { rd = 31 // ZR } op := uint32(0) // ADD S := uint32(0) if isSub { op = 1 } if isS { S = 1 } if v >= 0 && v <= 0xFFF { return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil } if v >= -2048 && v < 0 { // Encode as the opposite operation with positive immediate. opp := op ^ 1 return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil } // Try with shift by 12. if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 { return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil } return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v) } // ---- MOV pseudo-instruction ---- // encodeARM64Mov encodes the MOV family — the load/store/immediate workhorse // of Go's arm64 assembly. MOV is an alias of MOVD (the width mnemonics // select the access width). The forms, mirroring the toolchain: // // MOVx $imm, rd load immediate (MOVZ/MOVN/MOVK) // MOVx mem, rd load from memory // MOVx rd, mem store to memory // MOVx rs, rd register move (ORR Rd, ZR, Rs) // MOVx $sym(SB), rd address of a static symbol (ADRP+ADD) // MOVx sym(SB), rd load from a static symbol (ADRP+LDR) // MOVx rd, sym(SB) store to a static symbol (ADRP+STR) func encodeARM64Mov(instr *ast.Instr, mnem string, fi arm64FrameInfo, relocs *[]Reloc) ([]byte, error) { ops := instr.Operands if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } src, dst := ops[0], ops[1] // Immediate → register (including $sym(SB)). if isImmOperand(src) && !isMemOperand(src) { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem) } return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil } rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) } return encodeARM64LoadImm(rd, arm64Imm64(src), mnem) } // Static symbol load/store via ADRP. if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) { rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem) } return encodeARM64SBLoad(src.Addr.Sym, rd, mnem, relocs) } if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) { rs := arm64RegNum(operandRegName(src)) if rs < 0 { return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem) } return encodeARM64SBStore(dst.Addr.Sym, rs, mnem, relocs) } // Memory load/store with offset. if isMemOperand(src) && !isMemOperand(dst) { rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s: invalid destination register", mnem) } return encodeARM64MemOp(mnem, src, rd, true, fi) } if !isMemOperand(src) && isMemOperand(dst) { rs := arm64RegNum(operandRegName(src)) if rs < 0 { return nil, fmt.Errorf("%s: invalid source register", mnem) } return encodeARM64MemOp(mnem, dst, rs, false, fi) } // Register → register. return encodeARM64RegMove(mnem, src, dst) } // arm64MovSize returns the encoded size of a MOV instruction. func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { if len(ops) != 2 { return 4 } src, dst := ops[0], ops[1] switch { case isImmOperand(src): if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { return 8 // ADRP + ADD } v := arm64Imm64(src) if v == 0 { return 4 } if arm64Movcon(v) >= 0 || arm64Movcon(^v) >= 0 { return 4 } return 8 // MOVZ + MOVK case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB": return 8 // ADRP + LDR case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB": return 8 // ADRP + STR case isMemOperand(src) || isMemOperand(dst): mem := src if !isMemOperand(src) { mem = dst } _, off := arm64MemWithFrame(mem, fi) // Scaled unsigned offset fits if aligned and in range. lt := a64LoadTable[mnem] if lt.size == 0 { lt.size = 3 // default to64-bit for MOV } scale := int32(1) << uint(lt.size) if off >= 0 && off%scale == 0 && off/scale < 4096 { return 4 } if off >= -256 && off <= 255 { return 4 // unscaled } return 12 // materialise offset + LDR/STR default: return 4 // register move } } // encodeARM64LoadImm loads an immediate into a register, matching the // toolchain's MOVZ/MOVN/MOVK sequence. func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) { d := v // For 32-bit MOVW, zero-extend. if mnem == "MOVW" || mnem == "MOVWU" { d = int64(uint32(v)) } if d == 0 { // ORR Rd, ZR, ZR (MOV $0, Rd) op := uint32(1<<31 | 1<<29 | 0x0a<<24) // ORR 64-bit if mnem == "MOVW" || mnem == "MOVWU" { op = 0<<31 | 1<<29 | 0x0a<<24 // ORR 32-bit } return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil } sf := uint32(1) // 64-bit if mnem == "MOVW" || mnem == "MOVWU" { sf = 0 } // The Go toolchain classifies immediates: // - C_ABCON0 (0 < v ≤ 4095): bitmask first for positive values // - Negative values: MOVN first, then bitmask // - C_MOVCON (movcon-eligible, outside ABCON range): MOVZ/MOVN first tryBitmaskFirst := (d > 0 && d <= 0xFFF) if tryBitmaskFirst { // Small immediate: try bitmask first (Go uses ORR for values like $1, $256). N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf)) if ok { return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil } } // Try MOVZ (single non-zero 16-bit chunk). s := arm64Movcon(d) if s >= 0 { return a64wordLE(a64MoveWide(sf, 2, uint32(s>>4), uint32((d>>uint(s))&0xFFFF), uint32(rd))), nil } // Try MOVN (single non-0xFFFF 16-bit chunk of ^d). sn := arm64Movcon(^d) if sn >= 0 { return a64wordLE(a64MoveWide(sf, 0, uint32(sn>>4), uint32((^d>>uint(sn))&0xFFFF), uint32(rd))), nil } // For values outside the bitmask-first range that are not movcon: try bitmask. if !tryBitmaskFirst { N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf)) if ok { return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil } } // Multi-instruction: MOVZ + MOVK for each non-zero16-bit chunk. var ws []uint32 first := true for i := 0; i < 4; i++ { chunk := (d >> uint(i*16)) & 0xFFFF if chunk == 0 { continue } if first { ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ first = false } else { ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK } } if len(ws) == 0 { op := uint32(1<<31 | 1<<29 | 0x0a<<24) return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil } return a64WordsLE(ws...), nil } // arm64Bitmask checks whether a value can be encoded as an AArch64 logical // immediate (bitmask). Returns the N, immr, imms fields and true if // representable. sf is 0 for 32-bit or 1 for 64-bit. func arm64Bitmask(v uint64, sf int) (N, immr, imms uint32, ok bool) { if v == 0 { return } maxElem := uint(6) // 2^6 = 64 if sf == 0 { maxElem = 5 // 2^5 = 32 v &= 0xFFFFFFFF } for e := uint(0); e < maxElem; e++ { esize := uint(1) << (e + 1) // 2, 4, 8, 16, 32, 64 emask := uint64(1<> r) | ((pattern << (esize - r)) & emask) if rotated == 0 { continue } // Count trailing 1s (contiguous block of 1s from bit 0). tz := uint(0) tmp := ^rotated for tmp&1 == 0 && tz < esize { tz++ tmp >>= 1 } if tz == 0 || tz >= esize { continue } mask := uint64(1<= 0 && off%scale == 0 { imm12 := uint32(off / scale) if imm12 < 4096 { return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), imm12, uint32(rn), uint32(reg))), nil } } // Try unscaled (9-bit signed). if off >= -256 && off <= 255 { return a64wordLE(a64LSUnscaled(lt.size, lt.V, lt.opc, off, rn, reg)), nil } // Large offset: materialise in R20 (TMP) and use register-offset. return nil, fmt.Errorf("%s: offset %d out of range", mnem, off) } // Store: same encoding but opc bits indicate store. storeOpc := a64StoreOpc(lt) if off >= 0 && off%scale == 0 { imm12 := uint32(off / scale) if imm12 < 4096 { return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), imm12, uint32(rn), uint32(reg))), nil } } if off >= -256 && off <= 255 { return a64wordLE(a64LSUnscaled(lt.size, lt.V, storeOpc, off, rn, reg)), nil } return nil, fmt.Errorf("%s: offset %d out of range", mnem, off) } // ---- static symbol references (ADRP + offset) ---- // encodeARM64SBAddr emits ADRP Rd, 0; ADD Rd, Rd, 0 with the // R_ADDRARM64 relocation pair, loading a symbol's address. func encodeARM64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, ) } return a64WordsLE( a64ADR(1, 0, 0, uint32(rd)), // ADRP Rd, 0 a64AddSub(1, 0, 0, 0, 0, uint32(rd), uint32(rd)), // ADD $0, Rd, Rd ) } // encodeARM64SBLoad emits ADRP R20, 0; LDR Rd, [R20, 0] with relocations. func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([]byte, error) { lt, ok := a64LoadTable[mnem] if !ok { lt = a64LoadTable["MOVD"] } if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, ) } return a64WordsLE( a64ADR(1, 0, 0, 20), // ADRP R20, 0 a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 20, uint32(rd)), // LDR Rd, [R20, #0] ), nil } // encodeARM64SBStore emits ADRP R20, 0; STR Rs, [R20, 0] with relocations. func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ([]byte, error) { lt, ok := a64LoadTable[mnem] if !ok { lt = a64LoadTable["MOVD"] } storeOpc := a64StoreOpc(lt) if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, ) } return a64WordsLE( a64ADR(1, 0, 0, 20), // ADRP R20, 0 a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 20, uint32(rs)), // STR Rs, [R20, #0] ), nil } // ---- operand helpers ---- // arm64Reg returns the register number of an operand, or -1. func arm64Reg(op *ast.Operand) int { return arm64RegNum(operandRegName(op)) } // arm64Imm64 returns the full 64-bit immediate value of an operand. func arm64Imm64(op *ast.Operand) int64 { if op.Imm.HasVal { v := op.Imm.Val if op.Imm.Neg { v = -v } return v } return 0 } // arm64MemWithFrame resolves a memory operand, translating FP/SP pseudo- // registers via the frame mapping. func arm64MemWithFrame(op *ast.Operand, fi arm64FrameInfo) (rn int, off int32) { if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { return arm64ResolvePseudo(op.Addr.Sym, fi) } return arm64RegNum(op.Addr.Base), int32(op.Addr.Offset) } // arm64Label returns the label name of an operand. func arm64Label(op *ast.Operand) string { if op.Addr.Sym != nil { return op.Addr.Sym.Name } return op.Raw } // ---- FP instruction encoding ---- // encodeARM64FP3 encodes a FP 3-operand instruction (Rm, Rn, Rd). // FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL. func encodeARM64FP3(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rm := arm64RegNum(operandRegName(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[2])) if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64FPUnary encodes a FP unary instruction (Rn, Rd). // FMOV reg-reg, FABS, FNEG, FSQRT, FCVT cross-precision, FRINT*. func encodeARM64FPUnary(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rn := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[1])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64FP4 encodes a FP 4-operand FMA instruction (Ra, Rm, Rn, Rd). // FMADD, FMSUB, FNMADD, FNMSUB. func encodeARM64FP4(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { var ra, rm, rn, rd int switch len(ops) { case 4: ra = arm64RegNum(operandRegName(ops[0])) rm = arm64RegNum(operandRegName(ops[1])) rn = arm64RegNum(operandRegName(ops[2])) rd = arm64RegNum(operandRegName(ops[3])) case 3: // 3-operand form: Fa, Fm, Fd → Fd = Fa ± Fd*Fm (Rn = Rd) ra = arm64RegNum(operandRegName(ops[0])) rm = arm64RegNum(operandRegName(ops[1])) rd = arm64RegNum(operandRegName(ops[2])) rn = rd default: return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops)) } if ra < 0 || rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(ra)<<16 | uint32(rm)<<10 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64FPCmp encodes a FP compare instruction. // Go assembler syntax: FCMP Fn, Fm (register) or FCMP $0.0, Fn (compare with zero). // ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5]. // Go puts first operand → Rm, second → Rn. func encodeARM64FPCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } // Check if first operand is #0 (compare with zero): FCMP $0.0, Fn. if isImmOperand(ops[0]) && immFromOperand(ops[0]) == 0 { rn := arm64RegNum(operandRegName(ops[1])) if rn < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } // For compare with zero: Rm=0, op2 bit 3 set (|= 8). return a64wordLE((baseOp | 8) | 0<<16 | uint32(rn)<<5), nil } // Register compare: FCMP Fn, Fm. // Go puts first operand in Rm field, second in Rn field. rm := arm64RegNum(operandRegName(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) if rm < 0 || rn < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5), nil } // encodeARM64FPCCmp encodes a FP conditional compare. // Go assembler syntax: FCCMP cond, Fn, Fm, $nzcv // ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5]. // Go puts ops[1] in Rm field, ops[2] in Rn field. func encodeARM64FPCCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } // Go puts ops[1] in Rm (bits 20:16), ops[2] in Rn (bits 9:5). rm := arm64RegNum(operandRegName(ops[1])) rn := arm64RegNum(operandRegName(ops[2])) if rm < 0 || rn < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } nzcv := uint32(immFromOperand(ops[3])) return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | nzcv&0xF), nil } // encodeARM64FPSel encodes a FP conditional select. // Go assembler syntax: FCSEL cond, Fn, Fm, Fd func encodeARM64FPSel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } // Operand order: cond, Fn, Fm, Fd condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } rn := arm64RegNum(operandRegName(ops[1])) rm := arm64RegNum(operandRegName(ops[2])) rd := arm64RegNum(operandRegName(ops[3])) if rn < 0 || rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64FPCvt encodes a FP ↔ integer conversion instruction. // The operand order depends on direction: FCVTZS Fd, Rn (FP→int) or SCVTF Rd, Fn (int→FP). func encodeARM64FPCvt(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } src := arm64RegNum(operandRegName(ops[0])) dst := arm64RegNum(operandRegName(ops[1])) if src < 0 || dst < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(src)<<5 | uint32(dst)), nil } // encodeARM64CSEL encodes a conditional select instruction. // CSEL Rm, Rn, Rd, cond (4 operands) or CSET Rd, cond (2 operands). func encodeARM64CSEL(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { isAlias := mnem == "CSET" || mnem == "CSETW" || mnem == "CSETM" || mnem == "CSETMW" || mnem == "CINC" || mnem == "CINCW" || mnem == "CINV" || mnem == "CINVW" || mnem == "CNEG" || mnem == "CNEGW" if isAlias { is2op := mnem == "CSET" || mnem == "CSETW" || mnem == "CSETM" || mnem == "CSETMW" if is2op { // CSET cond, Rd → CSEL XZR, XZR, Rd, inverted_cond if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } rd := arm64RegNum(operandRegName(ops[1])) if rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } invCond := cond ^ 1 return a64wordLE(baseOp | 31<<16 | invCond<<12 | 31<<5 | uint32(rd)), nil } // CINC cond, Rn, Rd → CSINC Rn, Rn, Rd, inverted_cond if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[2])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } invCond := cond ^ 1 return a64wordLE(baseOp | uint32(rn)<<16 | invCond<<12 | uint32(rn)<<5 | uint32(rd)), nil } // CSEL cond, Rn, Rm, Rd (4 operands) — condition first. // Go assembler syntax: CSEL cond, Rn, Rm, Rd // ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5], Rd in bits[4:0]. if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } rn := arm64RegNum(operandRegName(ops[1])) rm := arm64RegNum(operandRegName(ops[2])) rd := arm64RegNum(operandRegName(ops[3])) if rn < 0 || rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64CRC32 encodes a CRC32 instruction. // Go assembler syntax: CRC32B Rm, Rd (2 operands, Rn=Rd). func encodeARM64CRC32(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) == 3 { // 3-operand form: CRC32B Rm, Rn, Rd → use Rm and Rd, Rn=Rd. rm := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[2])) if rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil } if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } rm := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[1])) if rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil } // ---- Atomics encoding ---- // encodeARM64Excl encodes an exclusive load/store instruction. // LDXR (Rn), Rt → LDXR Rt, [Rn] (2 operands: mem, reg or reg, mem) // STXR Rs, (Rn), Rt → STXR Rs, Rt, [Rn] (3 operands: Rs, mem, Rt-status) func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { // LDXR/STXR have different operand forms. isLoad := strings.HasPrefix(mnem, "LD") if isLoad { // LDXR (Rn), Rt → 2 operands: mem, reg if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rn, _ := arm64MemWithFrame(ops[0], arm64FrameInfo{}) rt := arm64RegNum(operandRegName(ops[1])) if rn < 0 || rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil } // STXR Rs, (Rn), Rt → 3 operands: Rs, mem, Rt if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rs := arm64RegNum(operandRegName(ops[0])) rn, _ := arm64MemWithFrame(ops[1], arm64FrameInfo{}) rt := arm64RegNum(operandRegName(ops[2])) if rs < 0 || rn < 0 || rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil } // encodeARM64LSEAtom encodes an LSE atomic instruction (LDADD, CAS, SWP). // LDADD Rs, (Rn), Rt → 3 operands: Rs, mem, Rt // CAS Rs, (Rn), Rt → 3 operands: Rs, mem, Rt func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rs := arm64RegNum(operandRegName(ops[0])) rn, _ := arm64MemWithFrame(ops[1], arm64FrameInfo{}) rt := arm64RegNum(operandRegName(ops[2])) if rs < 0 || rn < 0 || rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil } // ---- Bitfield/EXTR encoding ---- // encodeARM64Bitfield encodes a bitfield instruction. // ASR/LSL/LSR/ROR $shamt, Rn, Rd → 3 operands: $imm, Rn, Rd // BFI/BFXIL/SBFM/UBFM $immr, Rn, $imms, Rd → 4 operands func encodeARM64Bitfield(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { isShift := mnem == "ASR" || mnem == "ASRW" || mnem == "LSL" || mnem == "LSLW" || mnem == "LSR" || mnem == "LSRW" || mnem == "ROR" || mnem == "RORW" if isShift { // ASR $shamt, Rn, Rd → SBFM with immr=shamt, imms=31/63 if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } shamt := int(immFromOperand(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[2])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } // ASR: SBFM with immr=shamt, imms=31(32-bit) or 63(64-bit) is64 := mnem == "ASR" imms := 31 if is64 { imms = 63 } return a64wordLE(baseOp | uint32(shamt)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil } // BFI/BFXIL/SBFM/UBFM: 4 operands ($immr, Rn, $imms, Rd) if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } immr := int(immFromOperand(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) imms := int(immFromOperand(ops[2])) rd := arm64RegNum(operandRegName(ops[3])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64Extr encodes an EXTR instruction. // EXTR $lsb, Rm, Rn, Rd → 4 operands func encodeARM64Extr(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } lsb := int(immFromOperand(ops[0])) rm := arm64RegNum(operandRegName(ops[1])) rn := arm64RegNum(operandRegName(ops[2])) rd := arm64RegNum(operandRegName(ops[3])) if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(lsb)<<10 | uint32(rn)<<5 | uint32(rd)), nil } // ---- SIMD/NEON encoding ---- // encodeARM64SIMD3 encodes a SIMD 3-operand instruction. // VADD Vm, Vn, Vd → base | Rm<<16 | Rn<<5 | Rd (Q and size bits in base) func encodeARM64SIMD3(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rm := arm64RegNum(operandRegName(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[2])) if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil } // AssembleFileARM64 assembles every TEXT function of a parsed arm64 file // and lays out its static symbols (GLOBL/DATA) in a data section behind the // code. SB references in the code are encoded as ADRP pairs with zero // immediates; the object-file emitters record R_ADDRARM64 relocations for // the linker. func AssembleFileARM64(f *ast.File) (*Image, error) { dataSyms, err := collectData(f) if err != nil { return nil, err } img := &Image{Symbols: map[string]int{}} for _, d := range f.Decls { t, ok := d.(*ast.Text) if !ok { continue } code, labels, relocs, lines, spadj, err := assembleARM64(t) if err != nil { return nil, fmt.Errorf("%s: %w", t.Name.Name, err) } fl := FuncLayout{ Name: t.Name.Name, Pkg: t.Name.Pkg, Static: t.Name.Static, Offset: len(img.Code), Size: len(code), Frame: frameSize(t), Args: argsSize(t), Line: t.Pos().Line, Labels: labels, Lines: lines, Spadj: spadj, Relocs: relocs, } for _, f := range t.Flags { switch f { case "NOSPLIT": fl.NoSplit = true case "SPWRITE": fl.SPWrite = true } } img.Funcs = append(img.Funcs, fl) img.Code = append(img.Code, code...) } // Lay out the data section behind the code, 16-aligned. dataStart := len(img.Code) for _, d := range dataSyms { pos := dataStart + len(img.Data) for pos%16 != 0 { img.Data = append(img.Data, 0) pos++ } img.Symbols[d.name] = pos img.Data = append(img.Data, d.buf...) img.DataSyms = append(img.DataSyms, DataSymbol{ Name: d.name, Pkg: d.pkg, Offset: len(img.Data) - len(d.buf), Size: d.size, Static: d.static, Rodata: d.rodata, Dupok: d.dupok, }) } return img, nil }