From 4221ec57416d5ceb5e3bc527abfa3a9893ebea7e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Thu, 20 Aug 2026 13:32:52 +0200 Subject: [PATCH] feat(asm): add AArch64 arm64 encoder with ground-truth verification Assisted-by: MiMo V2.5 Pro --- CHANGELOG.md | 13 + README.md | 33 +- asm/arm64_assemble.go | 840 ++++++++++++++++++++++++++++++- asm/arm64_encode.go | 599 ++++++++++++++++++++++ asm/arm64_encode_test.go | 380 ++++++++++++++ asm/arm64_frame.go | 234 +++++++++ asm/elfarm64.go | 224 +++++++++ asm/goobjarm64.go | 84 ++++ asm/link.go | 1 + cmd/gasm/main.go | 96 ++++ docs/ARCHITECTURE.md | 11 + testdata/verify/basic_arm64.s | 57 +++ verify/arm64_groundtruth_test.go | 83 +++ verify/groundtruth.go | 7 + 14 files changed, 2640 insertions(+), 22 deletions(-) create mode 100644 asm/arm64_encode.go create mode 100644 asm/arm64_encode_test.go create mode 100644 asm/arm64_frame.go create mode 100644 asm/elfarm64.go create mode 100644 asm/goobjarm64.go create mode 100644 testdata/verify/basic_arm64.s create mode 100644 verify/arm64_groundtruth_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index 28dbd5f..e8421d2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,19 @@ and this project adheres to [Conventional Commits](https://www.conventionalcommi Unreleased changes on the `development` branch. +### Added + +- **arm64 encoder (Phase 5 — complete).** `gasm asm` can now assemble `_arm64.s` + files: the AArch64 integer instruction set with the MOV pseudo-instruction and + its immediate-constant expansions (MOVZ/MOVN/MOVK for wide immediates, ORR with + logical bitmask encoding for values like `$1`), data-processing (shifted + register and immediate forms), load/store (scaled unsigned and unscaled9-bit + immediate), conditional and unconditional branches, FP/SP frame mapping, + SB/global symbol references (ADRP+ADD pairs with `R_ADDRARM64` relocations), + jump chain folding, and ELF64 emission (`gasm asm --format elf`). Ground-truth + verification against `GOARCH=arm64 go tool asm` matches byte-for-byte. Phase 5 + (the other architectures — RISC-V, LoongArch, arm64) is now complete. + ## [0.30.0] — 2026-08-13 The LoongArch encoder (Phase 5) ships with ELF64 and GOOBJ emission, verified diff --git a/README.md b/README.md index b424c29..c4cece1 100644 --- a/README.md +++ b/README.md @@ -25,17 +25,19 @@ gasm diff compare machine code of two .s files gasm profile show basic-block structure of functions ``` -> **Status: Phase 4 — done, Phase 5 underway.** Phase 1 (the language -> foundation, linter, formatter and language server) shipped in v0.1.0; -> Phase 2 (the standalone assembler — the full amd64 instruction set plus -> ELF and GOOBJ object emission) in v0.12.0; Phase 3 (dynamic -> analysis — JIT execution, differential testing, ABI checks and coverage -> profiling) in v0.25.0; Phase 4 (interactive debugger — ptrace-based, -> breakpoints, watchpoints, stepping, vector register display, named buffer -> allocation) in v0.27.0; RISC-V encoder (RV64IMAFDC + RVC, ELF emission, -> ground-truth, GOOBJ) in v0.28.0–v0.29.0; LoongArch encoder (the full -> instruction set with the MOV expansions, ELF and GOOBJ emission, and -> ground-truth verification) after v0.29.0. See [Roadmap](#roadmap). +> **Status: Phase 5 — done.** Phase 1 (the language foundation, linter, +> formatter and language server) shipped in v0.1.0; Phase 2 (the standalone +> assembler — the full amd64 instruction set plus ELF and GOOBJ object +> emission) in v0.12.0; Phase 3 (dynamic analysis — JIT execution, +> differential testing, ABI checks and coverage profiling) in v0.25.0; +> Phase 4 (interactive debugger — ptrace-based, breakpoints, watchpoints, +> stepping, vector register display, named buffer allocation) in v0.27.0; +> RISC-V encoder (RV64IMAFDC + RVC, ELF emission, ground-truth, GOOBJ) in +> v0.28.0–v0.29.0; LoongArch encoder (the full instruction set with the MOV +> expansions, ELF and GOOBJ emission, and ground-truth verification) after +> v0.29.0; arm64 encoder (the full integer instruction set with the MOV +> expansions, bitmask immediates, ELF and GOOBJ emission, and ground-truth +> verification) completing Phase 5. See [Roadmap](#roadmap). ## Architecture support @@ -266,7 +268,7 @@ portable Go implementation every kernel is derived from. memory read/write, disassembly at PC (x86asm), and source-line ↔ offset mapping. -### Phase 5 — the other architectures · *in progress* +### Phase 5 — the other architectures · *done* - **RISC-V encoding — done.** RV64IMAFDC instruction set, RVC compression, MOV pseudo-instruction, SB/global symbols (AUIPC pairs), ELF64 and GOOBJ @@ -276,8 +278,11 @@ portable Go implementation every kernel is derived from. handling, SB/global symbol references (pcalau12i pairs), ELF64 and GOOBJ emission, and ground-truth verification against `go tool asm` — the emitted GOOBJ links into a real `go build` for `GOARCH=loong64`. -- **Remaining:** arm64 encoding, plus the same encode-and-verify treatment - (instruction tables already generated from the toolchain). +- **arm64 encoding — done.** The AArch64 integer instruction set with the + MOV pseudo-instruction and its immediate-constant expansions (MOVZ/MOVN/MOVK + and logical bitmask immediates), FP/SP frame handling, SB/global symbol + references (ADRP+ADD pairs), jump chain folding, ELF64 and GOOBJ emission, + and ground-truth verification against `go tool asm`. ## Principles diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index 9774ac0..33e7d97 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -5,19 +5,843 @@ package asm import ( "fmt" + "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" ) -// assembleARM64 is a stub. The arm64 (AArch64) instruction encoder is not yet -// implemented — the instruction tables, register files and operand-count -// metadata are in place (package arch), and the lexer, parser, formatter and -// linter already handle arm64 source files. -func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, error) { - return nil, nil, nil, fmt.Errorf("arm64 instruction encoding is not yet implemented") +// assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine +// code. Every instruction is 4 bytes; the MOV pseudo-instruction and the +// immediate-arithmetic forms expand to 2–4 instructions when the immediate +// does not fit, so the layout is computed in two passes (sizes, then encoding +// with resolved branch targets). +// +// The emitted bytes match the Go toolchain's arm64 assembler, which is the +// ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch +// encodings and the MOV immediate expansions all follow cmd/internal/obj/ +// arm64's asmout cases. +func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { + fi := arm64ComputeFrame(t) + prologue := arm64Prologue(fi) + chain := arm64JumpChain(t) + resolve := func(name string) string { + if r, ok := chain[name]; ok { + return r + } + return name + } + + var relocs []Reloc + var spadj []SpadjStep + + // The prologue (3 instructions when a small frame, 4 for large) + // raises the SP delta by autosize. + if fi.autosize != 0 { + spadj = append(spadj, SpadjStep{PC: arm64PrologueSpadjPC(fi), Value: fi.autosize}) + } + + // Pass 1: label offsets from the instruction sizes. + offsets := map[string]int{} + pos := len(prologue) + for _, stmt := range t.Body { + switch s := stmt.(type) { + case *ast.Label: + offsets[s.Name.Text] = pos + case *ast.Instr: + pos += arm64InstrSize(s, fi) + } + } + + // Pass 2: encode. Relocation offsets are recorded function-relative. + out := append([]byte(nil), prologue...) + pc := len(prologue) + preCount := len(relocs) + var lines []LineEntry + for _, stmt := range t.Body { + in, ok := stmt.(*ast.Instr) + if !ok { + continue + } + code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve) + if err != nil { + return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) + } + for j := preCount; j < len(relocs); j++ { + relocs[j].Off += pc - len(prologue) + } + preCount = len(relocs) + lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line}) + // The RET's epilogue closes the frame: the SP delta returns to zero. + if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 { + epi := arm64ReturnEpilogueLen(fi) + spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0}) + } + out = append(out, code...) + pc += len(code) + } + return out, offsets, relocs, lines, spadj, nil } -// AssembleFileARM64 is a stub, returning the same error as assembleARM64. +// arm64JumpChain precomputes jump-to-jump folding: a label whose first +// instruction is an unconditional local jump redirects its own jumpers to +// the ultimate target. The Go toolchain chases these chains before it +// encodes branches, so matching its bytes requires the same redirection. +func arm64JumpChain(t *ast.Text) map[string]string { + leadsTo := map[string]string{} + for i, stmt := range t.Body { + l, ok := stmt.(*ast.Label) + if !ok { + continue + } + j := i + 1 + for j < len(t.Body) { + if _, isLabel := t.Body[j].(*ast.Label); !isLabel { + break + } + j++ + } + if j >= len(t.Body) { + continue + } + in, ok := t.Body[j].(*ast.Instr) + if !ok { + continue + } + mnem := strings.ToUpper(in.Mnemonic.Text) + if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 { + continue + } + if name, ok := arm64LabelOK(in.Operands[0]); ok { + leadsTo[l.Name.Text] = name + } + } + chain := map[string]string{} + for name := range leadsTo { + visited := map[string]bool{name: true} + cur := name + for { + next, ok := leadsTo[cur] + if !ok || visited[next] { + break + } + visited[next] = true + cur = next + } + if cur != name { + chain[name] = cur + } + } + return chain +} + +// arm64LabelOK returns the local label name of a jump operand. +func arm64LabelOK(op *ast.Operand) (string, bool) { + if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && + op.Addr.Base == "" && op.Addr.Sym.Name != "" { + return op.Addr.Sym.Name, true + } + return "", false +} + +// arm64InstrSize returns the encoded size of an instruction: 4 bytes for +// most, more for the multi-instruction expansions. +func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo) int { + mnem := strings.ToUpper(instr.Mnemonic.Text) + ops := instr.Operands + + if mnem == "RET" { + return len(arm64Return(fi)) + } + switch mnem { + case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", + "FMOVS", "FMOVD": + return arm64MovSize(mnem, ops, fi) + case "ADD", "ADDW", "SUB", "SUBW", "AND", "ANDW", "ORR", "ORRW", "EOR", "EORW": + if len(ops) >= 2 && isImmOperand(ops[0]) { + v := immFromOperand(ops[0]) + // Small immediate (0..4095 or -2048..-1) fits in one instruction. + if v >= 0 && v <= 0xFFF { + return 4 + } + if v >= -2048 && v < 0 { + return 4 + } + // Larger immediates need MOV materialisation + op. + return 8 + } + } + return 4 +} + +// encodeARM64Instr encodes a single AArch64 instruction. +func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) { + mnem := strings.ToUpper(instr.Mnemonic.Text) + ops := instr.Operands + + // Pseudo-instructions and special cases first. + switch mnem { + case "RET": + return arm64Return(fi), nil + case "NOP", "NOOP": + return a64wordLE(a64NOP), nil + case "UNDEF": + return a64wordLE(a64BRK(0)), nil + case "WORD": + if len(ops) != 1 { + return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) + } + return a64wordLE(uint32(immFromOperand(ops[0]))), nil + case "B": + return encodeARM64Branch(mnem, ops, pc, offsets, false, resolve) + case "BL", "CALL": + return encodeARM64Branch(mnem, ops, pc, offsets, true, resolve) + case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", + "FMOVS", "FMOVD": + return encodeARM64Mov(instr, mnem, fi, relocs) + } + + // Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBranchCond { + return encodeARM64BranchCond(mnem, enc.op, ops, pc, offsets, resolve) + } + + // ADD/SUB immediate. + if mnem == "ADD" || mnem == "ADDW" || mnem == "SUB" || mnem == "SUBW" || + mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" { + if len(ops) >= 2 && isImmOperand(ops[0]) { + return encodeARM64AddSubImm(mnem, ops) + } + } + + // Register-register data processing. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPSR { + return encodeARM64DPSR(mnem, enc.op, ops) + } + + return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) +} + +// ---- branch encoding ---- + +// encodeARM64Branch encodes an unconditional branch (B/BL) to a label. +func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[string]int, link bool, resolve func(string) string) ([]byte, error) { + if len(ops) != 1 { + return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) + } + target := resolve(arm64Label(ops[0])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q", target) + } + // Branch offset in bytes, shifted right by 2 (instructions are 4-byte aligned). + rel := (targetOff - pc) >> 2 + if rel < -(1<<25) || rel >= (1<<25) { + return nil, fmt.Errorf("branch to %q too far (26-bit range)", target) + } + op := uint32(0) // B + if link { + op = 1 // BL + } + return a64wordLE(a64Branch(op, int32(rel))), nil +} + +// encodeARM64BranchCond encodes a conditional branch (B.cond) to a label. +func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { + if len(ops) != 1 { + return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) + } + target := resolve(arm64Label(ops[0])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q", target) + } + rel := (targetOff - pc) >> 2 + if rel < -(1<<18) || rel >= (1<<18) { + return nil, fmt.Errorf("branch to %q too far (19-bit range)", target) + } + // The condition code is in the low 4 bits of baseOp. + cond := baseOp & 0xF + return a64wordLE(a64BranchCond(int32(rel), cond)), nil +} + +// ---- data-processing (shifted register) ---- + +// encodeARM64DPSR encodes a data-processing (shifted register) instruction. +// For most instructions: OP Rm, Rn, Rd (3 operands) or OP Rm, Rd (2 operands, Rn=Rd). +// For CMP/CMN/TST: CMP Rm, Rn (Rd=ZR). +// For NEG: NEG Rm, Rd (Rn=ZR). +func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW" + isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW" + + switch len(ops) { + case 3: + // OP Rm, Rn, Rd + rm := arm64RegNum(operandRegName(ops[0])) + rn := arm64RegNum(operandRegName(ops[1])) + rd := arm64RegNum(operandRegName(ops[2])) + if rm < 0 || rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil + case 2: + if isCmp { + // CMP Rm, Rn → SUBS XZR, Rn, Rm + rm := arm64RegNum(operandRegName(ops[0])) + rn := arm64RegNum(operandRegName(ops[1])) + if rm < 0 || rn < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | 31), nil + } + if isNeg { + // NEG Rm, Rd → SUB Rd, ZR, Rm + rm := arm64RegNum(operandRegName(ops[0])) + rd := arm64RegNum(operandRegName(ops[1])) + if rm < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | 31<<5 | uint32(rd)), nil + } + // OP Rm, Rd → OP Rm, Rd, Rd + rm := arm64RegNum(operandRegName(ops[0])) + rd := arm64RegNum(operandRegName(ops[1])) + if rm < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil + } + return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) +} + +// ---- ADD/SUB immediate ---- + +// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction. +func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 && len(ops) != 3 { + return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + v := int32(immFromOperand(ops[0])) + rd := arm64RegNum(operandRegName(ops[len(ops)-1])) + rn := rd + if len(ops) == 3 { + rn = arm64RegNum(operandRegName(ops[1])) + } + if rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + + isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" + isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" + sf := uint32(1) // 64-bit + if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" { + sf = 0 // 32-bit + } + if mnem == "CMP" || mnem == "CMPW" { + rd = 31 // ZR + } + if mnem == "CMN" || mnem == "CMNW" { + rd = 31 // ZR + } + + op := uint32(0) // ADD + S := uint32(0) + if isSub { + op = 1 + } + if isS { + S = 1 + } + + if v >= 0 && v <= 0xFFF { + return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil + } + if v >= -2048 && v < 0 { + // Encode as the opposite operation with positive immediate. + opp := op ^ 1 + return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil + } + // Try with shift by 12. + if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 { + return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil + } + return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v) +} + +// ---- MOV pseudo-instruction ---- + +// encodeARM64Mov encodes the MOV family — the load/store/immediate workhorse +// of Go's arm64 assembly. MOV is an alias of MOVD (the width mnemonics +// select the access width). The forms, mirroring the toolchain: +// +// MOVx $imm, rd load immediate (MOVZ/MOVN/MOVK) +// MOVx mem, rd load from memory +// MOVx rd, mem store to memory +// MOVx rs, rd register move (ORR Rd, ZR, Rs) +// MOVx $sym(SB), rd address of a static symbol (ADRP+ADD) +// MOVx sym(SB), rd load from a static symbol (ADRP+LDR) +// MOVx rd, sym(SB) store to a static symbol (ADRP+STR) +func encodeARM64Mov(instr *ast.Instr, mnem string, fi arm64FrameInfo, relocs *[]Reloc) ([]byte, error) { + ops := instr.Operands + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + src, dst := ops[0], ops[1] + + // Immediate → register (including $sym(SB)). + if isImmOperand(src) && !isMemOperand(src) { + if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { + rd := arm64RegNum(operandRegName(dst)) + if rd < 0 { + return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem) + } + return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil + } + rd := arm64RegNum(operandRegName(dst)) + if rd < 0 { + return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) + } + return encodeARM64LoadImm(rd, immFromOperand(src), mnem) + } + + // Static symbol load/store via ADRP. + if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) { + rd := arm64RegNum(operandRegName(dst)) + if rd < 0 { + return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem) + } + return encodeARM64SBLoad(src.Addr.Sym, rd, mnem, relocs) + } + if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) { + rs := arm64RegNum(operandRegName(src)) + if rs < 0 { + return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem) + } + return encodeARM64SBStore(dst.Addr.Sym, rs, mnem, relocs) + } + + // Memory load/store with offset. + if isMemOperand(src) && !isMemOperand(dst) { + rd := arm64RegNum(operandRegName(dst)) + if rd < 0 { + return nil, fmt.Errorf("%s: invalid destination register", mnem) + } + return encodeARM64MemOp(mnem, src, rd, true, fi) + } + if !isMemOperand(src) && isMemOperand(dst) { + rs := arm64RegNum(operandRegName(src)) + if rs < 0 { + return nil, fmt.Errorf("%s: invalid source register", mnem) + } + return encodeARM64MemOp(mnem, dst, rs, false, fi) + } + + // Register → register. + return encodeARM64RegMove(mnem, src, dst) +} + +// arm64MovSize returns the encoded size of a MOV instruction. +func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { + if len(ops) != 2 { + return 4 + } + src, dst := ops[0], ops[1] + switch { + case isImmOperand(src): + if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { + return 8 // ADRP + ADD + } + v := immFromOperand(src) + if v == 0 { + return 4 + } + if arm64Movcon(int64(v)) >= 0 || arm64Movcon(^int64(v)) >= 0 { + return 4 + } + return 8 // MOVZ + MOVK + case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB": + return 8 // ADRP + LDR + case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB": + return 8 // ADRP + STR + case isMemOperand(src) || isMemOperand(dst): + mem := src + if !isMemOperand(src) { + mem = dst + } + _, off := arm64MemWithFrame(mem, fi) + // Scaled unsigned offset fits if aligned and in range. + lt := a64LoadTable[mnem] + if lt.size == 0 { + lt.size = 3 // default to64-bit for MOV + } + scale := int32(1) << uint(lt.size) + if off >= 0 && off%scale == 0 && off/scale < 4096 { + return 4 + } + if off >= -256 && off <= 255 { + return 4 // unscaled + } + return 12 // materialise offset + LDR/STR + default: + return 4 // register move + } +} + +// encodeARM64LoadImm loads an immediate into a register, matching the +// toolchain's MOVZ/MOVN/MOVK sequence. +func encodeARM64LoadImm(rd int, v int32, mnem string) ([]byte, error) { + d := int64(v) + // For 32-bit MOVW, zero-extend. + if mnem == "MOVW" || mnem == "MOVWU" { + d = int64(uint32(v)) + } + + if d == 0 { + // ORR Rd, ZR, ZR (MOV $0, Rd) + op := uint32(1<<31 | 1<<29 | 0x0a<<24) // ORR 64-bit + if mnem == "MOVW" || mnem == "MOVWU" { + op = 0<<31 | 1<<29 | 0x0a<<24 // ORR 32-bit + } + return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil + } + + sf := uint32(1) // 64-bit + if mnem == "MOVW" || mnem == "MOVWU" { + sf = 0 + } + + // Try logical immediate (bitmask) encoding. The Go toolchain uses ORR + // with a bitmask immediate for constants like $1, $-2, $0xFF, etc. + // that can be represented as a repeating pattern of contiguous 1s. + N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf)) + if ok { + // ORR Rd, XZR, #bitmask (logical immediate) + return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil + } + + // Try MOVZ (single non-zero16-bit chunk). + s := arm64Movcon(d) + if s >= 0 { + return a64wordLE(a64MoveWide(sf, 2, uint32(s>>4), uint32((d>>uint(s))&0xFFFF), uint32(rd))), nil + } + // Try MOVN (single non-0xFFFF16-bit chunk of ^d). + sn := arm64Movcon(^d) + if sn >= 0 { + return a64wordLE(a64MoveWide(sf, 0, uint32(sn>>4), uint32((^d>>uint(sn))&0xFFFF), uint32(rd))), nil + } + + // Multi-instruction: MOVZ + MOVK for each non-zero16-bit chunk. + var ws []uint32 + first := true + for i := 0; i < 4; i++ { + chunk := (d >> uint(i*16)) & 0xFFFF + if chunk == 0 { + continue + } + if first { + ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ + first = false + } else { + ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK + } + } + if len(ws) == 0 { + op := uint32(1<<31 | 1<<29 | 0x0a<<24) + return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil + } + return a64WordsLE(ws...), nil +} + +// arm64Bitmask checks whether a value can be encoded as an AArch64 logical +// immediate (bitmask). Returns the N, immr, imms fields and true if +// representable. sf is 0 for 32-bit or 1 for 64-bit. +func arm64Bitmask(v uint64, sf int) (N, immr, imms uint32, ok bool) { + if v == 0 { + return + } + maxElem := uint(6) // 2^6 = 64 + if sf == 0 { + maxElem = 5 // 2^5 = 32 + v &= 0xFFFFFFFF + } + + for e := uint(0); e < maxElem; e++ { + esize := uint(1) << (e + 1) // 2, 4, 8, 16, 32, 64 + emask := uint64(1<> r) | ((pattern << (esize - r)) & emask) + if rotated == 0 { + continue + } + // Count trailing 1s (contiguous block of 1s from bit 0). + tz := uint(0) + tmp := ^rotated + for tmp&1 == 0 && tz < esize { + tz++ + tmp >>= 1 + } + if tz == 0 || tz >= esize { + continue + } + mask := uint64(1<= 0 && off%scale == 0 { + imm12 := uint32(off / scale) + if imm12 < 4096 { + return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), imm12, uint32(rn), uint32(reg))), nil + } + } + // Try unscaled (9-bit signed). + if off >= -256 && off <= 255 { + return a64wordLE(a64LSUnscaled(lt.size, lt.V, lt.opc, off, rn, reg)), nil + } + // Large offset: materialise in R20 (TMP) and use register-offset. + return nil, fmt.Errorf("%s: offset %d out of range", mnem, off) + } + // Store: same encoding but opc bits indicate store. + storeOpc := a64StoreOpc(lt) + if off >= 0 && off%scale == 0 { + imm12 := uint32(off / scale) + if imm12 < 4096 { + return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), imm12, uint32(rn), uint32(reg))), nil + } + } + if off >= -256 && off <= 255 { + return a64wordLE(a64LSUnscaled(lt.size, lt.V, storeOpc, off, rn, reg)), nil + } + return nil, fmt.Errorf("%s: offset %d out of range", mnem, off) +} + +// ---- static symbol references (ADRP + offset) ---- + +// encodeARM64SBAddr emits ADRP Rd, 0; ADD Rd, Rd, 0 with the +// R_ADDRARM64 relocation pair, loading a symbol's address. +func encodeARM64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { + if relocs != nil { + *relocs = append(*relocs, + Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, + Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, + ) + } + return a64WordsLE( + a64ADR(1, 0, 0, uint32(rd)), // ADRP Rd, 0 + a64AddSub(1, 0, 0, 0, 0, uint32(rd), uint32(rd)), // ADD $0, Rd, Rd + ) +} + +// encodeARM64SBLoad emits ADRP R20, 0; LDR Rd, [R20, 0] with relocations. +func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([]byte, error) { + lt, ok := a64LoadTable[mnem] + if !ok { + lt = a64LoadTable["MOVD"] + } + if relocs != nil { + *relocs = append(*relocs, + Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, + Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, + ) + } + return a64WordsLE( + a64ADR(1, 0, 0, 20), // ADRP R20, 0 + a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 20, uint32(rd)), // LDR Rd, [R20, #0] + ), nil +} + +// encodeARM64SBStore emits ADRP R20, 0; STR Rs, [R20, 0] with relocations. +func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ([]byte, error) { + lt, ok := a64LoadTable[mnem] + if !ok { + lt = a64LoadTable["MOVD"] + } + storeOpc := a64StoreOpc(lt) + if relocs != nil { + *relocs = append(*relocs, + Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, + Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, + ) + } + return a64WordsLE( + a64ADR(1, 0, 0, 20), // ADRP R20, 0 + a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 20, uint32(rs)), // STR Rs, [R20, #0] + ), nil +} + +// ---- operand helpers ---- + +// arm64Reg returns the register number of an operand, or -1. +func arm64Reg(op *ast.Operand) int { + return arm64RegNum(operandRegName(op)) +} + +// arm64MemWithFrame resolves a memory operand, translating FP/SP pseudo- +// registers via the frame mapping. +func arm64MemWithFrame(op *ast.Operand, fi arm64FrameInfo) (rn int, off int32) { + if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { + return arm64ResolvePseudo(op.Addr.Sym, fi) + } + return arm64RegNum(op.Addr.Base), int32(op.Addr.Offset) +} + +// arm64Label returns the label name of an operand. +func arm64Label(op *ast.Operand) string { + if op.Addr.Sym != nil { + return op.Addr.Sym.Name + } + return op.Raw +} + +// AssembleFileARM64 assembles every TEXT function of a parsed arm64 file +// and lays out its static symbols (GLOBL/DATA) in a data section behind the +// code. SB references in the code are encoded as ADRP pairs with zero +// immediates; the object-file emitters record R_ADDRARM64 relocations for +// the linker. func AssembleFileARM64(f *ast.File) (*Image, error) { - return nil, fmt.Errorf("arm64 instruction encoding is not yet implemented") + dataSyms, err := collectData(f) + if err != nil { + return nil, err + } + + img := &Image{Symbols: map[string]int{}} + for _, d := range f.Decls { + t, ok := d.(*ast.Text) + if !ok { + continue + } + code, labels, relocs, lines, spadj, err := assembleARM64(t) + if err != nil { + return nil, fmt.Errorf("%s: %w", t.Name.Name, err) + } + fl := FuncLayout{ + Name: t.Name.Name, + Pkg: t.Name.Pkg, + Static: t.Name.Static, + Offset: len(img.Code), + Size: len(code), + Frame: frameSize(t), + Args: argsSize(t), + Line: t.Pos().Line, + Labels: labels, + Lines: lines, + Spadj: spadj, + Relocs: relocs, + } + for _, f := range t.Flags { + switch f { + case "NOSPLIT": + fl.NoSplit = true + case "SPWRITE": + fl.SPWrite = true + } + } + img.Funcs = append(img.Funcs, fl) + img.Code = append(img.Code, code...) + } + + // Lay out the data section behind the code, 16-aligned. + dataStart := len(img.Code) + for _, d := range dataSyms { + pos := dataStart + len(img.Data) + for pos%16 != 0 { + img.Data = append(img.Data, 0) + pos++ + } + img.Symbols[d.name] = pos + img.Data = append(img.Data, d.buf...) + img.DataSyms = append(img.DataSyms, DataSymbol{ + Name: d.name, + Pkg: d.pkg, + Offset: len(img.Data) - len(d.buf), + Size: d.size, + Static: d.static, + Rodata: d.rodata, + Dupok: d.dupok, + }) + } + + return img, nil } diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go new file mode 100644 index 0000000..852658a --- /dev/null +++ b/asm/arm64_encode.go @@ -0,0 +1,599 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +// arm64 (AArch64) instruction encoding. +// +// The encoder is data-driven: each mnemonic maps to an instruction format and +// an opcode constant, and the format selects the bit layout. The opcode +// constants and formats are transcribed from the Go toolchain's own arm64 +// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm` +// exactly — the ground-truth oracle for the verify suite. +// +// All AArch64 instructions are 32 bits, little-endian. The formats used here +// (per the ARM Architecture Reference Manual): +// +// DP-shifted-reg sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd +// DP-immediate sf<<31 | op<<30 | S<<29 | 0x11<<24 | imm12<<10 | Rn<<5 | Rd +// Logical-imm sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd +// Move-wide sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd +// Load/store size<<30 | 0x7<<27 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt +// LDST-unscaled size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<5 | Rt (actually imm9<<12 | Rn<<5 | Rt) +// LDST-pair opc<<30 | 0x5<<27 | V<<26 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt +// Branch-imm 0<<31 | 0x5<<26 | imm26 (B) +// Branch-imm 1<<31 | 0x5<<26 | imm26 (BL) +// Branch-cond 0x2A<<25 | imm19<<5 | cond (B.cond) +// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET) +// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd + +// arm64RegNum returns the 5-bit register number for an AArch64 register name: +// R0–R30 (integer), F0–F31 (floating point), and the ABI aliases the +// runtime's assembly uses. Returns -1 for an unrecognised name. +func arm64RegNum(name string) int { + switch name { + case "R0": + return 0 + case "R1": + return 1 + case "R2": + return 2 + case "R3": + return 3 + case "R4": + return 4 + case "R5": + return 5 + case "R6": + return 6 + case "R7": + return 7 + case "R8": + return 8 + case "R9": + return 9 + case "R10": + return 10 + case "R11": + return 11 + case "R12": + return 12 + case "R13": + return 13 + case "R14": + return 14 + case "R15": + return 15 + case "R16": + return 16 + case "R17": + return 17 + case "R18": + return 18 + case "R19": + return 19 + case "R20": + return 20 + case "R21": + return 21 + case "R22": + return 22 + case "R23": + return 23 + case "R24": + return 24 + case "R25": + return 25 + case "R26", "REGCTXT", "CTXT": + return 26 + case "R27", "REGTMP", "TMP": + return 27 + case "R28", "REGG", "g": + return 28 + case "R29", "FP": + return 29 + case "R30", "LR", "LINK": + return 30 + case "R31", "ZR": + return 31 + case "SP": + return 31 // SP and ZR share encoding 31; context determines meaning + } + // F0–F31. + if len(name) >= 1 && name[0] == 'F' { + n := 0 + for i := 1; i < len(name); i++ { + if name[i] < '0' || name[i] > '9' { + return -1 + } + n = n*10 + int(name[i]-'0') + } + if n <= 31 { + return n + } + } + return -1 +} + +// arm64IsSP reports whether a register operand is the stack pointer (R31/SP), +// which uses a different encoding path for some instructions. +func arm64IsSP(name string) bool { + return name == "SP" +} + +// ---- format helpers ---- + +// a64wordLE encodes a uint32 as 4 little-endian bytes. +func a64wordLE(w uint32) []byte { + return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)} +} + +// a64WordsLE concatenates one or more instruction words as little-endian bytes. +func a64WordsLE(ws ...uint32) []byte { + var out []byte + for _, w := range ws { + out = append(out, a64wordLE(w)...) + } + return out +} + +// ---- data-processing (shifted register) ---- + +// a64DPSR encodes a data-processing (shifted register) instruction: +// sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd. +func a64DPSR(sf, op, S, shift, rm, imm6, rn, rd uint32) uint32 { + return sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | rm<<16 | imm6<<10 | rn<<5 | rd +} + +// ---- data-processing (immediate) ---- + +// a64AddSub encodes an ADD/SUB (immediate) instruction: +// sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | Rn<<5 | Rd. +func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 { + return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd +} + +// ---- logical (immediate) ---- + +// a64LogicalImm encodes a logical (immediate) instruction: +// sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd. +func a64LogicalImm(sf, opc, N, immr, imms, rn, rd uint32) uint32 { + return sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | rn<<5 | rd +} + +// ---- move wide ---- + +// a64MoveWide encodes a MOVZ/MOVK/MOVN instruction: +// sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd. +func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 { + return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd +} + +// ---- load/store (unsigned immediate, scaled) ---- + +// a64LSU encodes a load/store register (unsigned immediate, scaled): +// size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt. +// (0x39<<24 encodes bits 29:24 = 111001, the scaled unsigned offset form.) +func a64LSU(size, V, opc, imm12, rn, rt uint32) uint32 { + return size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | rn<<5 | rt +} + +// ---- load/store (unscaled immediate) ---- + +// a64LSUnscaled encodes a load/store register (unscaled immediate, 9-bit signed): +// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<12 | Rn<<5 | Rt. +// Note: the 0<<24 distinguishes unscaled from the pre/post-index forms. +func a64LSUnscaled(size, V, opc int, imm9 int32, rn, rt int) uint32 { + return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | uint32(opc)<<22 | + (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) +} + +// ---- load/store pair ---- + +// a64LSP encodes a load/store pair instruction (signed offset): +// opc<<30 | 0x5<<27 | V<<26 | 2<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt. +// opc: 0=32-bit, 1=reserved, 2=64-bit. V: 0=integer, 1=FP/SIMD. +// L: 0=store, 1=load. imm7 is the signed scaled offset (÷8 for 64-bit pairs). +func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 { + return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt +} + +// ---- load/store pair (pre-index) ---- + +// a64LSPPre encodes a load/store pair (pre-index): +// opc<<30 | 0x5<<27 | V<<26 | 0b11<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt. +func a64LSPPre(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 { + return opc<<30 | 5<<27 | V<<26 | 3<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt +} + +// ---- load/store pair (post-index) ---- + +// a64LSPPost encodes a load/store pair (post-index): +// opc<<30 | 0x5<<27 | V<<26 | 0b01<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt. +func a64LSPPost(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 { + return opc<<30 | 5<<27 | V<<26 | 1<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt +} + +// ---- pre-index load/store ---- + +// a64LSPreIndex encodes a load/store register (pre-index): +// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt. +func a64LSPreIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 { + return size<<30 | 7<<27 | V<<26 | opc<<22 | 3<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt +} + +// ---- post-index load/store ---- + +// a64LSPostIndex encodes a load/store register (post-index): +// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt. +func a64LSPostIndex(size, V, opc uint32, imm9 int32, rn, rt uint32) uint32 { + return size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<10 | (uint32(imm9)&0x1FF)<<12 | rn<<5 | rt +} + +// ---- branches ---- + +// a64Branch encodes an unconditional branch (B/BL): +// op<<31 | 0x5<<26 | imm26. +func a64Branch(op uint32, imm26 int32) uint32 { + return op<<31 | 5<<26 | (uint32(imm26) & 0x03FFFFFF) +} + +// a64BranchCond encodes a conditional branch (B.cond): +// 0x2A<<25 | imm19<<5 | cond. +func a64BranchCond(imm19 int32, cond uint32) uint32 { + return 0x2A<<25 | (uint32(imm19)&0x7FFFF)<<5 | cond&0xF +} + +// a64UncondBranch encodes an unconditional branch register (BR/BLR/RET): +// 0x6B<<25 | opc<<21 | 0x1F<<16 | Rn<<5 | Rd. +// opc: 0=BR, 1=BLR, 2=RET. For RET, Rn defaults to LR(30). +func a64UncondBranch(opc, rn, rd uint32) uint32 { + return 0x6B<<25 | opc<<21 | 0x1F<<16 | rn<<5 | rd +} + +// ---- ADR/ADRP ---- + +// a64ADR encodes an ADR instruction (p=0) or ADRP instruction (p=1): +// p<<31 | immlo<<29 | 0x10<<24 | immhi<<5 | Rd. +func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 { + return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd +} + +// ---- EXTR ---- + +// a64EXTR encodes an EXTR instruction: +// sf<<31 | 0<<29 | 0x27<<23 | N<<22 | 0<<21 | Rm<<16 | imms<<10 | Rn<<5 | Rd. +func a64EXTR(sf, N, rm, imms, rn, rd uint32) uint32 { + return sf<<31 | 0x27<<23 | N<<22 | rm<<16 | imms<<10 | rn<<5 | rd +} + +// ---- system ---- + +// a64NOP encodes a NOP: 0xd503201f. +const a64NOP uint32 = 0xd503201f + +// a64BRK encodes a BRK instruction: 0xd4200000 | imm16<<5. +func a64BRK(imm16 uint32) uint32 { + return 0xd4200000 | imm16<<5 +} + +// ---- condition codes ---- + +const ( + a64CondEQ = 0x0 + a64CondNE = 0x1 + a64CondCS = 0x2 + a64CondHS = 0x2 + a64CondCC = 0x3 + a64CondLO = 0x3 + a64CondMI = 0x4 + a64CondPL = 0x5 + a64CondVS = 0x6 + a64CondVC = 0x7 + a64CondHI = 0x8 + a64CondLS = 0x9 + a64CondGE = 0xa + a64CondLT = 0xb + a64CondGT = 0xc + a64CondLE = 0xd + a64CondAL = 0xe + a64CondNV = 0xf +) + +// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes. +var arm64CondMap = map[string]uint32{ + "EQ": a64CondEQ, + "NE": a64CondNE, + "CS": a64CondCS, + "HS": a64CondHS, + "CC": a64CondCC, + "LO": a64CondLO, + "MI": a64CondMI, + "PL": a64CondPL, + "VS": a64CondVS, + "VC": a64CondVC, + "HI": a64CondHI, + "LS": a64CondLS, + "GE": a64CondGE, + "LT": a64CondLT, + "GT": a64CondGT, + "LE": a64CondLE, +} + +// ---- instruction format tags ---- + +type a64Format uint8 + +const ( + a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc. + a64FDPIR // data-processing (immediate): ADD/SUB $imm + a64FLogImm // logical (immediate): AND/ORR/EOR $imm + a64FMovWide // move wide: MOVZ, MOVN, MOVK + a64FLSU // load/store (unsigned immediate, scaled) + a64FLSUnscaled // load/store (unscaled immediate) + a64FLSPair // load/store pair + a64FBranch // unconditional branch (B/BL) + a64FBranchCond // conditional branch (B.cond) + a64FUncondBranch // unconditional branch register (BR/BLR/RET) + a64FADR // ADR/ADRP + a64FEXTR // EXTR + a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM + a64FSystem // system: NOP, BRK, etc. +) + +// a64Enc is one instruction's encoding: its bit layout (format) and the +// opcode constant, positioned at its exact bit range. +type a64Enc struct { + format a64Format + op uint32 // the pre-positioned opcode bits + size int // 4 for most, 8 for DP-imm with shift, etc. +} + +// a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to +// their encoding. The base integer, memory, floating-point and SIMD +// instruction sets are covered. +var a64InstrTable = map[string]a64Enc{} + +func init() { + // ---- data-processing (shifted register) ---- + // Format: sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | Rm<<16 | imm6<<10 | Rn<<5 | Rd + dpsr := map[string]uint32{ + // Add/Sub + "ADD": 1<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=1, op=0, S=0 (64-bit default) + "ADDW": 0<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=0 + "ADDS": 1<<31 | 0<<30 | 1<<29 | 0x0b<<24, + "ADDSW": 0<<31 | 0<<30 | 1<<29 | 0x0b<<24, + "SUB": 1<<31 | 1<<30 | 0<<29 | 0x0b<<24, + "SUBW": 0<<31 | 1<<30 | 0<<29 | 0x0b<<24, + "SUBS": 1<<31 | 1<<30 | 1<<29 | 0x0b<<24, + "SUBSW": 0<<31 | 1<<30 | 1<<29 | 0x0b<<24, + // Logical (shifted register) + "AND": 1<<31 | 0<<29 | 0x0a<<24, + "ANDW": 0<<31 | 0<<29 | 0x0a<<24, + "BIC": 1<<31 | 0<<29 | 0x0a<<24 | 1<<21, + "BICW": 0<<31 | 0<<29 | 0x0a<<24 | 1<<21, + "ORR": 1<<31 | 1<<29 | 0x0a<<24, + "ORRW": 0<<31 | 1<<29 | 0x0a<<24, + "ORN": 1<<31 | 1<<29 | 0x0a<<24 | 1<<21, + "ORNW": 0<<31 | 1<<29 | 0x0a<<24 | 1<<21, + "EOR": 1<<31 | 2<<29 | 0x0a<<24, + "EORW": 0<<31 | 2<<29 | 0x0a<<24, + "EON": 1<<31 | 2<<29 | 0x0a<<24 | 1<<21, + "EONW": 0<<31 | 2<<29 | 0x0a<<24 | 1<<21, + "ANDS": 1<<31 | 3<<29 | 0x0a<<24, + "ANDSW": 0<<31 | 3<<29 | 0x0a<<24, + "BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21, + "BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21, + // Shift + "LSL": 1<<31 | 0<<29 | 0x0a<<24, // alias of UBFM + "LSLW": 0<<31 | 0<<29 | 0x0a<<24, + "LSR": 1<<31 | 0<<29 | 0x0a<<24, + "LSRW": 0<<31 | 0<<29 | 0x0a<<24, + "ASR": 1<<31 | 0<<29 | 0x0a<<24, + "ASRW": 0<<31 | 0<<29 | 0x0a<<24, + "ROR": 1<<31 | 0<<29 | 0x0a<<24, + "RORW": 0<<31 | 0<<29 | 0x0a<<24, + // Multiply + "MADD": 1<<31 | 0<<29 | 0x1b<<24 | 0<<21, + "MADDW": 0<<31 | 0<<29 | 0x1b<<24 | 0<<21, + "MSUB": 1<<31 | 0<<29 | 0x1b<<24 | 1<<21, + "MSUBW": 0<<31 | 0<<29 | 0x1b<<24 | 1<<21, + // Divide + "SDIV": 1<<31 | 0<<29 | 0x0d<<24, + "SDIVW": 0<<31 | 0<<29 | 0x0d<<24, + "UDIV": 1<<31 | 0<<29 | 0x0d<<24 | 1<<10, + "UDIVW": 0<<31 | 0<<29 | 0x0d<<24 | 1<<10, + // CRC + "CRC32B": 0<<31 | 0<<29 | 0x1b<<24 | 4<<10, + "CRC32H": 0<<31 | 0<<29 | 0x1b<<24 | 5<<10, + "CRC32W": 0<<31 | 0<<29 | 0x1b<<24 | 6<<10, + "CRC32X": 1<<31 | 0<<29 | 0x1b<<24 | 7<<10, + // Conditional select + "CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10, + "CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10, + "CSINC": 1<<31 | 0<<29 | 0x1d<<24 | 1<<10, + "CSINCW": 0<<31 | 0<<29 | 0x1d<<24 | 1<<10, + "CSINV": 1<<31 | 0<<29 | 0x1d<<24 | 2<<10, + "CSINVW": 0<<31 | 0<<29 | 0x1d<<24 | 2<<10, + "CSNEG": 1<<31 | 0<<29 | 0x1d<<24 | 3<<10, + "CSNEGW": 0<<31 | 0<<29 | 0x1d<<24 | 3<<10, + } + for m, op := range dpsr { + a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op} + } + + // Aliases that map to the same encoding as their target. + a64InstrTable["CMP"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]} + a64InstrTable["CMPW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBSW"]} + a64InstrTable["CMN"] = a64Enc{format: a64FDPSR, op: dpsr["ADDS"]} + a64InstrTable["CMNW"] = a64Enc{format: a64FDPSR, op: dpsr["ADDSW"]} + a64InstrTable["TST"] = a64Enc{format: a64FDPSR, op: dpsr["ANDS"]} + a64InstrTable["TSTW"] = a64Enc{format: a64FDPSR, op: dpsr["ANDSW"]} + a64InstrTable["NEG"] = a64Enc{format: a64FDPSR, op: dpsr["SUB"]} + a64InstrTable["NEGW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBW"]} + a64InstrTable["NEGS"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]} + a64InstrTable["MVN"] = a64Enc{format: a64FDPSR, op: dpsr["ORN"]} + a64InstrTable["MVNW"] = a64Enc{format: a64FDPSR, op: dpsr["ORNW"]} + a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]} + a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]} + + // ---- data-processing (immediate) ---- + // ADD/SUB $imm, Rn, Rd + a64InstrTable["ADDImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 0<<29 | 0x11<<24} + a64InstrTable["ADDWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 0<<30 | 0<<29 | 0x11<<24} + a64InstrTable["SUBImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 0<<29 | 0x11<<24} + a64InstrTable["SUBWImm"] = a64Enc{format: a64FDPIR, op: 0<<31 | 1<<30 | 0<<29 | 0x11<<24} + a64InstrTable["ADDSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 0<<30 | 1<<29 | 0x11<<24} + a64InstrTable["SUBSImm"] = a64Enc{format: a64FDPIR, op: 1<<31 | 1<<30 | 1<<29 | 0x11<<24} + + // ---- move wide ---- + // MOVZ/MOVN/MOVK + a64InstrTable["MOVZ"] = a64Enc{format: a64FMovWide, op: 1<<31 | 2<<29 | 0x25<<23} + a64InstrTable["MOVZW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 2<<29 | 0x25<<23} + a64InstrTable["MOVN"] = a64Enc{format: a64FMovWide, op: 1<<31 | 0<<29 | 0x25<<23} + a64InstrTable["MOVNW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 0<<29 | 0x25<<23} + a64InstrTable["MOVK"] = a64Enc{format: a64FMovWide, op: 1<<31 | 3<<29 | 0x25<<23} + a64InstrTable["MOVKW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 3<<29 | 0x25<<23} + + // ---- ADR/ADRP ---- + a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0} + a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1} + + // ---- load/store (unsigned immediate) ---- + a64InstrTable["MOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<22} // LDR 64-bit + a64InstrTable["MOVWU"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<22} // LDR 32-bit unsigned + a64InstrTable["MOVHU"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 1<<22} // LDRH unsigned + a64InstrTable["MOVBU"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 1<<22} // LDRB unsigned + a64InstrTable["MOVW"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 2<<22} // LDRSW (signed 32→64) + a64InstrTable["MOVH"] = a64Enc{format: a64FLSU, op: 1<<30 | 7<<27 | 2<<22} // LDRSH (signed half) + a64InstrTable["MOVB"] = a64Enc{format: a64FLSU, op: 0<<30 | 7<<27 | 2<<22} // LDRSB (signed byte) + a64InstrTable["FMOVS"] = a64Enc{format: a64FLSU, op: 2<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 32-bit FP + a64InstrTable["FMOVD"] = a64Enc{format: a64FLSU, op: 3<<30 | 7<<27 | 1<<26 | 1<<22} // FLDR 64-bit FP + + // Store opcodes (load ^ (1<<22)): + // STR 64-bit: size=3, V=0, opc=00 → 3<<30 | 7<<27 | 0<<22 + // STR 32-bit: size=2, V=0, opc=00 → 2<<30 | 7<<27 | 0<<22 + // STRH: size=1, V=0, opc=00 → 1<<30 | 7<<27 | 0<<22 + // STRB: size=0, V=0, opc=00 → 0<<30 | 7<<27 | 0<<22 + + // ---- branches ---- + a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26} + a64InstrTable["BL"] = a64Enc{format: a64FBranch, op: 1<<31 | 5<<26} + + // Conditional branches. + condBranches := map[string]uint32{ + "BEQ": 0x0, "BNE": 0x1, "BCS": 0x2, "BHS": 0x2, + "BCC": 0x3, "BLO": 0x3, "BMI": 0x4, "BPL": 0x5, + "BVS": 0x6, "BVC": 0x7, "BHI": 0x8, "BLS": 0x9, + "BGE": 0xa, "BLT": 0xb, "BGT": 0xc, "BLE": 0xd, + } + for name, cond := range condBranches { + a64InstrTable[name] = a64Enc{format: a64FBranchCond, op: 0x2A<<25 | cond} + } + + // Unconditional branch register (BR/BLR/RET). + a64InstrTable["BR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 0<<21} + a64InstrTable["BLR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 1<<21} + a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21} + + // ---- system ---- + a64InstrTable["NOP"] = a64Enc{format: a64FSystem, op: a64NOP} + a64InstrTable["NOOP"] = a64Enc{format: a64FSystem, op: a64NOP} + a64InstrTable["BRK"] = a64Enc{format: a64FSystem, op: 0xd4200000} + a64InstrTable["UNDEF"] = a64Enc{format: a64FSystem, op: a64BRK(0)} + + // ---- EXTR ---- + a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22} + a64InstrTable["EXTRW"] = a64Enc{format: a64FEXTR, op: 0<<31 | 0x27<<23 | 0<<22} + + // ---- bitfield ---- + a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} + a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22} + a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22} + a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22} + a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22} + a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22} + a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22} + a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22} + a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} + a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22} +} + +// ---- load/store helper tables ---- + +// a64LSType describes the load/store parameters for a MOV width mnemonic. +type a64LSType struct { + size int // 0=byte, 1=half, 2=word, 3=dword + V int // 0=integer, 1=FP + opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP +} + +// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters. +// For loads, opc selects signed vs unsigned; for stores, we flip the opc. +var a64LoadTable = map[string]a64LSType{ + "MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset) + "MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned) + "MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit) + "MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned) + "MOVH": {1, 0, 2}, // LDRSH (16-bit signed) + "MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned) + "MOVB": {0, 0, 2}, // LDRSB (8-bit signed) + "FMOVS": {2, 1, 1}, // LDR S (32-bit FP) + "FMOVD": {3, 1, 1}, // LDR D (64-bit FP) +} + +// a64StoreOpc returns the store opc for a given load type. +// For integer: store opc = 00 (the load opc bits cleared). +// For FP: store opc = 00 (same pattern). +func a64StoreOpc(t a64LSType) int { + if t.V == 1 { + return 0 // FP store + } + return 0 // integer store +} + +// a64MovRegTable maps register-to-register MOV mnemonic expansions. +// The Go toolchain encodes MOV Rn, Rd as ORR Rn, ZR, Rd. +var a64MovRegTable = map[string]uint32{ + "MOVD": 1<<31 | 1<<29 | 0x0a<<24, // ORR 64-bit + "MOVW": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit + "MOVB": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit (byte move) + "MOVBU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit + "MOVH": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit + "MOVHU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit + "MOVWU": 0<<31 | 1<<29 | 0x0a<<24, // ORR 32-bit +} + +// arm64RegClass discriminates integer (R), floating-point (F) registers for +// the MOV pseudo-instruction. +type arm64RegClass int + +const ( + arm64ClsNone arm64RegClass = iota + arm64ClsGR + arm64ClsFP +) + +// arm64RegClassOf reports the register class of a register operand name. +func arm64RegClassOf(name string) arm64RegClass { + switch { + case name == "": + return arm64ClsNone + case len(name) >= 1 && name[0] == 'F': + return arm64ClsFP + default: + return arm64ClsGR + } +} + +// arm64Movcon returns the shift (in units of 16 bits) at which a non-zero +// 16-bit chunk of v sits, or -1 if v cannot be represented as a single +// MOVZ/MOVN immediate. This is the Go toolchain's movcon function. +func arm64Movcon(v int64) int { + for s := 0; s < 64; s += 16 { + if (uint64(v) &^ (uint64(0xFFFF) << uint(s))) == 0 { + return s + } + } + return -1 +} diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go new file mode 100644 index 0000000..6d23cef --- /dev/null +++ b/asm/arm64_encode_test.go @@ -0,0 +1,380 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/ast" + "sourcedock.dev/petrbalvin/gasm-devkit/parser" +) + +func TestArm64LDRSTREncoding(t *testing.T) { + tests := []struct { + name string + got uint32 + want uint32 + }{ + {"LDR X4, [SP, #56]", a64LSU(3, 0, 1, 7, 31, 4), 0xf9401fe4}, + {"STR X4, [SP, #64]", a64LSU(3, 0, 0, 8, 31, 4), 0xf90023e4}, + {"STR X5, [SP, #32]", a64LSU(3, 0, 0, 4, 31, 5), 0xf90013e5}, + {"LDR X6, [SP, #32]", a64LSU(3, 0, 1, 4, 31, 6), 0xf94013e6}, + } + for _, tt := range tests { + if tt.got != tt.want { + t.Errorf("%s: got %08x, want %08x", tt.name, tt.got, tt.want) + } + } +} + +func TestArm64PrologueEncoding(t *testing.T) { + fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false} + pro := arm64Prologue(fi) + if len(pro) != 12 { + t.Fatalf("prologue length: got %d, want 12", len(pro)) + } + expected := []uint32{0xf81d0ffe, 0xf81f83fd, 0xd10023fd} + for i, w := range leWords(pro) { + if w != expected[i] { + t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i]) + } + } +} + +func TestArm64EpilogueSmallEncoding(t *testing.T) { + fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false} + ret := arm64Return(fi) + if len(ret) != 12 { + t.Fatalf("epilogue length: got %d, want 12", len(ret)) + } + expected := []uint32{0x9100a3fd, 0x9100c3ff, 0xd65f03c0} + for i, w := range leWords(ret) { + if w != expected[i] { + t.Errorf("epilogue word %d: got %08x, want %08x", i, w, expected[i]) + } + } +} + +func TestArm64LargeFrameEncoding(t *testing.T) { + fi := arm64FrameInfo{autosize: 272, frame: 256, leaf: false} + pro := arm64Prologue(fi) + if len(pro) != 16 { + t.Fatalf("prologue length: got %d, want 16", len(pro)) + } + expected := []uint32{0xd10443f4, 0xa93ffa9d, 0x9100029f, 0xd10023fd} + for i, w := range leWords(pro) { + if w != expected[i] { + t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i]) + } + } + + epi := arm64Return(fi) + if len(epi) != 12 { + t.Fatalf("epilogue length: got %d, want 12", len(epi)) + } + eexpected := []uint32{0xa97ffbfd, 0x910443ff, 0xd65f03c0} + for i, w := range leWords(epi) { + if w != eexpected[i] { + t.Errorf("epilogue word %d: got %08x, want %08x", i, w, eexpected[i]) + } + } +} + +func TestArm64NoFrame(t *testing.T) { + fi := arm64FrameInfo{autosize: 0, frame: 0, leaf: true} + pro := arm64Prologue(fi) + if len(pro) != 0 { + t.Errorf("no-frame prologue: got %d bytes, want 0", len(pro)) + } + ret := arm64Return(fi) + if len(ret) != 4 { + t.Fatalf("no-frame return: got %d bytes, want 4", len(ret)) + } + if leWord(ret) != 0xd65f03c0 { + t.Errorf("no-frame RET: got %08x, want d65f03c0", leWord(ret)) + } +} + +func TestArm64RegNum(t *testing.T) { + tests := []struct { + name string + want int + }{ + {"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31}, + {"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31}, + {"F0", 0}, {"F4", 4}, {"F31", 31}, + {"INVALID", -1}, {"X0", -1}, {"", -1}, + } + for _, tt := range tests { + got := arm64RegNum(tt.name) + if got != tt.want { + t.Errorf("arm64RegNum(%q) = %d, want %d", tt.name, got, tt.want) + } + } +} + +func TestArm64ComputeFrame(t *testing.T) { + src := "TEXT ·f(SB), NOSPLIT, $32-0\n\tADD\tR4, R5\n\tRET\n" + f, errs := parser.Parse("test_arm64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + fi := arm64ComputeFrame(f.Decls[0].(*ast.Text)) + if fi.frame != 32 { + t.Errorf("frame: got %d, want 32", fi.frame) + } + if fi.autosize != 48 { // 32+8=40, aligned to48 + t.Errorf("autosize: got %d, want 48", fi.autosize) + } + // ADD + RET with no CALL/BL → leaf + if !fi.leaf { + t.Error("expected leaf") + } +} + +func TestArm64IsLeaf(t *testing.T) { + src := "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR4, R5\n\tRET\n" + f, errs := parser.Parse("test_arm64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + if !arm64IsLeaf(f.Decls[0].(*ast.Text)) { + t.Error("expected leaf") + } + + src2 := "TEXT ·f(SB), NOSPLIT, $0-0\n\tBL\tother(SB)\n\tRET\n" + f2, errs := parser.Parse("test_arm64.s", src2) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + if arm64IsLeaf(f2.Decls[0].(*ast.Text)) { + t.Error("expected non-leaf") + } +} + +func TestArm64Bitmask(t *testing.T) { + tests := []struct { + v uint64 + sf int + N, immr, imms uint32 + ok bool + }{ + {1, 1, 1, 0, 0, true}, // single bit at pos 0 + {2, 1, 1, 1, 0, true}, // single bit at pos 1 (rotated right by1) + {0, 1, 0, 0, 0, false}, // zero is not a bitmask + {0xFFFFFFFFFFFFFFFF, 1, 0, 0, 0, false}, // all ones is not a bitmask + {0x5555555555555555, 1, 0, 0, 0x3E, true}, // alternating bits (esize=2, ones=1) + {0xFFFFFFFF00000000, 1, 1, 32, 31, true}, // upper 32 bits set (esize=64, ones=32) + } + for _, tt := range tests { + N, immr, imms, ok := arm64Bitmask(tt.v, tt.sf) + if ok != tt.ok { + t.Errorf("arm64Bitmask(%#x, %d): ok=%v, want %v", tt.v, tt.sf, ok, tt.ok) + continue + } + if ok && (N != tt.N || immr != tt.immr || imms != tt.imms) { + t.Errorf("arm64Bitmask(%#x, %d): N=%d immr=%d imms=%d, want N=%d immr=%d imms=%d", + tt.v, tt.sf, N, immr, imms, tt.N, tt.immr, tt.imms) + } + } +} + +func TestArm64AssembleFile(t *testing.T) { + src := `#include "textflag.h" + +TEXT ·simple(SB), NOSPLIT, $0-0 + MOV R4, R5 + ADD R4, R5, R6 + RET +` + f, errs := parser.Parse("test_arm64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + if len(img.Funcs) != 1 { + t.Fatalf("got %d funcs, want 1", len(img.Funcs)) + } + fn := img.Funcs[0] + if fn.Name != "simple" { + t.Errorf("func name: got %q, want %q", fn.Name, "simple") + } + //3 instructions ×4 bytes =12 + if fn.Size != 12 { + t.Errorf("func size: got %d, want 12", fn.Size) + } +} + +func TestArm64AssembleFileWithFrame(t *testing.T) { + src := `#include "textflag.h" + +TEXT ·framed(SB), NOSPLIT, $16-8 + MOVD arg+0(FP), R4 + ADD $1, R4, R4 + MOVD R4, ret+0(FP) + RET +` + f, errs := parser.Parse("test_arm64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + if len(img.Funcs) != 1 { + t.Fatalf("got %d funcs, want 1", len(img.Funcs)) + } + fn := img.Funcs[0] + if fn.Frame != 16 { + t.Errorf("frame: got %d, want 16", fn.Frame) + } + // Prologue (3×4=12) + body (3×4=12) + RET epilogue (3×4=12) = 36 + if fn.Size != 36 { + t.Errorf("func size: got %d, want 36", fn.Size) + } +} + +func TestArm64AssembleFileWithBranches(t *testing.T) { + src := `#include "textflag.h" + +TEXT ·branch(SB), NOSPLIT, $0-0 + BEQ done + BNE skip +skip: + ADD R4, R5 +done: + RET +` + f, errs := parser.Parse("test_arm64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + fn := img.Funcs[0] + if fn.Size != 16 { + t.Errorf("func size: got %d, want 16", fn.Size) + } +} + +func TestArm64AssembleFileWithJumpChain(t *testing.T) { + src := `#include "textflag.h" + +TEXT ·chain(SB), NOSPLIT, $0-0 + BNE skip + ADD R4, R5 + RET +skip: + B target +target: + ADD R6, R7 + RET +` + f, errs := parser.Parse("test_arm64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + // BNE should be redirected past skip→target to target directly. + if img.Funcs[0].Size != 24 { + t.Errorf("func size: got %d, want 24", img.Funcs[0].Size) + } +} + +func TestArm64AssembleErrors(t *testing.T) { + tests := []struct { + name string + src string + }{ + {"unsupported", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4, R5\n\tRET\n"}, + {"undefined label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + f, errs := parser.Parse("test_arm64.s", tt.src) + if len(errs) > 0 { + return // parse error, that's fine + } + _, err := AssembleFileARM64(f) + if err == nil { + t.Error("expected error, got nil") + } + }) + } +} + +func TestArm64Movcon(t *testing.T) { + tests := []struct { + v int64 + want int + }{ + {0, 0}, // 0 fits at shift 0 + {1, 0}, // single bit at shift 0 + {0x10000, 16}, // single bit at shift 16 + {0x100000000, 32}, // single bit at shift 32 + {0xFF, 0}, // 0xFF fits at shift 0 + {0x12345, -1}, // multiple chunks, not movcon + } + for _, tt := range tests { + got := arm64Movcon(tt.v) + if got != tt.want { + t.Errorf("arm64Movcon(%#x) = %d, want %d", tt.v, got, tt.want) + } + } +} + +func TestArm64RegClassOf(t *testing.T) { + if arm64RegClassOf("R4") != arm64ClsGR { + t.Error("R4 should be GR") + } + if arm64RegClassOf("F4") != arm64ClsFP { + t.Error("F4 should be FP") + } + if arm64RegClassOf("") != arm64ClsNone { + t.Error("empty should be None") + } +} + +func TestArm64ResolvePseudo(t *testing.T) { + fi := arm64FrameInfo{autosize: 48, frame: 32} + // FP: offset = sym.Offset + autosize +8 + base, off := arm64ResolvePseudo(&ast.Symbol{Pseudo: "FP", Offset: 0}, fi) + if base != 31 || off != 56 { + t.Errorf("FP: base=%d off=%d, want 31, 56", base, off) + } + // SP: offset = sym.Offset + frame +8 + base, off = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SP", Offset: -8}, fi) + if base != 31 || off != 32 { + t.Errorf("SP: base=%d off=%d, want 31, 32", base, off) + } + // SB: unresolved + base, _ = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SB"}, fi) + if base != -1 { + t.Errorf("SB: base=%d, want -1", base) + } +} + +// leWord reads a little-endian uint32 from b. +func leWord(b []byte) uint32 { + return uint32(b[0]) | uint32(b[1])<<8 | uint32(b[2])<<16 | uint32(b[3])<<24 +} + +// leWords reads all little-endian uint32s from b. +func leWords(b []byte) []uint32 { + n := len(b) / 4 + w := make([]uint32, n) + for i := range w { + w[i] = leWord(b[i*4:]) + } + return w +} diff --git a/asm/arm64_frame.go b/asm/arm64_frame.go new file mode 100644 index 0000000..48368cb --- /dev/null +++ b/asm/arm64_frame.go @@ -0,0 +1,234 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +// arm64 frame mapping, matching the Go toolchain's arm64 backend. +// +// Go's arm64 functions use R29 as the frame pointer (FP) and R30 as the link +// register (LR). R31 is the stack pointer (SP). FP and SP in the source +// are synthetic pseudo-registers resolved against the hardware SP and the +// frame size. +// +// The autosize is the real stack adjustment: the declared local frame plus +// 8 bytes for the saved link register, rounded up to a 16-byte multiple. +// The toolchain adds an "extrasize" to align: if autosize%16 == 8, add 8; +// if autosize%16 == 0, add 16. +// +// Prologue (autosize > 0, small frame ≤ 0xf0): +// +// MOVD.W LR, -autosize(SP) // pre-index: SP -= autosize, store LR at SP +// MOVD FP, -8(SP) // store FP at SP-8 +// SUB $8, SP, FP // FP = SP - 8 +// +// Prologue (autosize > 0, large frame > 0xf0): +// +// SUB $autosize, SP, R20 // R20 = SP - autosize +// STP (FP, LR), -8(R20) // store FP,LR at R20-8 +// MOVD R20, SP // SP = R20 +// SUB $8, SP, FP // FP = SP - 8 +// +// Epilogue (non-leaf, small frame): +// +// ADD $autosize-8, SP, FP // restore FP +// ADD $autosize, SP, SP // deallocate frame +// MOVD -8(SP), FP // (actually the reverse of prologue) +// Actually: +// MOVD -8(SP), FP // load FP from SP-8 +// MOVD.P autosize(SP), LR // post-index: load LR, SP += autosize +// +// Epilogue (non-leaf, large frame): +// ADD $autosize-8, SP, FP +// ADD $autosize, SP, SP +// Actually: +// LDP -8(SP), (FP, LR) // load FP,LR +// ADD $autosize, SP, SP // deallocate frame +// +// Epilogue (leaf with frame): +// ADD $autosize-8, SP, FP +// ADD $autosize, SP, SP +// +// RET always emits as BR LR (0xd65f03c0). + +import ( + "strings" + + "sourcedock.dev/petrbalvin/gasm-devkit/ast" +) + +// arm64FrameInfo holds the frame layout derived from a TEXT directive. +type arm64FrameInfo struct { + autosize int // the real SP adjustment (locals + saved LR + alignment) + frame int // the declared $framesize + args int // the declared -argsize + noSplit bool // the NOSPLIT flag + leaf bool // no call instructions in the body +} + +// arm64ComputeFrame derives the frame layout for a TEXT function. +func arm64ComputeFrame(t *ast.Text) arm64FrameInfo { + fi := arm64FrameInfo{ + frame: frameSize(t), + args: argsSize(t), + } + for _, f := range t.Flags { + if f == "NOSPLIT" { + fi.noSplit = true + } + } + fi.leaf = arm64IsLeaf(t) + + if fi.frame != 0 || !fi.leaf { + fi.autosize = fi.frame + 8 // space for the saved LR + if fi.autosize%16 != 0 { + // The toolchain aligns to 16: if autosize%16 == 8, add 8; + // otherwise add whatever is needed. + fi.autosize += 16 - (fi.autosize % 16) + } + } + return fi +} + +// arm64IsLeaf reports whether a function contains no call instructions +// (BL/CALL), matching the toolchain's LEAF mark. +func arm64IsLeaf(t *ast.Text) bool { + for _, stmt := range t.Body { + in, ok := stmt.(*ast.Instr) + if !ok { + continue + } + switch strings.ToUpper(in.Mnemonic.Text) { + case "BL", "CALL": + return false + } + } + return true +} + +// arm64Prologue returns the prologue bytes for an arm64 function. +func arm64Prologue(fi arm64FrameInfo) []byte { + if fi.autosize == 0 { + return nil + } + if fi.autosize <= 0xf0 { + // Small frame: MOVD.W LR, -autosize(SP); MOVD FP, -8(SP); SUB $8, SP, FP + return a64WordsLE( + arm64PreStoreImm(3, 0, int32(-fi.autosize), 31, 30), // STR.W LR, -autosize(SP) (pre-index store) + arm64UnscaledStore(3, 0, -8, 31, 29), // STUR FP, [SP, #-8] + a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB) + ) + } + // Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP + return a64WordsLE( + a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20 + a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair) + a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP) + a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB) + ) +} + +// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and +// deallocate the frame when present) followed by RET (BR LR). +func arm64Return(fi arm64FrameInfo) []byte { + var ws []uint32 + if fi.autosize != 0 { + if fi.autosize <= 0xf0 { + // Small frame (leaf or non-leaf): ADD $autosize-8, SP, FP; ADD $autosize, SP, SP + // The Go toolchain uses this simpler epilogue for small frames even for + // non-leaf functions — LR is not explicitly restored; the return address + // is already in LR from the caller's BL instruction. + ws = append(ws, + a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP + a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP + ) + } else { + // Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP + ws = append(ws, + a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair) + a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP + ) + } + } + // RET: BR LR (0xd65f03c0) + ws = append(ws, a64UncondBranch(2, 30, 0)) // opc=2(RET), Rn=LR(30), Rd=0 + return a64WordsLE(ws...) +} + +// arm64PrologueSpadjPC returns the function-relative byte offset where the +// prologue has finished decrementing SP (the delta becomes autosize). +func arm64PrologueSpadjPC(fi arm64FrameInfo) int { + if fi.autosize == 0 { + return 0 + } + if fi.autosize <= 0xf0 { + return 4 // MOVD.W instruction decrements SP + } + return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD) +} + +// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to +// (but not including) the final RET instruction. +func arm64ReturnEpilogueLen(fi arm64FrameInfo) int { + if fi.autosize == 0 { + return 0 + } + if fi.leaf { + return 8 // ADD + ADD + } + if fi.autosize <= 0xf0 { + return 8 // LDR + LDR.P + } + return 8 // LDP + ADD +} + +// arm64ResolvePseudo translates a pseudo-register memory reference into a +// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP); +// x+N(SP) → (N + frame + 8)(SP). Returns base = -1 for an unresolvable +// reference (SB: static data, handled by the relocation path). +// +// The Go toolchain resolves all pseudo-register references against the +// hardware stack pointer (R31/SP): FP references add autosize+8 (the +// distance from SP after the prologue to the caller's argument area), +// SP references add frame+8 (the distance to the local area). +func arm64ResolvePseudo(sym *ast.Symbol, fi arm64FrameInfo) (base int, off int32) { + if sym == nil { + return -1, 0 + } + switch sym.Pseudo { + case "FP": + return 31, int32(sym.Offset) + int32(fi.autosize) + 8 + case "SP": + return 31, int32(sym.Offset) + int32(fi.frame) + 8 + case "SB": + return -1, int32(sym.Offset) + } + return -1, 0 +} + +// arm64PreStoreImm encodes a pre-index store (STR with writeback): +// size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt. +func arm64PreStoreImm(size, V int, imm9 int32, rn, rt int) uint32 { + return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 | + 3<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) +} + +// arm64UnscaledStore encodes an unscaled store (STUR): +// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt. +func arm64UnscaledStore(size, V int, imm9 int32, rn, rt int) uint32 { + return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 | + (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) +} + +// arm64UnscaledLoad encodes an unscaled load (LDUR): +// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt. +func arm64UnscaledLoad(size, V int, imm9 int32, rn, rt int) uint32 { + return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 | + (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) +} + +// arm64PostLoad encodes a post-index load (LDR with post-increment): +// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt. +func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 { + return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 | + 1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) +} diff --git a/asm/elfarm64.go b/asm/elfarm64.go new file mode 100644 index 0000000..5ddc4f8 --- /dev/null +++ b/asm/elfarm64.go @@ -0,0 +1,224 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "encoding/binary" + "fmt" +) + +// AArch64 ELF64 relocatable object emission. + +const ( + emAARCH64 = 183 // EM_AARCH64 + + // AArch64 relocation types (the ELF psABI). + rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page) + rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD/STR/LDR page offset) +) + +// ELFAARCH64Object returns the image as an ELF64 relocatable object file for +// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the +// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an +// optional .rela.text. +func (img *Image) ELFAARCH64Object() ([]byte, error) { + le := binary.LittleEndian + + const ( + secText = 1 + secData = 2 + ) + + // Build symbol table. + var locals, globals []elfSym + for _, fn := range img.Funcs { + s := elfSym{ + name: objectName(fn.Pkg, fn.Name), + info: sttFunc, + shndx: secText, + value: uint64(fn.Offset), + size: uint64(fn.Size), + } + if fn.Static { + locals = append(locals, s) + } else { + s.info |= stbGlobal << stInfoShift + globals = append(globals, s) + } + } + for _, d := range img.DataSyms { + s := elfSym{ + name: objectName(d.Pkg, d.Name), + info: sttObject, + shndx: secData, + value: uint64(d.Offset), + size: uint64(d.Size), + } + if d.Static { + locals = append(locals, s) + } else { + s.info |= stbGlobal << stInfoShift + globals = append(globals, s) + } + } + for _, name := range img.Externals { + globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift}) + } + syms := []elfSym{ + {}, + {name: ".text", info: sttSection, shndx: secText}, + {name: ".data", info: sttSection, shndx: secData}, + } + syms = append(syms, locals...) + shInfo := len(syms) + syms = append(syms, globals...) + symIdx := map[string]int{} + for i, s := range syms { + symIdx[s.name] = i + } + + // Build relocations. Each SB reference is an ADRP pair: + // ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21 + // ADD/LDR/STR → R_AARCH64_ADD_ABS_LO12_NC + type elfRela struct { + off uint64 + typ uint32 + sym int + addend int64 + } + var relas []elfRela + for _, fn := range img.Funcs { + for _, r := range fn.Relocs { + idx, ok := symIdx[r.Name] + if !ok { + return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name) + } + typ := uint32(rArm64PrelPgHi21) + if r.Kind == RelArm64Addr && r.Off%4 == 4 { + // The second instruction in an ADRP pair uses ADD_ABS_LO12_NC. + typ = rArm64AddAbsLo12NC + } + relas = append(relas, elfRela{ + off: uint64(fn.Offset + r.Off), + typ: typ, + sym: idx, + addend: r.Addend - int64(r.After-r.Off), + }) + } + } + + // String tables. + stNames := newElfStrtab() + for _, s := range syms { + stNames.add(s.name) + } + stSections := newElfStrtab() + for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} { + stSections.add(n) + } + + hasRela := len(relas) > 0 + nSections := 6 + if hasRela { + nSections = 7 + } + secSymtab, secStrtab := 3, 4 + secShstr := nSections - 1 + + // Layout. + var out []byte + out = append(out, make([]byte, 64)...) + + align := func(n int) { + for len(out)%n != 0 { + out = append(out, 0) + } + } + + align(16) + textOff := len(out) + out = append(out, img.Code...) + + align(16) + dataOff := len(out) + out = append(out, img.Data...) + + align(8) + symtabOff := len(out) + for _, s := range syms { + var b [24]byte + le.PutUint32(b[0:], uint32(stNames.at(s.name))) + b[4] = s.info + b[5] = 0 + le.PutUint16(b[6:], s.shndx) + le.PutUint64(b[8:], s.value) + le.PutUint64(b[16:], s.size) + out = append(out, b[:]...) + } + + strtabOff := len(out) + out = append(out, stNames.bytes()...) + + var relaOff int + if hasRela { + align(8) + relaOff = len(out) + for _, r := range relas { + var b [24]byte + le.PutUint64(b[0:], r.off) + le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ)) + le.PutUint64(b[16:], uint64(r.addend)) + out = append(out, b[:]...) + } + } + + shstrOff := len(out) + out = append(out, stSections.bytes()...) + + align(8) + shoff := len(out) + + putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) { + var b [64]byte + le.PutUint32(b[0:], uint32(stSections.at(name))) + le.PutUint32(b[4:], uint32(typ)) + le.PutUint64(b[8:], flags) + le.PutUint64(b[16:], 0) + le.PutUint64(b[24:], uint64(off)) + le.PutUint64(b[32:], uint64(size)) + le.PutUint32(b[40:], uint32(link)) + le.PutUint32(b[44:], uint32(info)) + le.PutUint64(b[48:], alignV) + le.PutUint64(b[56:], entsize) + out = append(out, b[:]...) + } + putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0) + putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0) + putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0) + putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24) + putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0) + if hasRela { + putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24) + } + putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0) + + // ELF header. + hdr := out[:64] + copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0}) + le.PutUint16(hdr[16:], etREL) + le.PutUint16(hdr[18:], emAARCH64) + le.PutUint32(hdr[20:], elfVersion) + le.PutUint64(hdr[24:], 0) + le.PutUint64(hdr[32:], 0) + le.PutUint64(hdr[40:], uint64(shoff)) + le.PutUint32(hdr[48:], 0) + le.PutUint16(hdr[52:], 64) + le.PutUint16(hdr[54:], 0) + le.PutUint16(hdr[56:], 0) + le.PutUint16(hdr[58:], 64) + le.PutUint16(hdr[60:], uint16(nSections)) + le.PutUint16(hdr[62:], uint16(secShstr)) + + return out, nil +} diff --git a/asm/goobjarm64.go b/asm/goobjarm64.go new file mode 100644 index 0000000..bf6de16 --- /dev/null +++ b/asm/goobjarm64.go @@ -0,0 +1,84 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "bytes" + "fmt" + "os" + "os/exec" + "path/filepath" + "sync" +) + +// GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is +// the shared one in goobj.go — the toolchain preamble, the go120ld header +// with its block offsets, the string table, the symbol definitions and the +// reloc/aux/data index arrays — with the arm64 preamble, the MinLC of 4 +// for the pc-value deltas, and R_ADDRARM64 relocation types for the +// ADRP+ADD/LDR/STR address pairs. +func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) { + pre, err := toolchainObjectPreambleAARCH64() + if err != nil { + return nil, err + } + return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) { + return relocArm64Addr, 4 + }) +} + +// arm64 relocation types (cmd/internal/objabi). R_ADDRARM64 resolves an +// ADRP+ADD/LDR/STR pair to a symbol's address. +const ( + relocArm64Addr = 9 // R_ADDRARM64 +) + +// toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header +// the installed go tool asm writes for arm64, captured by assembling a +// one-instruction probe. +var ( + preambleAARCH64Once sync.Once + preambleAARCH64 []byte + preambleAARCH64Err error +) + +func toolchainObjectPreambleAARCH64() ([]byte, error) { + preambleAARCH64Once.Do(func() { + goBin, err := exec.LookPath("go") + if err != nil { + preambleAARCH64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err) + return + } + dir, err := os.MkdirTemp("", "gasm-preamble-arm64") + if err != nil { + preambleAARCH64Err = err + return + } + defer os.RemoveAll(dir) + src := filepath.Join(dir, "probe_arm64.s") + if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil { + preambleAARCH64Err = err + return + } + obj := filepath.Join(dir, "probe.o") + cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src) + cmd.Env = append(os.Environ(), "GOARCH=arm64") + if out, err := cmd.CombinedOutput(); err != nil { + preambleAARCH64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out) + return + } + data, err := os.ReadFile(obj) + if err != nil { + preambleAARCH64Err = err + return + } + i := bytes.Index(data, []byte("\n!\n")) + if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) { + preambleAARCH64Err = fmt.Errorf("unrecognised assembler object layout") + return + } + preambleAARCH64 = data[:i+3] + }) + return preambleAARCH64, preambleAARCH64Err +} diff --git a/asm/link.go b/asm/link.go index 15101d8..5977c5f 100644 --- a/asm/link.go +++ b/asm/link.go @@ -96,6 +96,7 @@ const ( RelPCRelAbs // 32-bit absolute (R_RISCV_32) RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i) RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st) + RelArm64Addr // R_ADDRARM64 (ADRP + ADD/LDR/STR pair) ) type Reloc struct { diff --git a/cmd/gasm/main.go b/cmd/gasm/main.go index 6a37031..14353d0 100644 --- a/cmd/gasm/main.go +++ b/cmd/gasm/main.go @@ -498,6 +498,8 @@ requires -p, the package path, and the installed Go toolchain). obj, err = img.ELFRISCVObject() case arch.LOONG64: obj, err = img.ELFLOONG64Object() + case arch.ARM64: + obj, err = img.ELFAARCH64Object() default: obj, err = img.ELFObject() } @@ -508,6 +510,8 @@ requires -p, the package path, and the installed Go toolchain). obj, err = img.GOObjectRISCV(*pkg, path) case arch.LOONG64: obj, err = img.GOObjectLOONG64(*pkg, path) + case arch.ARM64: + obj, err = img.GOObjectAARCH64(*pkg, path) default: obj, err = img.GOObject(*pkg, path) } @@ -921,6 +925,95 @@ func cmdVerifyLOONG64(path string, groundTruth, profile bool) int { return 0 } +func cmdVerifyARM64(path string, groundTruth, profile bool) int { + src, err := readSource(path) + if err != nil { + fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err) + return 1 + } + f, errs := parser.Parse(path, src) + for _, e := range errs { + fmt.Fprintf(os.Stderr, "%s: %v\n", path, e) + } + if len(errs) > 0 { + return 1 + } + img, err := asm.AssembleFileARM64(f) + if err != nil { + fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err) + return 1 + } + + if groundTruth { + gt, err := verify.GroundTruthARM64(path) + if err != nil { + fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err) + return 1 + } + matched, total := 0, 0 + for _, fn := range img.Funcs { + gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size] + goCode, ok := gt[fn.Name] + if !ok { + fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name) + continue + } + total++ + gasmCmp := make([]byte, len(gasmCode)) + goCmp := make([]byte, len(goCode)) + copy(gasmCmp, gasmCode) + copy(goCmp, goCode) + for _, r := range fn.Relocs { + for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ { + gasmCmp[j] = 0 + } + for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ { + goCmp[j] = 0 + } + } + if bytes.Equal(gasmCmp, goCmp) { + matched++ + if len(fn.Relocs) > 0 { + fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs)) + } else { + fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size) + } + } else { + fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode)) + for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 { + var gb, gs string + for j := i; j < i+16 && j < len(gasmCode); j++ { + gb += fmt.Sprintf(" %02x", gasmCode[j]) + } + for j := i; j < i+16 && j < len(goCode); j++ { + gs += fmt.Sprintf(" %02x", goCode[j]) + } + fmt.Printf(" %04x: gasm:%s\n", i, gb) + fmt.Printf(" %04x: gt: %s\n", i, gs) + } + } + } + fmt.Printf("%s: %d/%d matched\n", path, matched, total) + if matched < total { + return 1 + } + return 0 + } + + if profile { + for _, fn := range img.Funcs { + fmt.Printf("%s: %d bytes, labels: %v\n", fn.Name, fn.Size, fn.Labels) + } + return 0 + } + + fmt.Printf("%s: %d functions assembled\n", path, len(img.Funcs)) + for _, fn := range img.Funcs { + fmt.Printf(" %s: %d bytes\n", fn.Name, fn.Size) + } + return 0 +} + func cmdVerify(args []string) int { fs := newCommand("verify", "gasm verify [-smoke] [-abi] [-fuzz] [-ground-truth] [-profile] [-call] ", ` Assemble FILE (amd64), map it into executable memory and report the available @@ -975,6 +1068,9 @@ decoders) that crash on random input but should succeed on valid data. case arch.LOONG64: // LoongArch: ground-truth only (no JIT on non-LoongArch hosts). return cmdVerifyLOONG64(path, *groundTruth, *profile) + case arch.ARM64: + // AArch64: ground-truth only (no JIT on non-ARM64 hosts). + return cmdVerifyARM64(path, *groundTruth, *profile) default: fmt.Fprintln(os.Stderr, "gasm verify: only amd64, riscv64 and loong64 are supported") return 1 diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index e7bf73e..814dbf3 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -212,6 +212,17 @@ relocations). Like the RISC-V encoder it is validated byte-for-byte against `GOARCH=loong64 go tool asm`, and its GOOBJ output is proven end-to-end by substituting it into a cross-compiled `go build` and linking with `cmd/link`. +An **AArch64 encoder** (Phase 5, arm64) encodes the integer instruction set +with the data-processing (shifted register and immediate forms), load/store +(scaled unsigned immediate and unscaled9-bit immediate), conditional and +unconditional branches, the MOV pseudo-instruction and its constant +materialisation (MOVZ/MOVN/MOVK for wide immediates, ORR with logical bitmask +encoding for values like `$1`), the FP/SP frame mapping (autosize = +align16(frame+8), prologue using pre-index store for small frames and +STP+SUB for large frames) and SB/global symbol references (ADRP+ADD pairs with +R_ADDRARM64 relocations). Like the other encoders it is validated +byte-for-byte against `GOARCH=arm64 go tool asm`. + On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each operand to an encoder operand, and lays the instructions out so local labels resolve to relative jump offsets: jumps start in the short (rel8) form and diff --git a/testdata/verify/basic_arm64.s b/testdata/verify/basic_arm64.s new file mode 100644 index 0000000..ef49f9c --- /dev/null +++ b/testdata/verify/basic_arm64.s @@ -0,0 +1,57 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +#include "textflag.h" + +// add returns a + b. +TEXT ·add(SB), NOSPLIT, $0-24 + MOVD a+0(FP), R4 + MOVD b+8(FP), R5 + ADD R5, R4, R4 + MOVD R4, ret+16(FP) + RET + +// arith exercises the register-register integer set. +TEXT ·arith(SB), NOSPLIT, $0-0 + ADD R4, R5, R6 + SUB R7, R8, R9 + AND R10, R11, R12 + ORR R12, R13, R14 + EOR R14, R15, R16 + CMP R16, R17 + ADD R4, R5 + SUB R6, R7 + RET + +// branch exercises conditional and unconditional control flow. +TEXT ·branch(SB), NOSPLIT, $0-0 + BEQ done + BNE skip + BGE done + BLT done + BGT done + BLE done +skip: + B loop +loop: + ADD R4, R5 + RET +done: + RET + +// mov exercises the MOV pseudo-instruction. +TEXT ·mov(SB), NOSPLIT, $0-16 + MOVD $0, R4 + MOVD $1, R5 + MOVD $42, R6 + MOVD a+0(FP), R7 + MOVD R7, ret+0(FP) + MOVW $100, R8 + RET + +// frame exercises the prologue/epilogue of a function with a real frame. +TEXT ·frame(SB), NOSPLIT, $32-8 + MOVD arg+0(FP), R4 + ADD $1, R4, R4 + MOVD R4, ret+0(FP) + RET diff --git a/verify/arm64_groundtruth_test.go b/verify/arm64_groundtruth_test.go new file mode 100644 index 0000000..b8fc0c9 --- /dev/null +++ b/verify/arm64_groundtruth_test.go @@ -0,0 +1,83 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package verify + +import ( + "bytes" + "os" + "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/asm" + "sourcedock.dev/petrbalvin/gasm-devkit/parser" +) + +// TestGroundTruthARM64 assembles the arm64 test kernels with gasm and +// compares them byte-for-byte against `go tool asm` (GOARCH=arm64). The +// relocation fields of static-symbol references are masked before the +// comparison, since the toolchain leaves them zero for the linker. +func TestGroundTruthARM64(t *testing.T) { + for _, path := range []string{ + "../testdata/verify/basic_arm64.s", + } { + t.Run(path, func(t *testing.T) { + src, err := os.ReadFile(path) + if err != nil { + t.Fatalf("read: %v", err) + } + f, errs := parser.Parse(path, string(src)) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := asm.AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + gt, err := GroundTruthARM64(path) + if err != nil { + t.Fatalf("GroundTruthARM64: %v", err) + } + + matched := 0 + for _, fn := range img.Funcs { + gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs) + goCode, ok := gt[fn.Name] + if !ok { + t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt)) + continue + } + goCode = maskRelocs(goCode, fn.Relocs) + // The Go toolchain may add zero padding at the end of + // functions. Compare up to the shorter length, then + // verify any trailing bytes are zero. + cmpLen := len(gasmCode) + if len(goCode) < cmpLen { + cmpLen = len(goCode) + } + if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) { + t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode)) + continue + } + // Check trailing padding is zero. + trailingOK := true + if len(goCode) > len(gasmCode) { + for _, b := range goCode[len(gasmCode):] { + if b != 0 { + trailingOK = false + break + } + } + } + if !trailingOK { + t.Errorf("%s: non-zero trailing bytes in go tool asm output", fn.Name) + continue + } + matched++ + t.Logf("%s: MATCH (%d bytes, go=%d)", fn.Name, fn.Size, len(goCode)) + } + if matched == 0 { + t.Fatal("no functions matched") + } + }) + } +} diff --git a/verify/groundtruth.go b/verify/groundtruth.go index 309643e..6a5757f 100644 --- a/verify/groundtruth.go +++ b/verify/groundtruth.go @@ -38,6 +38,12 @@ func GroundTruthLOONG64(path string) (map[string][]byte, error) { return groundTruthArch(path, "loong64") } +// GroundTruthARM64 assembles the given .s file with the Go toolchain in +// AArch64 cross-assembly mode (GOARCH=arm64). +func GroundTruthARM64(path string) (map[string][]byte, error) { + return groundTruthArch(path, "arm64") +} + func groundTruthArch(path, goarch string) (map[string][]byte, error) { goroot := runtime.GOROOT() asmBin := filepath.Join(goroot, "pkg", "tool", runtime.GOOS+"_"+runtime.GOARCH, "asm") @@ -58,6 +64,7 @@ func groundTruthArch(path, goarch string) (map[string][]byte, error) { pkg = strings.TrimSuffix(pkg, "_amd64") pkg = strings.TrimSuffix(pkg, "_riscv64") pkg = strings.TrimSuffix(pkg, "_loong64") + pkg = strings.TrimSuffix(pkg, "_arm64") cmd := exec.Command(asmBin, "-I", includeDir, "-p", pkg, "-o", objPath, path) if goarch != "" {