// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "fmt" "math/bits" "strconv" "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" ) // assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine // code. Every instruction is 4 bytes; the MOV pseudo-instruction and the // immediate-arithmetic forms expand to 2-4 instructions when the immediate // does not fit, so the layout is computed in two passes (sizes, then encoding // with resolved branch targets). The returned literals carry the read-only // constants any VMOVS/VMOVD/VMOVQ load refers to; the file assembler lays // them out in the data section. // // The emitted bytes match the Go toolchain's arm64 assembler, which is the // ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch // encodings and the MOV immediate expansions all follow cmd/internal/obj/ // arm64's asmout cases. One deliberate difference: the stack-growth guard // (the morestack check in the prologue and the call back into the runtime in // the epilogue) is not emitted, so the bytes match only for NOSPLIT functions // or zero-frame leaves, where the toolchain emits no guard either. func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, []Arm64Literal, error) { fi := arm64ComputeFrame(t) prologue := arm64Prologue(fi) guardLen := arm64GuardLen(fi) chain := arm64JumpChain(t) resolve := func(name string) string { if r, ok := chain[name]; ok { return r } return name } var relocs []Reloc var spadj []SpadjStep lits := &arm64Literals{} // The prologue (3 instructions when a small frame, 4 for large) // raises the SP delta by autosize. The guard prefix shifts its PC. if fi.autosize != 0 { spadj = append(spadj, SpadjStep{PC: guardLen + arm64PrologueSpadjPC(fi), Value: fi.autosize}) } // Pass 1: label offsets from the instruction sizes. offsets := map[string]int{} pos := guardLen + len(prologue) for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: offsets[s.Name.Text] = pos case *ast.Instr: if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" { pos += arm64PCAlignPad(pos, s) } else { pos += arm64InstrSize(s, fi, pos) } } } // Pass 2: encode. The guard prefix precedes the prologue; its branches // target the morestack block at the end of the function, whose position // the first pass has settled. bodyLen := 0 { p := guardLen + len(prologue) for _, stmt := range t.Body { if in, ok := stmt.(*ast.Instr); ok { p += arm64InstrSize(in, fi, p) } } bodyLen = p - (guardLen + len(prologue)) } var out []byte if fi.needSplit { out = append(out, arm64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...) } out = append(out, prologue...) pc := guardLen + len(prologue) preCount := len(relocs) var lines []LineEntry for _, stmt := range t.Body { in, ok := stmt.(*ast.Instr) if !ok { continue } if strings.ToUpper(in.Mnemonic.Text) == "PCALIGN" { pad := arm64PCAlignPad(pc, in) for i := 0; i < pad/4; i++ { out = append(out, a64wordLE(a64NOP)...) pc += 4 } continue } if strings.ToUpper(in.Mnemonic.Text) == "BYTE" { for _, op := range in.Operands { out = append(out, byte(arm64Imm64(op))) pc++ } continue } code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve, lits) if err != nil { return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) } for j := preCount; j < len(relocs); j++ { // Make the relocation offsets function-relative: each instruction // records its reloc offset relative to its own start, and pc is // that instruction's offset from the function start (prologue // included). After shifts by the same amount. relocs[j].Off += pc relocs[j].After += pc } preCount = len(relocs) lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line}) // The RET's epilogue closes the frame: the SP delta returns to zero. if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 { epi := arm64ReturnEpilogueLen(fi) spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0}) } out = append(out, code...) pc += len(code) } if fi.needSplit { block, blReloc := arm64MoreStackBlock(pc) out = append(out, block...) relocs = append(relocs, blReloc) pc += len(block) } return out, offsets, relocs, lines, spadj, lits.list(), nil } // arm64JumpChain precomputes jump-to-jump folding: a label whose first // instruction is an unconditional local jump redirects its own jumpers to // the ultimate target. The Go toolchain chases these chains before it // encodes branches, so matching its bytes requires the same redirection. func arm64JumpChain(t *ast.Text) map[string]string { leadsTo := map[string]string{} for i, stmt := range t.Body { l, ok := stmt.(*ast.Label) if !ok { continue } j := i + 1 for j < len(t.Body) { if _, isLabel := t.Body[j].(*ast.Label); !isLabel { break } j++ } if j >= len(t.Body) { continue } in, ok := t.Body[j].(*ast.Instr) if !ok { continue } mnem := strings.ToUpper(in.Mnemonic.Text) if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 { continue } if name, ok := arm64LabelOK(in.Operands[0]); ok { leadsTo[l.Name.Text] = name } } chain := map[string]string{} for name := range leadsTo { visited := map[string]bool{name: true} cur := name for { next, ok := leadsTo[cur] if !ok || visited[next] { break } visited[next] = true cur = next } if cur != name { chain[name] = cur } } return chain } // arm64LabelOK returns the local label name of a jump operand. func arm64LabelOK(op *ast.Operand) (string, bool) { if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Sym.Name != "" { return op.Addr.Sym.Name, true } return "", false } // arm64WritebackSuffix splits a mnemonic carrying the toolchain's post-index // (.P) or pre-index (.W) suffix, as in MOVD.P or LDP.W. It reports the base // mnemonic, the suffix letter and whether a suffix was present. func arm64WritebackSuffix(mnem string) (base, wb string, ok bool) { if before, ok0 := strings.CutSuffix(mnem, ".P"); ok0 { return before, "P", true } if before, ok0 := strings.CutSuffix(mnem, ".W"); ok0 { return before, "W", true } return mnem, "", false } // arm64PCAlignPad returns the padding PCALIGN inserts before the next // instruction so that it starts at the requested boundary relative to the // function start. The boundary must be a power of two between 8 and 2048, // as the toolchain requires. func arm64PCAlignPad(pos int, instr *ast.Instr) int { if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) { return 0 } align := int(arm64Imm64(instr.Operands[0])) if align < 8 || align > 2048 || align&(align-1) != 0 { return 0 } return (align - pos%align) % align } // isARM64MovMnemonic reports whether m is one of the MOV-family spellings the // arm64 encoder treats as the MOV pseudo-instruction. func isARM64MovMnemonic(m string) bool { switch m { case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", "FMOVS", "FMOVD": return true } return false } // arm64InstrSize returns the encoded size of an instruction: 4 bytes for // most, more for the multi-instruction expansions. func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int { mnem := strings.ToUpper(instr.Mnemonic.Text) ops := instr.Operands if mnem == "RET" { return len(arm64Return(fi)) } if mnem == "PCALIGN" { return arm64PCAlignPad(pos, instr) } if mnem == "BYTE" { return len(ops) } switch mnem { case "VMOVS", "VMOVD", "VMOVQ": // ADRP + ADD + wide load against a pooled literal. return 12 } // Writeback (.P/.W) forms are always a single instruction. if base, _, ok := arm64WritebackSuffix(mnem); ok { if isARM64MovMnemonic(base) || a64InstrTable[base].format == a64FPair { return 4 } } // The funcdata pseudo-statements contribute no bytes, the expanded // FUNCDATA/PCDATA forms included. switch mnem { case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA": return 0 } switch mnem { case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", "FMOVS", "FMOVD": return arm64MovSize(mnem, ops, fi) case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW", "ADDS", "ADDSW", "SUBS", "SUBSW": if len(ops) >= 2 && isImmOperand(ops[0]) { // Size exactly as the encoder will emit: a single imm12 word, the // two-word ADDCON2 split, or a materialisation into REGTMP plus // the register form. Anything else would desynchronise the label // offsets of pass 1 from the bytes pass 2 lays down. if v, ok := arm64ImmOperandValue(ops[0]); ok { rn, rd := 0, 0 if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 { rd = n } if len(ops) == 3 { if n := arm64RegNum(operandRegName(ops[1])); n >= 0 { rn = n } } if ws, err := arm64AddSubImmWords(mnem, v, rn, rd); err == nil { return 4 * len(ws) } } return 4 } } return 4 } // encodeARM64Instr encodes a single AArch64 instruction. lits collects the // read-only literals a VMOVS/VMOVD/VMOVQ constant load needs; the file // assembler lays them out once every function is encoded. func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64FrameInfo, relocs *[]Reloc, resolve func(string) string, lits *arm64Literals) ([]byte, error) { mnem := strings.ToUpper(instr.Mnemonic.Text) ops := instr.Operands // Pseudo-instructions and special cases first. switch mnem { case "RET": return arm64Return(fi), nil case "NOP", "NOOP": return a64wordLE(a64NOP), nil case "UNDEF": return a64wordLE(a64BRK(0)), nil case "WORD": if len(ops) != 1 { return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) } w := arm64Imm64(ops[0]) if w < 0 || w > 0xFFFFFFFF { return nil, fmt.Errorf("WORD: immediate %d does not fit a 32-bit word", w) } return a64wordLE(uint32(w)), nil case "B", "JMP": return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve) case "BL", "CALL": return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve) case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", "FMOVS", "FMOVD": return encodeARM64Mov(instr, mnem, "", fi, relocs) } // Post-index (.P) and pre-index (.W) writeback forms: the MOV family and // the load/store pair family carry the suffix on the mnemonic itself. // (The SIMD VLD1.P/VST1.P/VLD1R.P/VLD4R.P spellings also end in .P, but // for them the suffix is part of the mnemonic and the table routes them.) if base, wb, ok := arm64WritebackSuffix(mnem); ok { switch { case isARM64MovMnemonic(base): return encodeARM64Mov(instr, base, wb, fi, relocs) case a64InstrTable[base].format == a64FPair: return encodeARM64Pair(base, a64InstrTable[base].op, ops, fi, wb, relocs) } } // The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS, // GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine // code: the toolchain emits zero instruction bytes for them, and so does // the encoder here. Files that include funcdata.h spell them after // macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are // bookkeeping too (the same treatment the loong64 encoder applies). switch mnem { case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED": return nil, nil case "END": if len(ops) != 0 { return nil, fmt.Errorf("END expects no operands, got %d", len(ops)) } return nil, nil case "FUNCDATA": if len(ops) != 2 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)") } return nil, nil case "PCDATA": if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) { return nil, fmt.Errorf("PCDATA expects $n, $n") } return nil, nil } // Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBranchCond { return encodeARM64BranchCond(mnem, enc.op, ops, pc, offsets, resolve) } // Unconditional register branches (BR, BLR). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FUncondBranch { return encodeARM64RegBranch(mnem, enc.op, ops) } // ADD/SUB immediate. if mnem == "ADD" || mnem == "ADDW" || mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW" { if len(ops) >= 2 && isImmOperand(ops[0]) { return encodeARM64AddSubImm(mnem, ops) } } // Shifts: immediate forms alias SBFM/UBFM/EXTR, register forms are the // two-source LSLV/LSRV/ASRV/RORV. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FShift { return encodeARM64Shift(mnem, enc.op, ops) } // Multiply-accumulate: MADD/MSUB Rm, Ra, Rn, Rd. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPR4 { return encodeARM64MAddSub(mnem, enc.op, ops) } // Register-register data processing. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPSR { return encodeARM64DPSR(mnem, enc.op, ops) } // FP 3-operand (Rm, Rn, Rd). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFP3 { return encodeARM64FP3(mnem, enc.op, ops) } // FP unary (Rn, Rd). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPUnary { return encodeARM64FPUnary(mnem, enc.op, ops) } // FP 4-operand FMA (Ra, Rm, Rn, Rd). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFP4 { return encodeARM64FP4(mnem, enc.op, ops) } // FP compare (Rm, Rn or #0, Rn). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCmp { return encodeARM64FPCmp(mnem, enc.op, ops) } // FP conditional compare (Rm, Rn, #nzcv, cond). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCCmp { return encodeARM64FPCCmp(mnem, enc.op, ops) } // FP conditional select (Rm, Rn, Rd, cond). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPSel { return encodeARM64FPSel(mnem, enc.op, ops) } // FP ↔ integer conversion. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCvt { return encodeARM64FPCvt(mnem, enc.op, ops) } // Conditional select (CSEL, CSINC, CSINV, CSNEG, CSET, CSETM, CINC, CINV, CNEG). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCSEL { return encodeARM64CSEL(mnem, enc.op, ops) } // CRC32. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCRC32 { return encodeARM64CRC32(mnem, enc.op, ops) } // Exclusive load/store (LDXR, STXR, LDAXR, STLXR and the register-pair // forms LDXP, STXP). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl { return encodeARM64Excl(mnem, enc.op, ops) } // LSE atomics (LDADD, CAS, SWP). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FLSE { return encodeARM64LSEAtom(mnem, enc.op, ops) } // Bitfield/shift (ASR, LSL, LSR, ROR, BFI, BFXIL, SBFM, UBFM). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield { return encodeARM64Bitfield(mnem, enc.op, ops) } // Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias { return encodeARM64BitfieldAlias(mnem, enc.op, ops) } // EXTR. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR { return encodeARM64Extr(mnem, enc.op, ops) } // Acquire/release loads and stores (LDAR family, STLR family). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FAcqRel { return encodeARM64AcqRel(mnem, enc.op, ops) } // Load/store pairs (LDP, STP, LDPW, STPW, FLDPD, FSTPD). The .P/.W // writeback forms are routed earlier, straight from the mnemonic. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FPair { return encodeARM64Pair(mnem, enc.op, ops, fi, "", relocs) } // Compare-and-branch and test-and-branch to a label. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBranch19 { return encodeARM64Branch19(mnem, enc.op, ops, pc, offsets, resolve) } if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FTestBranch { return encodeARM64TestBranch(mnem, enc.op, ops, pc, offsets, resolve) } // Data-processing (1 source): RBIT, REV, CLZ, CLS. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDP1 { return encodeARM64DP1(mnem, enc.op, ops) } // ADR/ADRP: (label, Rd) with the byte distance split into immlo and // immhi. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FADR { return encodeARM64ADR(mnem, enc.op, ops, pc, offsets, resolve) } // Bitfield extract with wrapping immr: UBFX, SBFX. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield2 { return encodeARM64Bitfield2(mnem, enc.op, ops) } // Conditional compare: CCMP, CCMN. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCondCmp { return encodeARM64CondCmp(mnem, enc.op, ops) } // System operations: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSys { return encodeARM64Sys(mnem, ops) } // Crypto: AESD, AESE, SHA1C, SHA256H and friends. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCrypto2 { return encodeARM64Crypto(mnem, enc.op, ops, 2) } if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCrypto3 { return encodeARM64Crypto(mnem, enc.op, ops, 3) } // Move wide with an explicit immediate: MOVK. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMovWide { return encodeARM64MoveWide(mnem, enc.op, ops) } // SIMD element moves (VDUP, VMOV with lane indices) take precedence // over the plain arrangement paths, which carry no index. if mnem == "VDUP" || mnem == "VMOV" { if arm64SimdHasElement(ops) { return encodeARM64Dup(mnem, ops) } // VMOV/VDUP Rn, Vd.: a general register into an arranged whole // vector (asm7.go case 82, shared by both mnemonics). The element // paths above only run when a lane index is spelled, so this is the // whole-vector shape's only route. if b, ok, err := encodeARM64GPToVec(mnem, ops); ok { return b, err } } // Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1, // VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so // this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist // only in the zero-immediate form (a64SimdVZero), so they route here with // an empty register-form spec. if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 { return encodeARM64SimdV(mnem, spec, ops) } // Arrangement-aware SIMD two-register (VREV32, VREV64, VUADDLV, VMOV). if spec, ok := a64SimdV2Table[mnem]; ok { return encodeARM64SimdV2(mnem, spec, ops) } // SIMD four-register and immediate three-register (VEOR3, VBCAX, VXAR, // VEXT). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMDV4 { return encodeARM64SimdV4(mnem, enc.op, ops) } // SIMD table lookup. if mnem == "VTBL" || mnem == "VTBX" { return encodeARM64VTBL(mnem, ops) } // SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R, // VLD4R). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FVLDST { return encodeARM64VLDST(mnem, enc.op, ops) } // SIMD shift by immediate (VSHL, VUSHR, VSRI). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FShiftImm { return encodeARM64ShiftImm(mnem, enc.op, ops) } // VMOVS/VMOVD/VMOVQ with a large constant: a literal pool load. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMoviLit { return encodeARM64MoviLit(mnem, enc.op, ops, relocs, lits) } return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) } // ---- branch encoding ---- // encodeARM64Branch encodes an unconditional branch (B/BL) to a label. func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[string]int, link bool, relocs *[]Reloc, resolve func(string) string) ([]byte, error) { if len(ops) != 1 { return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) } op := ops[0] // Branch to the program counter: JMP (PC) spins forever, and a spelled // offset (CALL -1(PC), the return stub) rides the imm26 field in word // units. The toolchain encodes both as a plain branch of that offset. if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") { rel := op.Addr.Offset if rel < -(1<<25) || rel >= (1<<25) { return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel) } bop := uint32(0) // B if link { bop = 1 // BL } return a64wordLE(a64Branch(bop, int32(rel))), nil } // Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The // toolchain's spelling carries no offset and no index; anything else // is reported rather than silently dropped. if op.Addr.Sym == nil && op.Addr.Base != "" { if op.Addr.Offset != 0 || op.Addr.Index != "" { return nil, fmt.Errorf("%s: invalid indirect branch operand %q", mnem, op.Raw) } rn := arm64RegNum(op.Addr.Base) if rn < 0 { return nil, fmt.Errorf("%s: unknown branch register %q", mnem, op.Addr.Base) } opc := uint32(0) // BR if link { opc = 1 // BLR } return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil } // The bare spelling BL R9 is the same indirect branch: the parser reads // a bare identifier as a symbol, and one named for a register is an // indirect branch through it, which the toolchain accepts alongside the // parenthesised form (BL (R3) and BL R3 both encode BLR R3). if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Index == "" { if rn := arm64RegNum(op.Addr.Sym.Name); rn >= 0 { opc := uint32(0) // BR if link { opc = 1 // BLR } return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil } } // Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a // relocation (R_CALLARM64 either way). if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" { if relocs != nil { *relocs = append(*relocs, Reloc{ Off: 0, After: 4, Name: op.Addr.Sym.Name, Addend: op.Addr.Sym.Offset, Kind: RelArm64Branch, }) } // Emit B/BL with zero offset; the linker fills in the target. bop := uint32(0) // B if link { bop = 1 // BL } return a64wordLE(a64Branch(bop, 0)), nil } target := resolve(arm64Label(op)) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q", target) } rel := (targetOff - pc) >> 2 if rel < -(1<<25) || rel >= (1<<25) { return nil, fmt.Errorf("branch to %q too far (26-bit range)", target) } bop := uint32(0) // B if link { bop = 1 // BL } return a64wordLE(a64Branch(bop, int32(rel))), nil } // encodeARM64RegBranch encodes BR/BLR through a register operand: // BR Xn = 0xd61f0000 | Rn<<5, BLR Xn = 0xd63f0000 | Rn<<5. func encodeARM64RegBranch(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 1 { return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) } rn := arm64RegNum(operandRegName(ops[0])) if rn < 0 { return nil, fmt.Errorf("%s expects a register operand", mnem) } return a64wordLE(uint32(baseOp) | 31<<16 | uint32(rn)<<5), nil } // encodeARM64BranchCond encodes a conditional branch (B.cond) to a label. func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { if len(ops) != 1 { return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) } rel, pcRel := arm64PCRelOffset(ops[0]) if !pcRel { target := resolve(arm64Label(ops[0])) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q", target) } rel = (targetOff - pc) >> 2 } if rel < -(1<<18) || rel >= (1<<18) { return nil, fmt.Errorf("%s: branch offset %d out of 19-bit range", mnem, rel) } // The condition code is in the low 4 bits of baseOp. cond := baseOp & 0xF return a64wordLE(a64BranchCond(int32(rel), cond)), nil } // ---- data-processing (shifted register) ---- // encodeARM64DPSR encodes a data-processing (shifted register) instruction. // For most instructions: OP Rm, Rn, Rd (3 operands) or OP Rm, Rd (2 operands, Rn=Rd). // For CMP/CMN/TST: CMP Rm, Rn (Rd=ZR). // For NEG: NEG Rm, Rd (Rn=ZR). // The first operand may carry the toolchain's modifier shapes: a shifted // register (R0<<2, R1>>3) or an extend modifier (R0.UXTW, R3.SXTW<<2). // Logical instructions (AND/ANDS/BIC/…) also accept a bitmask immediate, and // the ADD/SUB-with-flags family an add/sub immediate. func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW" isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" || mnem == "MVN" || mnem == "MVNW" || mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW" // Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating // bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the // complement of the written value. if len(ops) >= 2 && len(ops) <= 3 && isImmOperand(ops[0]) { var logical bool switch mnem { case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW", "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW", "TST", "TSTW": logical = true } if logical { v, ok := arm64ImmOperandValue(ops[0]) if !ok { return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw) } inverted := false switch mnem { case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW": inverted = true } if inverted { v = ^v } width := 64 if strings.HasSuffix(mnem, "W") { width = 32 } n, immr, imms, ok := a64LogicalImm(v, width) if !ok { // Beyond the bitmask immediates the toolchain materialises // the constant into REGTMP (R27) and uses the register form // (asm7.go cases 62 and 13). BIC/ORN/EON read the written // value, so the materialisation uses v before any inversion. written := v if inverted { written = ^v } width := mnem if strings.HasSuffix(mnem, "W") { width = "MOVW" } else { width = "MOVD" } mw, merr := encodeARM64LoadImm(27, written, width) var rn, rd int switch len(ops) { case 3: rn = arm64RegNum(operandRegName(ops[1])) rd = arm64RegNum(operandRegName(ops[2])) default: rd = arm64RegNum(operandRegName(ops[1])) rn = rd } if isCmp { rd = 31 } if merr != nil || rn < 0 || rd < 0 { return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " ")) } return append(mw, a64wordLE(baseOp|27<<16|uint32(rn)<<5|uint32(rd))...), nil } opc := (baseOp >> 29) & 7 sf := (baseOp >> 31) & 1 var rn, rd int switch len(ops) { case 3: rn = arm64RegNum(operandRegName(ops[1])) rd = arm64RegNum(operandRegName(ops[2])) default: rd = arm64RegNum(operandRegName(ops[1])) rn = rd } if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(sf<<31 | opc<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | uint32(rn)<<5 | uint32(rd)), nil } } // Shifted-register and extend-modifier first operand: OP Rm<= 2 && arm64RegMod(ops[0]) { rm, shiftBits, extendOpt, isExtend, amount, ok := arm64RegModifier(ops[0]) if !ok { return nil, fmt.Errorf("%s: invalid register modifier %q", mnem, ops[0].Raw) } isAddSub := strings.HasPrefix(mnem, "ADD") || strings.HasPrefix(mnem, "SUB") || isCmp || isNeg || mnem == "ADC" || mnem == "ADCS" || mnem == "SBC" || mnem == "SBCS" || mnem == "ADCW" || mnem == "ADCSW" || mnem == "SBCW" || mnem == "SBCSW" if isExtend && !isAddSub { return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem) } if !isExtend { // ROR rides the shifted-register field only for the logical // group; the toolchain reports "unsupported shift operator" for // the arithmetic forms, whose shift=11 encoding is unallocated. if shiftBits == 3 && !arm64LogicalShifted(mnem) { return nil, fmt.Errorf("%s: unsupported shift operator", mnem) } // The imm6 field is 5 bits and truncates at the 32-bit width. limit := 63 if strings.HasSuffix(mnem, "W") { limit = 31 } if amount < 0 || amount > limit { return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount) } // SP-based ADD/SUB have no shifted-register encoding: the // toolchain canonicalises LSL #n to the extend form (UXTX, or // UXTW in the 32-bit forms) and rejects a right shift. spInvolved := false for _, op := range ops[1:] { if n := operandRegName(op); n == "SP" || n == "RSP" { spInvolved = true } } if spInvolved && strings.HasPrefix(mnem, "ADD") || spInvolved && strings.HasPrefix(mnem, "SUB") { if shiftBits != 0 { return nil, fmt.Errorf("%s: right shift not encodable against SP", mnem) } opt := uint32(3) // UXTX if strings.HasSuffix(mnem, "W") { opt = 2 // UXTW } baseOp |= 1<<21 | opt<<13 | uint32(amount)<<10 } else { baseOp |= shiftBits<<22 | uint32(amount)<<10 } } else { baseOp |= 1<<21 | extendOpt<<13 | uint32(amount)<<10 } rd := arm64RegNum(operandRegName(ops[len(ops)-1])) rn := rd if len(ops) == 3 { rn = arm64RegNum(operandRegName(ops[1])) } if isCmp { rd = 31 } if isNeg && len(ops) == 2 { rn = 31 } if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil } switch len(ops) { case 3: // The carry family carries an immediate spelling in three operands // too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register // operand, the same shape the two-operand form takes. if isImmOperand(ops[0]) && arm64CarryOp(mnem) { if v := arm64Imm64(ops[0]); v != 0 { return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem) } rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[2])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil } // OP Rm, Rn, Rd rm := arm64RegNum(operandRegName(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[2])) if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil case 2: // ADC family carries an immediate spelling: ADC $0, Rd reads the // carry into Rd and takes ZR as the register operand. The // encoding has no immediate field, so $0 is the only value. if isImmOperand(ops[0]) { v := arm64Imm64(ops[0]) if v != 0 { return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem) } rd := arm64RegNum(operandRegName(ops[1])) if rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | 31<<16 | uint32(rd)<<5 | uint32(rd)), nil } if isCmp { // CMP Rm, Rn → SUBS XZR, Rn, Rm rm := arm64RegNum(operandRegName(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) if rm < 0 || rn < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | 31), nil } if isNeg { // NEG Rm, Rd → SUB Rd, ZR, Rm rm := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[1])) if rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | 31<<5 | uint32(rd)), nil } // OP Rm, Rd → OP Rm, Rd, Rd rm := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[1])) if rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil } return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } // arm64LogicalShifted reports whether a mnemonic belongs to the logical // shifted-register group, the only forms whose register operand accepts the // ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms). func arm64LogicalShifted(mnem string) bool { switch mnem { case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW", "ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW", "TST", "TSTW", "MVN", "MVNW": return true } return false } // arm64CarryOp reports whether a mnemonic belongs to the carry-using // arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only // data-processing instructions the toolchain accepts an immediate $0 // operand spelling for. func arm64CarryOp(mnem string) bool { switch mnem { case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW": return true } return false } // arm64RegMod reports whether a register operand carries the shifted-register // or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend // option (R0.UXTW, R3.SXTW<<2). func arm64RegMod(op *ast.Operand) bool { if op.Addr.Shift != "" { return true } name := operandRegName(op) if _, after, ok := strings.Cut(name, "."); ok { return strings.IndexByte(after, '[') < 0 // element selectors are not extend modifiers } return false } // arm64RegModifier resolves a modified register operand: the register number, // the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend // option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives // from the parser with the raw token spacing ("@ > 7"), so it is compacted // before the operator match. func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) { name := operandRegName(op) shift := strings.Join(strings.Fields(op.Addr.Shift), "") if before, after, ok0 := strings.Cut(name, "."); ok0 { switch strings.ToUpper(strings.TrimSpace(after)) { case "UXTB": extendOpt = 0 case "UXTH": extendOpt = 1 case "UXTW", "UXTW32": extendOpt = 2 case "UXTX": extendOpt = 3 case "SXTB": extendOpt = 4 case "SXTH": extendOpt = 5 case "SXTW": extendOpt = 6 case "SXTX": extendOpt = 7 default: return 0, 0, 0, false, 0, false } extend = true rm = arm64RegNum(strings.TrimSpace(before)) if rm < 0 { return 0, 0, 0, false, 0, false } amount, ok = arm64ShiftAmount(shift) if !ok || amount < 0 || amount > 4 { return 0, 0, 0, false, 0, false } return rm, 0, extendOpt, true, amount, true } shiftKind = 0 // LSL switch { case strings.HasPrefix(shift, "<<"): shiftKind = 0 case strings.HasPrefix(shift, ">>"): shiftKind = 1 // LSR case strings.HasPrefix(shift, "->"): shiftKind = 2 // ASR case strings.HasPrefix(shift, "@>"): shiftKind = 3 // ROR default: return 0, 0, 0, false, 0, false } amount, ok = arm64ShiftAmount(shift) if !ok { return 0, 0, 0, false, 0, false } rm = arm64RegNum(name) if rm < 0 { return 0, 0, 0, false, 0, false } return rm, shiftKind, 0, false, amount, true } // arm64ShiftAmount extracts the integer after the shift operator in a // shift suffix (<<, >>, ->, @>). func arm64ShiftAmount(shift string) (int, bool) { s := strings.TrimSpace(shift) s = strings.TrimPrefix(s, "<<") s = strings.TrimPrefix(s, ">>") s = strings.TrimPrefix(s, "->") s = strings.TrimPrefix(s, "@>") s = strings.TrimSpace(s) if s == "" { if shift == "" { return 0, true } return 0, false } v, err := strconv.Atoi(s) if err != nil { return 0, false } return v, true } // encodeARM64Shift encodes LSL/LSR/ASR/ROR in both widths. The operand order // is source first, destination last: OP $sh|Rm, Rn, Rd or OP $sh|Rm, Rd. // With an immediate the shift is the SBFM/UBFM (ROR: EXTR) alias, with a // register it is the data-processing (2 source) LSLV/LSRV/ASRV/RORV; the // two-source opcode rides the same 0xd6<<21 field as SDIV/UDIV, with // LSLV=0b001000, LSRV=0b001001, ASRV=0b001010, RORV=0b001011 at bits 15:10. func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[len(ops)-1])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } if isImmOperand(ops[0]) { width := uint32(64) if strings.HasSuffix(mnem, "W") { width = 32 } sh := arm64Imm64(ops[0]) if sh < 0 || uint32(sh) >= width { return nil, fmt.Errorf("%s: shift amount %d out of range for %d-bit form", mnem, sh, width) } switch mnem { case "LSL", "LSLW": // UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh immr := (width - uint32(sh)) % width return a64wordLE(baseOp | immr<<16 | (width-1-uint32(sh))<<10 | uint32(rn)<<5 | uint32(rd)), nil case "LSR", "LSRW": // UBFM Rd, Rn, #sh, #(W-1) return a64wordLE(baseOp | uint32(sh)<<16 | (width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil case "ASR", "ASRW": // SBFM Rd, Rn, #sh, #(W-1) return a64wordLE(baseOp | uint32(sh)<<16 | (width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil default: // ROR, RORW: EXTR Rd, Rn, Rn, #sh (Rm = Rn, imms = sh). return a64wordLE(baseOp | uint32(rn)<<16 | uint32(sh)<<10 | uint32(rn)<<5 | uint32(rd)), nil } } rm := arm64RegNum(operandRegName(ops[0])) if rm < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } op2 := uint32(8) // LSLV switch mnem { case "LSR", "LSRW": op2 = 9 // LSRV case "ASR", "ASRW": op2 = 10 // ASRV case "ROR", "RORW": op2 = 11 // RORV } sf := uint32(1) if strings.HasSuffix(mnem, "W") { sf = 0 } return a64wordLE(sf<<31 | 0xd6<<21 | op2<<10 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64MAddSub encodes MADD/MSUB/MADDW/MSUBW. The toolchain's operand // order is Rm, Ra, Rn, Rd (its optab case 15 comment says exactly that), so // the accumulate register is the SECOND operand: base | Rm<<16 | Ra<<10 | // Rn<<5 | Rd. The optab has no shorter row for these mnemonics, so all four // operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the // MUL mnemonic, not to these. func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { // The widening three-operand forms (SMULL, UMNEGL, …) read the // accumulate register as ZR, already preset in the table's base word. if len(ops) == 3 { switch mnem { case "SMULL", "UMULL", "SMNEGL", "UMNEGL": default: return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem) } rm := arm64RegNum(operandRegName(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[2])) if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil } if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops)) } rm := arm64RegNum(operandRegName(ops[0])) ra := arm64RegNum(operandRegName(ops[1])) rn := arm64RegNum(operandRegName(ops[2])) rd := arm64RegNum(operandRegName(ops[3])) if rm < 0 || rn < 0 || ra < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(ra)<<10 | uint32(rn)<<5 | uint32(rd)), nil } // ---- ADD/SUB immediate ---- // encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction, // following the toolchain's immediate classification (asm7.go conclass and // optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an // ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and // otherwise a constant materialisation into REGTMP (R27) followed by the // register form. func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } v, ok := arm64ImmOperandValue(ops[0]) if !ok { return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw) } rd := arm64RegNum(operandRegName(ops[len(ops)-1])) rn := rd if len(ops) == 3 { rn = arm64RegNum(operandRegName(ops[1])) } if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } // CMP/CMN discard the destination. if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" { rd = 31 // ZR } ws, err := arm64AddSubImmWords(mnem, v, rn, rd) if err != nil { return nil, fmt.Errorf("%s: %w", mnem, err) } return a64WordsLE(ws...), nil } // arm64AddSubImmWords returns the word sequence the toolchain emits for an // ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and // their W forms. rn and rd are resolved register numbers (a comparison // discards rd, so the caller passes 31). func arm64AddSubImmWords(mnem string, v int64, rn, rd int) ([]uint32, error) { w := strings.HasSuffix(mnem, "W") sf := uint32(1) // 64-bit d := v if w { sf = 0 // 32-bit // The W forms classify the 32-bit value (asm7.go con32class). d = int64(uint32(v)) } isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW" isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW" op := uint32(0) // ADD S := uint32(0) if isSub { op = 1 } if isS { S = 1 } single := func(sh, imm12 uint32) []uint32 { return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))} } // imm12: plain, then the one-shifted-by-12 form. if d >= 0 && d <= 0xFFF { return single(0, uint32(d)), nil } if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF { return single(1, uint32(d>>12)), nil } // ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB // split into two imm12 instructions, low half first (asm7.go case 48). // The encoding is complete in itself: no REGTMP, no register form. The S // forms must not break addition/subtraction, so the toolchain // reclassifies them and falls through to the materialisation below. dm := ^d if w { dm = ^d & 0xFFFFFFFF } _, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf)) if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon { return []uint32{ a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)), a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)), }, nil } // Constant into REGTMP (R27), then the register form. The first word // mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for // the complement form, the bitmask ORR otherwise, and the full // omovlconst sequence when no single word carries the value. var seq []uint32 switch s := arm64Movcon(d); { case s >= 0: seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)} case arm64Movcon(dm) >= 0: s := arm64Movcon(dm) seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)} case isBitcon: n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf)) seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5} default: seq = arm64MovLConst(d, sf) } // The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr). seq = append(seq, a64InstrTable[mnem].op|27<<16|uint32(rn)<<5|uint32(rd)) for i := range seq[:len(seq)-1] { seq[i] |= 27 // REGTMP } return seq, nil } // arm64MovLConst returns the toolchain's multi-word constant sequence for a // value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go // omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is // returned with the destination field clear so the caller can OR its own // register in. movcon and movcon-of-complement must fail for d before this // is reached, so no branch sees all-zero or all-0xFFFF chunks. func arm64MovLConst(d int64, sf uint32) []uint32 { if sf == 0 { // omovlconst AMOVW: both 16-bit halves, low first. return []uint32{ a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0), a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0), } } dn := ^d var immh [4]uint64 zero, neg := 0, 0 for i := range immh { immh[i] = uint64(d>>(i*16)) & 0xFFFF switch immh[i] { case 0: zero++ case 0xFFFF: neg++ } } mw := func(opc uint32, val int64, chunk int) uint32 { return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0) } var os []uint32 switch { case zero == 2: // one MOVZ and one MOVK i := 0 for ; i < 4; i++ { if immh[i] != 0 { os = append(os, mw(2, d, i)) i++ break } } for ; i < 4; i++ { if immh[i] != 0 { os = append(os, mw(3, d, i)) } } case neg == 2: // one MOVN and one MOVK i := 0 for ; i < 4; i++ { if immh[i] != 0xFFFF { os = append(os, mw(0, dn, i)) i++ break } } for ; i < 4; i++ { if immh[i] != 0xFFFF { os = append(os, mw(3, d, i)) } } default: // A two-word shortcut: a bitmask in every chunk but one, fixed up by // a single MOVK (constants from strength-reduced division). if zero == 0 && neg == 0 { for i := range 4 { mask := uint64(0xFFFF) << (i * 16) for period := 2; period <= 32; period *= 2 { x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask if n, immr, imms, ok := arm64Bitmask(x, 1); ok { os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5) os = append(os, mw(3, d, i)) return os } } } } switch { case zero >= 1: // one MOVZ and up to three MOVKs i := 0 for ; i < 4; i++ { if immh[i] != 0 { os = append(os, mw(2, d, i)) i++ break } } for ; i < 4; i++ { if immh[i] != 0 { os = append(os, mw(3, d, i)) } } case neg >= 1: // one MOVN and up to three MOVKs i := 0 for ; i < 4; i++ { if immh[i] != 0xFFFF { os = append(os, mw(0, dn, i)) i++ break } } for ; i < 4; i++ { if immh[i] != 0xFFFF { os = append(os, mw(3, d, i)) } } default: // one MOVZ and three MOVKs os = append(os, mw(2, d, 0)) for i := 1; i < 4; i++ { os = append(os, mw(3, d, i)) } } } return os } // ---- MOV pseudo-instruction ---- // encodeARM64Mov encodes the MOV family, the load/store/immediate workhorse // of Go's arm64 assembly. MOV is an alias of MOVD (the width mnemonics // select the access width). The forms, mirroring the toolchain: // // MOVx $imm, rd load immediate (MOVZ/MOVN/MOVK) // MOVx mem, rd load from memory // MOVx rd, mem store to memory // MOVx rs, rd register move (ORR Rd, ZR, Rs) // MOVx $sym(SB), rd address of a static symbol (ADRP+ADD) // MOVx sym(SB), rd load from a static symbol (ADRP+LDR) // MOVx rd, sym(SB) store to a static symbol (ADRP+STR) // // With the writeback suffix (MOVD.P, MOVD.W, …) the memory form becomes a // post-index or pre-index access whose offset is the base writeback amount. // Storing a $0 immediate stores ZR; any other immediate is rejected, matching // the toolchain. func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo, relocs *[]Reloc) ([]byte, error) { ops := instr.Operands if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } src, dst := ops[0], ops[1] if wb != "" { switch { case isMemOperand(src) && !isMemOperand(dst): rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s: invalid destination register", mnem) } return encodeARM64MemOp(mnem, src, rd, true, fi, wb) case isMemOperand(dst) && !isMemOperand(src): rs := arm64RegNum(operandRegName(src)) if rs < 0 { if !isImmOperand(src) || arm64Imm64(src) != 0 { return nil, fmt.Errorf("%s: invalid source register", mnem) } // Storing a constant zero stores the zero register. rs = 31 } return encodeARM64MemOp(mnem, dst, rs, false, fi, wb) default: return nil, fmt.Errorf("%s: writeback form needs a register and a memory operand", mnem) } } // Immediate → register (including $sym(SB)). if isImmOperand(src) && !isMemOperand(src) { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem) } return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil } // Immediate → memory: only storing zero is encodable (the ZR // register); the toolchain rejects any other immediate-to-memory // combination ("illegal combination"). if isMemOperand(dst) { if arm64Imm64(src) != 0 { return nil, fmt.Errorf("%s: illegal combination: an immediate store must be zero", mnem) } return encodeARM64MemOp(mnem, dst, 31, false, fi, "") } rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) } return encodeARM64LoadImm(rd, arm64Imm64(src), mnem) } // Static symbol load/store via ADRP. if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) { rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem) } return encodeARM64SBLoad(src.Addr.Sym, rd, mnem, relocs) } if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) { rs := arm64RegNum(operandRegName(src)) if rs < 0 { return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem) } return encodeARM64SBStore(dst.Addr.Sym, rs, mnem, relocs) } // System-register moves: MOVD NZCV, R0 reads (MRS) and MOVD R0, NZCV // writes (MSR) the flag and FP status registers. if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "" && src.Addr.Base == "" { if base, ok := a64MRSOps[src.Addr.Sym.Name]; ok { rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s %s: invalid destination register", mnem, src.Addr.Sym.Name) } return a64wordLE(base | uint32(rd)&31), nil } } if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "" && dst.Addr.Base == "" { if base, ok := a64MSRRegOps[dst.Addr.Sym.Name]; ok { rs := arm64RegNum(operandRegName(src)) if rs < 0 { return nil, fmt.Errorf("%s %s: invalid source register", mnem, dst.Addr.Sym.Name) } return a64wordLE(base | uint32(rs)&31), nil } } // Memory load/store with offset. if arm64IsMemOperand(src) && !arm64IsMemOperand(dst) { rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s: invalid destination register", mnem) } return encodeARM64MemOp(mnem, src, rd, true, fi, "") } if !arm64IsMemOperand(src) && arm64IsMemOperand(dst) { rs := arm64RegNum(operandRegName(src)) if rs < 0 { return nil, fmt.Errorf("%s: invalid source register", mnem) } return encodeARM64MemOp(mnem, dst, rs, false, fi, "") } // Register → register. return encodeARM64RegMove(mnem, src, dst) } // arm64MovSize returns the encoded size of a MOV instruction. func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { if len(ops) != 2 { return 4 } src, dst := ops[0], ops[1] switch { case isImmOperand(src): if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { return 8 // ADRP + ADD } if isMemOperand(dst) { // Only the $0 (ZR store) immediate reaches memory, in one word. return 4 } // Size the immediate exactly as the encoder will emit it: multi-chunk // values expand to up to four words and the W forms truncate first. // Anything else would desynchronise the label offsets of pass 1 from // the bytes pass 2 lays down, corrupting every later branch. b, err := encodeARM64LoadImm(31, arm64Imm64(src), mnem) if err != nil { return 4 } return len(b) case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB": return 8 // ADRP + LDR case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB": return 8 // ADRP + STR case isMemOperand(src) || isMemOperand(dst): mem := src if !isMemOperand(src) { mem = dst } _, off := arm64MemWithFrame(mem, fi) // Scaled unsigned offset fits if aligned and in range. lt, ok := a64LoadTable[mnem] if !ok { lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access } scale := int64(1) << uint(lt.size) if off >= 0 && off%scale == 0 && off/scale < 4096 { return 4 } if off >= -256 && off <= 255 { return 4 // unscaled } if _, _, _, ok := arm64SplitOffset(off, scale); ok { return 8 // ADD base, REGTMP + access } return 12 // literal pool range: encoding reports it as unsupported default: return 4 // register move } } // encodeARM64LoadImm loads an immediate into a register, matching the // toolchain's MOVZ/MOVN/MOVK sequence. W forms truncate to 32 bits first and // every classification (movcon, complement, chunk count) runs on the truncated // value, so a 32-bit immediate never reaches the 64-bit halves: MOVW $-1 // truncates to 0xFFFFFFFF, whose complement is a single zero chunk, and encodes // as MOVN W, #0. func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) { d := v sf := uint32(1) // 64-bit if mnem == "MOVW" || mnem == "MOVWU" { d = int64(uint32(v)) sf = 0 } if d == 0 { // ORR Rd, ZR, ZR (MOV $0, Rd) op := uint32(1<<31 | 1<<29 | 0x0a<<24) // ORR 64-bit if sf == 0 { op = 0<<31 | 1<<29 | 0x0a<<24 // ORR 32-bit } return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil } // The Go toolchain classifies immediates (asm7.go conclass): // - inside the imm12/shifted-imm12 "addcon" band (C_ABCON0/C_ABCON, // 0 < v ≤ 4095 or a 4096 multiple up to 0xFFF000): bitmask first, so // `MOVD $4096, R27` is ORR $4096, not MOVZ $(1<<12) // - outside that band: MOVZ/MOVN first (C_MOVCON before C_BITCON), and // negative values reach MOVN before the bitmask test tryBitmaskFirst := d > 0 && (d <= 0xFFF || (d&0xFFF == 0 && d <= 0xFFF000)) if tryBitmaskFirst { // Addcon-band immediate: try bitmask first (Go uses ORR for values // like $1, $256 and $65536). N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf)) if ok { return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil } } // Try MOVZ (single non-zero 16-bit chunk) and MOVN (single non-0xFFFF // chunk of the complement). The W forms must look inside the 32-bit // window only, so the complement is masked to the operand width; d is // already truncated and needs no mask. width := uint64(0xFFFFFFFF) if sf == 1 { width = 0xFFFFFFFFFFFFFFFF } s := arm64Movcon(d) if s >= 0 { return a64wordLE(a64MoveWide(sf, 2, uint32(s>>4), uint32((d>>uint(s))&0xFFFF), uint32(rd))), nil } sn := arm64Movcon(^d & int64(width)) if sn >= 0 { return a64wordLE(a64MoveWide(sf, 0, uint32(sn>>4), uint32(((^d)>>uint(sn))&0xFFFF), uint32(rd))), nil } // For values outside the bitmask-first range that are not movcon: try bitmask. if !tryBitmaskFirst { N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf)) if ok { return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil } } // Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN // for the first special 16-bit chunk, then MOVK per remaining one, with // the bitmask-plus-fixup shortcut for strength-reduced constants). ws := arm64MovLConst(d, sf) for i := range ws { ws[i] |= uint32(rd) } return a64WordsLE(ws...), nil } // arm64Bitmask checks whether a value can be encoded as an AArch64 logical // immediate (bitmask). Returns the N, immr, imms fields and true if // representable. sf is 0 for 32-bit or 1 for 64-bit. func arm64Bitmask(v uint64, sf int) (N, immr, imms uint32, ok bool) { if v == 0 { return } maxElem := uint(6) // 2^6 = 64 if sf == 0 { maxElem = 5 // 2^5 = 32 v &= 0xFFFFFFFF } for e := uint(0); e < maxElem; e++ { esize := uint(1) << (e + 1) // 2, 4, 8, 16, 32, 64 emask := uint64(1<> r) | ((pattern << (esize - r)) & emask) if rotated == 0 { continue } // Count trailing 1s (contiguous block of 1s from bit 0). tz := uint(0) tmp := ^rotated for tmp&1 == 0 && tz < esize { tz++ tmp >>= 1 } if tz == 0 || tz >= esize { continue } mask := uint64(1< 255 { return nil, fmt.Errorf("%s: writeback offset %d out of range (-256..255)", mnem, off) } w := a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg) if wb == "P" { w |= 1 << 10 // post-index } else { w |= 3 << 10 // pre-index } return a64wordLE(w), nil } // Scaled unsigned offset first, then the unscaled ±255 form. if off >= 0 && off%scale == 0 && off/scale < 4096 { return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil } if off >= -256 && off <= 255 { return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg)), nil } // Large offset: materialise the base in REGTMP (R27) the way the // toolchain does and access what remains. The ADD offsets from the // operand's own base register, [SP] and [Rn] alike. addImm, addShift, access, ok := arm64SplitOffset(off, scale) if !ok { return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off) } return a64WordsLE( a64AddSub(1, 0, 0, addShift, addImm, uint32(rn), 27), // ADD $addImm<= 0 && rest>>12 <= 4095 { return uint32(rest >> 12), 1, off - rest, true } return 0, 0, 0, false } // ---- static symbol references (ADRP + offset) ---- // encodeARM64SBAddr emits ADRP Rd, 0; ADD Rd, Rd, 0 with the // R_ADDRARM64 relocation pair, loading a symbol's address. func encodeARM64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset}, ) } return a64WordsLE( a64ADR(1, 0, 0, uint32(rd)), // ADRP Rd, 0 a64AddSub(1, 0, 0, 0, 0, uint32(rd), uint32(rd)), // ADD $0, Rd, Rd ) } // encodeARM64SBLoad emits ADRP R27, 0; LDR Rd, [R27, 0] with relocations, // matching the toolchain: the scratch register is REGTMP (R27) and the pair // carries R_ARM64_PCREL_LDST64. func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([]byte, error) { lt, ok := a64LoadTable[mnem] if !ok { lt = a64LoadTable["MOVD"] } if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset}, ) } return a64WordsLE( a64ADR(1, 0, 0, 27), // ADRP R27, 0 a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 27, uint32(rd)), // LDR Rd, [R27, #0] ), nil } // encodeARM64SBStore emits ADRP R27, 0; STR Rs, [R27, 0] with relocations, // matching the toolchain's R27 scratch and R_ARM64_PCREL_LDST64 pair. func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ([]byte, error) { lt, ok := a64LoadTable[mnem] if !ok { lt = a64LoadTable["MOVD"] } storeOpc := a64StoreOpc(lt) if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset}, ) } return a64WordsLE( a64ADR(1, 0, 0, 27), // ADRP R27, 0 a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 27, uint32(rs)), // STR Rs, [R27, #0] ), nil } // ---- operand helpers ---- // arm64Imm64 returns the full 64-bit immediate value of an operand. func arm64Imm64(op *ast.Operand) int64 { if op.Imm.HasVal { v := op.Imm.Val if op.Imm.Neg { v = -v } return v } return 0 } // arm64ImmOperandValue returns the immediate an operand stands for, falling // back to a raw evaluation for the spellings the parser leaves unevaluated: // the one's-complement form $~n and parenthesised constant expressions. // The second result reports whether a value could be recovered. func arm64ImmOperandValue(op *ast.Operand) (int64, bool) { if op.Imm.HasVal { v := op.Imm.Val if op.Imm.Neg { v = -v } return v, true } s := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(op.Raw), "$")) inverted := false if i := strings.IndexByte(s, '~'); i >= 0 { inverted = true s = s[i+1:] } v, ok := arm64EvalExpr(s) if !ok { return 0, false } if inverted { v = ^v } return v, true } // arm64MemWithFrame resolves a memory operand, translating FP/SP pseudo- // registers via the frame mapping. The offset stays 64-bit: the AST carries // int64 displacements and truncating here would wrap offsets beyond 2^31 // silently. func arm64MemWithFrame(op *ast.Operand, fi arm64FrameInfo) (rn int, off int64) { if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { base, pseudo := arm64ResolvePseudo(op.Addr.Sym, fi) return base, int64(pseudo) } if rn, off, ok := arm64ExprMem(op); ok { return rn, off } return arm64RegNum(op.Addr.Base), op.Addr.Offset } // arm64IsMemOperand is the arm64-side memory test: the shared syntactic test // plus the parenthesised-expression form (8*1)(RSP), which the parser leaves // unstructured (empty base) because the offset is not a plain integer. func arm64IsMemOperand(op *ast.Operand) bool { if isMemOperand(op) { return true } _, _, ok := arm64ExprMem(op) return ok } // arm64ExprMem recovers a base register and an evaluated offset from a // parenthesised-expression memory operand such as (8*22)(RSP) or (0*8)(R0). // It reports ok=false for anything else. func arm64ExprMem(op *ast.Operand) (rn int, off int64, ok bool) { if op.Addr.Base != "" || op.Addr.Sym != nil { return 0, 0, false } s := strings.Join(strings.Fields(op.Raw), " ") if !strings.HasSuffix(s, ")") { return 0, 0, false } // Split the trailing "( REG )" from the leading "( EXPR )". inner := strings.LastIndex(s, "(") if inner <= 0 { return 0, 0, false } regPart := strings.TrimSpace(s[inner+1 : len(s)-1]) head := strings.TrimSpace(s[:inner]) if !strings.HasPrefix(head, "(") || !strings.HasSuffix(head, ")") { return 0, 0, false } expr := strings.TrimSpace(head[1 : len(head)-1]) v, ok := arm64EvalExpr(expr) if !ok { return 0, 0, false } rn = arm64RegNum(regPart) if rn < 0 { return 0, 0, false } return rn, v, true } // arm64EvalExpr evaluates the Plan 9 constant arithmetic the assembler // accepts inside memory operands: integers with unary minus and the + - * << // >> & | ^ operators. Operator precedence follows the Plan 9 convention // (shifts bind tighter than +, * tighter than shifts); expressions it cannot // fully reduce report ok=false. func arm64EvalExpr(s string) (int64, bool) { type parser struct { toks []string pos int } var scan func(string) []string scan = func(s string) []string { var out []string for s = strings.TrimSpace(s); s != ""; s = strings.TrimSpace(s) { switch { case s[0] == '(' || s[0] == ')': out = append(out, s[:1]) s = s[1:] case s[0] >= '0' && s[0] <= '9': i := 0 for i < len(s) && ((s[i] >= '0' && s[i] <= '9') || s[i] == 'x' || s[i] == 'X' || s[i] >= 'a' && s[i] <= 'f' || s[i] >= 'A' && s[i] <= 'F') { i++ } out = append(out, s[:i]) s = s[i:] case strings.HasPrefix(s, "<<"), strings.HasPrefix(s, ">>"): out = append(out, s[:2]) s = s[2:] case strings.IndexByte("+-*&|^~", s[0]) >= 0: out = append(out, s[:1]) s = s[1:] default: return nil } } return out } toks := scan(s) if toks == nil { return 0, false } p := &parser{toks: toks} var primary func() (int64, bool) var expr func() (int64, bool) primary = func() (int64, bool) { if p.pos >= len(p.toks) { return 0, false } t := p.toks[p.pos] switch { case t == "-": p.pos++ v, ok := primary() return -v, ok case t == "+": p.pos++ return primary() case t == "~": p.pos++ v, ok := primary() return ^v, ok case t == "(": p.pos++ v, ok := expr() if !ok || p.pos >= len(p.toks) || p.toks[p.pos] != ")" { return 0, false } p.pos++ return v, true } if t[0] < '0' || t[0] > '9' { return 0, false } v, err := strconv.ParseInt(strings.TrimPrefix(strings.TrimPrefix(t, "0X"), "0x"), 0, 64) if err != nil { return 0, false } p.pos++ return v, true } var binop func(minLevel int) (int64, bool) level := func(op string) int { switch op { case "|", "^": return 1 case "&": return 2 case "<<", ">>": return 3 case "*": return 4 case "+", "-": return 5 } return 0 } var apply func(v int64, op string, w int64) (int64, bool) apply = func(v int64, op string, w int64) (int64, bool) { switch op { case "+": return v + w, true case "-": return v - w, true case "*": return v * w, true case "<<": return v << uint(w), true case ">>": return v >> uint(w), true case "&": return v & w, true case "|": return v | w, true case "^": return v ^ w, true } return 0, false } binop = func(minLevel int) (int64, bool) { v, ok := primary() if !ok { return 0, false } for p.pos < len(p.toks) { op := p.toks[p.pos] lv := level(op) if lv == 0 || lv < minLevel { break } p.pos++ w, ok := binop(lv + 1) if !ok { return 0, false } v, ok = apply(v, op, w) if !ok { return 0, false } } return v, true } expr = func() (int64, bool) { return binop(1) } v, ok := binop(1) if !ok || p.pos != len(p.toks) { return 0, false } return v, true } // arm64Label returns the label name of an operand. func arm64Label(op *ast.Operand) string { if op.Addr.Sym != nil { return op.Addr.Sym.Name } return op.Raw } // ---- FP instruction encoding ---- // encodeARM64FP3 encodes a FP 3-operand instruction (Rm, Rn, Rd). // FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL. The two-operand spelling // (FMULD F3, F5: multiply into the second operand) folds Rn into Rd. func encodeARM64FP3(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 3 && len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } rm := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[len(ops)-1])) rn := rd if len(ops) == 3 { rn = arm64RegNum(operandRegName(ops[1])) } if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64FPUnary encodes a FP unary instruction (Rn, Rd). // FMOV reg-reg, FABS, FNEG, FSQRT, FCVT cross-precision, FRINT*. func encodeARM64FPUnary(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rn := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[1])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64FP4 encodes a FP 4-operand FMA instruction (Ra, Rm, Rn, Rd). // FMADD, FMSUB, FNMADD, FNMSUB. func encodeARM64FP4(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { var ra, rm, rn, rd int switch len(ops) { case 4: ra = arm64RegNum(operandRegName(ops[0])) rm = arm64RegNum(operandRegName(ops[1])) rn = arm64RegNum(operandRegName(ops[2])) rd = arm64RegNum(operandRegName(ops[3])) case 3: // 3-operand form: Fa, Fm, Fd → Fd = Fa ± Fd*Fm (Rn = Rd) ra = arm64RegNum(operandRegName(ops[0])) rm = arm64RegNum(operandRegName(ops[1])) rd = arm64RegNum(operandRegName(ops[2])) rn = rd default: return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops)) } if ra < 0 || rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(ra)<<16 | uint32(rm)<<10 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64FPCmp encodes a FP compare instruction. // Go assembler syntax: FCMP Fn, Fm (register) or FCMP $0.0, Fn (compare with zero). // ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5]. // Go puts first operand → Rm, second → Rn. func encodeARM64FPCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } // Check if first operand is #0 (compare with zero): FCMP $0.0, Fn. if isImmOperand(ops[0]) && arm64Imm64(ops[0]) == 0 { rn := arm64RegNum(operandRegName(ops[1])) if rn < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } // For compare with zero: Rm=0, op2 bit 3 set (|= 8). return a64wordLE((baseOp | 8) | 0<<16 | uint32(rn)<<5), nil } // Register compare: FCMP Fn, Fm. // Go puts first operand in Rm field, second in Rn field. rm := arm64RegNum(operandRegName(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) if rm < 0 || rn < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5), nil } // encodeARM64FPCCmp encodes a FP conditional compare. // Go assembler syntax: FCCMP cond, Fn, Fm, $nzcv // ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5]. // Go puts ops[1] in Rm field, ops[2] in Rn field. func encodeARM64FPCCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } // Go puts ops[1] in Rm (bits 20:16), ops[2] in Rn (bits 9:5). rm := arm64RegNum(operandRegName(ops[1])) rn := arm64RegNum(operandRegName(ops[2])) if rm < 0 || rn < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } nzcv := arm64Imm64(ops[3]) if nzcv < 0 || nzcv > 0xF { return nil, fmt.Errorf("%s: nzcv %d out of range (0..15)", mnem, nzcv) } return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(nzcv)&0xF), nil } // encodeARM64FPSel encodes a FP conditional select. // Go assembler syntax: FCSEL cond, Fn, Fm, Fd func encodeARM64FPSel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } // Operand order: cond, Fn, Fm, Fd condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } rn := arm64RegNum(operandRegName(ops[1])) rm := arm64RegNum(operandRegName(ops[2])) rd := arm64RegNum(operandRegName(ops[3])) if rn < 0 || rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64FPCvt encodes a FP ↔ integer conversion instruction. // The operand order depends on direction: FCVTZS Fd, Rn (FP→int) or SCVTF Rd, Fn (int→FP). func encodeARM64FPCvt(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } src := arm64RegNum(operandRegName(ops[0])) dst := arm64RegNum(operandRegName(ops[1])) if src < 0 || dst < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(src)<<5 | uint32(dst)), nil } // encodeARM64CSEL encodes a conditional select instruction. // CSEL Rm, Rn, Rd, cond (4 operands) or CSET Rd, cond (2 operands). func encodeARM64CSEL(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { isAlias := mnem == "CSET" || mnem == "CSETW" || mnem == "CSETM" || mnem == "CSETMW" || mnem == "CINC" || mnem == "CINCW" || mnem == "CINV" || mnem == "CINVW" || mnem == "CNEG" || mnem == "CNEGW" if isAlias { is2op := mnem == "CSET" || mnem == "CSETW" || mnem == "CSETM" || mnem == "CSETMW" if is2op { // CSET cond, Rd → CSEL XZR, XZR, Rd, inverted_cond if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } rd := arm64RegNum(operandRegName(ops[1])) if rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } invCond := cond ^ 1 return a64wordLE(baseOp | 31<<16 | invCond<<12 | 31<<5 | uint32(rd)), nil } // CINC cond, Rn, Rd → CSINC Rn, Rn, Rd, inverted_cond if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[2])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } invCond := cond ^ 1 return a64wordLE(baseOp | uint32(rn)<<16 | invCond<<12 | uint32(rn)<<5 | uint32(rd)), nil } // CSEL cond, Rn, Rm, Rd (4 operands), condition first. // Go assembler syntax: CSEL cond, Rn, Rm, Rd // ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5], Rd in bits[4:0]. if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } rn := arm64RegNum(operandRegName(ops[1])) rm := arm64RegNum(operandRegName(ops[2])) rd := arm64RegNum(operandRegName(ops[3])) if rn < 0 || rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64CRC32 encodes a CRC32 instruction. // Go assembler syntax: CRC32B Rm, Rd (2 operands, Rn=Rd). func encodeARM64CRC32(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) == 3 { // 3-operand form: CRC32B Rm, Rn, Rd → use Rm and Rd, Rn=Rd. rm := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[2])) if rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil } if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } rm := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[1])) if rm < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil } // ---- Atomics encoding ---- // arm64ExclMem resolves the memory operand of an exclusive or atomic // instruction. These encodings have no immediate field: the toolchain // rejects `LDXR 8(R1), R2` as an illegal combination, so a non-zero offset is // reported rather than silently dropped (which would read the wrong address). func arm64ExclMem(mnem string, op *ast.Operand) (int, error) { rn, off := arm64MemWithFrame(op, arm64FrameInfo{}) if rn < 0 { return 0, fmt.Errorf("invalid memory operand in %s", mnem) } if off != 0 { return 0, fmt.Errorf("%s: offset %d not supported, exclusive and atomic accesses take a plain (Rn) operand", mnem, off) } return rn, nil } // arm64PairOf parses a register-pair operand `(R1, R2)`, reporting false // when the operand is not a pair. The toolchain takes the second register of // the pair from the operand's Offset (its C_PAIR class, // cmd/internal/obj/arm64/asm7.go cases 58/59). func arm64PairOf(op *ast.Operand) (int, int, bool) { raw := strings.TrimSpace(op.Raw) if !strings.HasPrefix(raw, "(") || !strings.HasSuffix(raw, ")") { return -1, -1, false } parts := strings.Split(raw[1:len(raw)-1], ",") if len(parts) != 2 { return -1, -1, false } r1 := arm64RegNum(strings.TrimSpace(parts[0])) r2 := arm64RegNum(strings.TrimSpace(parts[1])) if r1 < 0 || r2 < 0 { return -1, -1, false } return r1, r2, true } // encodeARM64Excl encodes the exclusive load/store family with the operand // order the toolchain parses (cmd/internal/obj/arm64/asm7.go cases 58 and 59, // and its own spellings in arm64enc.s): // // STXR Rt, (Rn), Rs store, single register // STXP (Rt1, Rt2), (Rn), Rs store, register pair // LDXR (Rn), Rt load, single register // LDXP (Rn), (Rt1, Rt2) load, register pair // // Decoded toolchain evidence: `STXR R1, (R2), R3` assembles to 0xc8037c41, // whose fields are Rs=3, Rn=2, Rt=1: the FIRST register operand is the data // register and the LAST the status register. func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { isLoad := strings.HasPrefix(mnem, "LD") if isLoad { // LDXR (Rn), Rt / LDXP (Rn), (Rt1, Rt2): 2 operands. if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rn, err := arm64ExclMem(mnem, ops[0]) if err != nil { return nil, err } if rt1, rt2, ok := arm64PairOf(ops[1]); ok { // The single-register opcodes pre-set the unused Rs (bits 20:16) // and Rt2 (bits 14:10) fields to 31; the pair forms carry a real // Rt2 and keep Rs at 31. return a64wordLE(baseOp | 0x1F<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil } rt := arm64RegNum(operandRegName(ops[1])) if rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil } // STXR Rt, (Rn), Rs / STXP (Rt1, Rt2), (Rn), Rs: 3 operands. if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rn, err := arm64ExclMem(mnem, ops[1]) if err != nil { return nil, err } rs := arm64RegNum(operandRegName(ops[2])) if rs < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } if rt1, rt2, ok := arm64PairOf(ops[0]); ok { return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil } rt := arm64RegNum(operandRegName(ops[0])) if rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil } // encodeARM64LSEAtom encodes an LSE atomic instruction (LDADD, CAS, SWP). // LDADD Rs, (Rn), Rt → 3 operands: Rs, mem, Rt // CAS Rs, (Rn), Rt → 3 operands: Rs, mem, Rt func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rs := arm64RegNum(operandRegName(ops[0])) if rs < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } rn, err := arm64ExclMem(mnem, ops[1]) if err != nil { return nil, err } rt := arm64RegNum(operandRegName(ops[2])) if rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil } // encodeARM64DP1 encodes a data-processing (1 source) instruction: // RBIT, REV, CLZ and friends take (Rn, Rd), word = base | Rn<<5 | Rd. func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rn := arm64RegNum(operandRegName(ops[0])) rd := arm64RegNum(operandRegName(ops[1])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64ADR encodes ADR/ADRP: (label, Rd), the byte distance to the // target split into immlo (bits 30:29) and immhi (bits 23:5), with bit 31 // selecting the page form. func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rd := arm64RegNum(operandRegName(ops[1])) if rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } target := resolve(arm64Label(ops[0])) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q", target) } rel := int64(targetOff - pc) if rel < -(1<<20) || rel >= 1<<20 { return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target) } return a64wordLE(a64ADR(page, int32(rel>>2), uint32(rel)&3, uint32(rd))), nil } // encodeARM64Bitfield2 encodes UBFX/SBFX ($lsb, Rn, $width, Rd): immr // carries the lsb (six bits with the N flag on the X forms) and imms the // lsb plus width minus one. A sum beyond the register width is the // toolchain's "illegal bit number" error. func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) { return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem) } lsb := arm64Imm64(ops[0]) width := arm64Imm64(ops[2]) rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[3])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } bits := int64(32) << (baseOp >> 31 & 1) if lsb < 0 || lsb >= bits { return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1) } if width < 1 || lsb+width > bits { return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits) } immr := uint32(lsb) n := uint32(0) if immr >= 32 { n = 1 << 21 immr &^= 32 } return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64BitfieldAlias encodes the four-operand bitfield aliases // ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the // field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb // with imms = lsb+width-1. func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) { return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem) } lsb := arm64Imm64(ops[0]) width := arm64Imm64(ops[2]) rn := arm64RegNum(operandRegName(ops[1])) rd := arm64RegNum(operandRegName(ops[3])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } bits := int64(32) << (baseOp >> 31 & 1) if lsb < 0 || lsb >= bits { return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1) } if width < 1 || width > bits || lsb+width > bits { return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits) } var immr, imms int64 switch mnem { case "BFXIL", "BFXILW": immr, imms = lsb, lsb+width-1 default: // BFI, SBFIZ, UBFIZ immr, imms = (-lsb)%bits, width-1 } return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The // third field carries Rm or a 5-bit immediate in the same bits, at the // toolchain's choice of register or immediate operand. func encodeARM64CondCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 4 || !isImmOperand(ops[3]) { return nil, fmt.Errorf("%s expects 4 operands (cond, Rn, Rm|$imm, $nzcv)", mnem) } condName := operandRegName(ops[0]) cond, ok := arm64CondMap[condName] if !ok { return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) } rn := arm64RegNum(operandRegName(ops[1])) if rn < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } var v2 uint32 if isImmOperand(ops[2]) { v := arm64Imm64(ops[2]) if v < 0 || v > 31 { return nil, fmt.Errorf("%s: immediate %d out of range (0..31)", mnem, v) } v2 = uint32(v) } else { rm := arm64RegNum(operandRegName(ops[2])) if rm < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } v2 = uint32(rm) } nzcv := arm64Imm64(ops[3]) if nzcv < 0 || nzcv > 15 { return nil, fmt.Errorf("%s: nzcv %d out of range (0..15)", mnem, nzcv) } // Bit 11 carries the immediate-vs-register choice for the third field. op2 := uint32(0) if isImmOperand(ops[2]) { op2 = 1 << 11 } return a64wordLE(baseOp | v2<<16 | cond<<12 | op2 | uint32(rn)<<5 | uint32(nzcv)&0xF), nil } // encodeARM64Branch19 encodes CBZ/CBNZ: (Rt, label), // word = base | imm19<<5 | Rt with imm19 = (target - pc) >> 2. // arm64PCRelOffset recognises the toolchain's forward branch spelling // n(PC): the number counts INSTRUCTIONS from the branch itself. It reports // ok for operands spelled that way and leaves everything else alone. func arm64PCRelOffset(op *ast.Operand) (int, bool) { if op.Addr.Base != "PC" && !(op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") { return 0, false } off := op.Addr.Offset if !op.Addr.HasOff { off = 0 } return int(off), true } func encodeARM64Branch19(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rt := arm64RegNum(operandRegName(ops[0])) if rt < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } rel, pcRel := arm64PCRelOffset(ops[1]) if !pcRel { target := resolve(arm64Label(ops[1])) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q", target) } rel = (targetOff - pc) >> 2 } if rel < -(1<<18) || rel >= (1<<18) { return nil, fmt.Errorf("%s: branch offset %d out of 19-bit range", mnem, rel) } return a64wordLE(baseOp | uint32(rel)&0x7FFFF<<5 | uint32(rt)), nil } // encodeARM64TestBranch encodes TBZ/TBNZ: ($bit, Rt, label). Bits 32 to 63 // set the b5 flag at bit 31; there is one mnemonic per polarity, no width // suffix. func encodeARM64TestBranch(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { if len(ops) != 3 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects 3 operands ($bit, Rt, label)", mnem) } bit := arm64Imm64(ops[0]) if bit < 0 || bit > 63 { return nil, fmt.Errorf("%s: bit number %d out of range (0..63)", mnem, bit) } rt := arm64RegNum(operandRegName(ops[1])) if rt < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } rel, pcRel := arm64PCRelOffset(ops[2]) if !pcRel { target := resolve(arm64Label(ops[2])) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q", target) } rel = (targetOff - pc) >> 2 } if rel < -(1<<13) || rel >= (1<<13) { return nil, fmt.Errorf("%s: branch offset %d out of 14-bit range", mnem, rel) } return a64wordLE(baseOp | uint32(bit>>5)<<31 | uint32(bit&31)<<19 | uint32(rel)&0x3FFF<<5 | uint32(rt)), nil } // encodeARM64Pair encodes load/store pair instructions. Loads spell // (mem, (Rt1, Rt2)), stores (Rt1, Rt2), mem; the scaled immediate rides // imm7 at bits 21:15 and must fit -64..63 after division by the access // size (8 bytes for the D forms, 4 for the W forms). // encodeARM64Pair encodes the load/store pair family. wb selects the // addressing mode: "" the signed-offset form, "P" post-index, "W" pre-index; // in the writeback forms the immediate is the amount added to the base // register around the access. func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64FrameInfo, wb string, relocs *[]Reloc) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } scale := int64(8) if strings.HasSuffix(mnem, "W") { scale = 4 } load := strings.Contains(mnem, "LDP") memOp, pairOp := ops[0], ops[1] if !load { memOp, pairOp = ops[1], ops[0] } if !arm64IsMemOperand(memOp) { return nil, fmt.Errorf("%s: invalid memory operand", mnem) } rt1, rt2, ok := arm64PairOf(pairOp) if !ok { return nil, fmt.Errorf("%s expects a register pair (Rt1, Rt2)", mnem) } // Pair access against a static symbol: ADRP R27, sym; ADD R27, R27, #lo; // LDP/STP (R27), (Rt1, Rt2), with the R_ADDRARM64 pair riding the first // two words, exactly like the toolchain lays it out. if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo == "SB" { if wb != "" { return nil, fmt.Errorf("%s: writeback is not supported on a symbol operand", mnem) } if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: memOp.Addr.Sym.Name, Kind: RelArm64Addr, Addend: memOp.Addr.Sym.Offset}, Reloc{Off: 4, After: 4, Name: memOp.Addr.Sym.Name, Kind: RelArm64Addr, Addend: memOp.Addr.Sym.Offset}, ) } return a64WordsLE( a64ADR(1, 0, 0, 27), // ADRP R27, 0 a64AddSub(1, 0, 0, 0, 0, 27, 27), // ADD $0, R27, R27 baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (Rt1, Rt2) ), nil } rn, off := arm64MemWithFrame(memOp, fi) if rn < 0 { return nil, fmt.Errorf("%s: invalid memory operand", mnem) } if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo != "" && wb != "" { return nil, fmt.Errorf("%s: writeback is not supported on a frame-relative operand", mnem) } if off%scale != 0 || off < -64*scale || off > 63*scale { return nil, fmt.Errorf("%s: offset %d out of pair range or not a multiple of %d", mnem, off, scale) } imm7 := off / scale // The signed-offset form carries bits 24:23 = 10; post-index drops bit 24 // and pre-index sets both, with the base carrying the opc, V and L halves. switch wb { case "P": baseOp = baseOp&^(1<<24) | 1<<23 case "W": baseOp |= 1 << 23 } return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil } // encodeARM64AcqRel encodes the acquire/release loads and stores. Loads // spell (Rn), Rt; stores Rt, (Rn). Both lay the base register at bits 9:5 // and the data register at bits 4:0 over a base that carries no offset // field, so a displaced operand is reported the way arm64ExclMem reports // one for the exclusive family. func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } memOp, regOp := ops[0], ops[1] if !strings.HasPrefix(mnem, "LD") { memOp, regOp = ops[1], ops[0] } rn, err := arm64ExclMem(mnem, memOp) if err != nil { return nil, err } rt := arm64RegNum(operandRegName(regOp)) if rt < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil } // encodeARM64Sys encodes the system operations: // // BRK [$imm16] SVC $imm16 // DMB|DSB|ISB $imm4 DC , Rn // MRS , Rd MSR $imm4, // PRFM (Rn), $imm| func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { // Operand-less returns and pointer-authentication hints. if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0, "AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff, "AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f, "YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f, "SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok { if len(ops) != 0 { return nil, fmt.Errorf("%s expects no operand", mnem) } return a64wordLE(w), nil } switch mnem { case "BRK", "SVC": base := uint32(0xd4200000) if mnem == "SVC" { base = 0xd4000001 } if len(ops) == 0 { return a64wordLE(base), nil } if len(ops) != 1 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects no operand or $immediate", mnem) } v := arm64Imm64(ops[0]) if v < 0 || v > 0xFFFF { return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v) } return a64wordLE(base | uint32(v)<<5), nil case "DMB", "DSB", "ISB", "CLREX": if len(ops) != 1 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects $immediate", mnem) } v := arm64Imm64(ops[0]) if v < 0 || v > 15 { return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v) } base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem] return a64wordLE(base | uint32(v)<<8), nil case "HINT": if len(ops) != 1 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects $immediate", mnem) } v := arm64Imm64(ops[0]) if v < 0 || v > 127 { return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v) } return a64wordLE(0xd503201f | uint32(v)<<5), nil case "BTI": // The toolchain requires the landing-pad kind: bare BTI is // rejected ("missing operand"), and only the uppercase C/J/JC // spellings assemble (0xd503245f/49f/4df). if len(ops) != 1 { return nil, fmt.Errorf("%s expects C, J, or JC", mnem) } op := operandRegName(ops[0]) base, ok := map[string]uint32{"C": 0xd503245f, "J": 0xd503249f, "JC": 0xd50324df}[op] if !ok { return nil, fmt.Errorf("%s: unknown kind %q", mnem, op) } return a64wordLE(base), nil case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3": if len(ops) != 1 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects $immediate", mnem) } v := arm64Imm64(ops[0]) if v < 0 || v > 0xFFFF { return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v) } base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003, "HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002, "DCPS3": 0xd4a00003}[mnem] return a64wordLE(base | uint32(v)<<5), nil case "DC": if len(ops) != 2 { return nil, fmt.Errorf("DC expects , Rn") } base, ok := a64DCOps[operandRegName(ops[0])] if !ok { return nil, fmt.Errorf("DC: unknown cache operation %q", operandRegName(ops[0])) } rn := arm64RegNum(operandRegName(ops[1])) if rn < 0 { return nil, fmt.Errorf("DC: invalid register operand") } return a64wordLE(base | uint32(rn)&31), nil case "MRS": if len(ops) != 2 { return nil, fmt.Errorf("MRS expects , Rd") } base, ok := a64MRSOps[operandRegName(ops[0])] if !ok { return nil, fmt.Errorf("MRS: unknown system register %q", operandRegName(ops[0])) } rd := arm64RegNum(operandRegName(ops[1])) if rd < 0 { return nil, fmt.Errorf("MRS: invalid register operand") } return a64wordLE(base | uint32(rd)&31), nil case "MSR": if len(ops) != 2 { return nil, fmt.Errorf("MSR expects $immediate, or Rn, ") } if !isImmOperand(ops[0]) { // Register form: MSR Rn, (the a64MSRRegOps words). base, ok := a64MSRRegOps[operandRegName(ops[1])] if !ok { return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1])) } rs := arm64RegNum(operandRegName(ops[0])) if rs < 0 { return nil, fmt.Errorf("MSR: invalid source register") } return a64wordLE(base | uint32(rs)&31), nil } base, ok := a64MSROps[operandRegName(ops[1])] if !ok { return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1])) } v := arm64Imm64(ops[0]) if v < 0 || v > 15 { return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v) } return a64wordLE(base | uint32(v)<<8 | 31), nil case "PRFM": if len(ops) != 2 { return nil, fmt.Errorf("PRFM expects (Rn), $immediate|") } rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{}) if rn < 0 || off != 0 { return nil, fmt.Errorf("PRFM: invalid memory operand") } var prfop int64 if isImmOperand(ops[1]) { prfop = arm64Imm64(ops[1]) if prfop < 0 || prfop > 31 { return nil, fmt.Errorf("PRFM: immediate %d out of range (0..31)", prfop) } } else { p, ok := a64PRFOps[operandRegName(ops[1])] if !ok { return nil, fmt.Errorf("PRFM: unknown prefetch operation %q", operandRegName(ops[1])) } prfop = int64(p) } return a64wordLE(0xf9800000 | uint32(rn)<<5 | uint32(prfop)), nil } return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) } // encodeARM64Crypto encodes the crypto instructions. Two-register forms // spell (Rn, Rd), three-register forms (Rm, Rn, Rd); the arrangements, when // spelled, must match the instruction's own (B16 for AES, S4 for the SHA1 // and SHA256 families, D2 for SHA512). func encodeARM64Crypto(mnem string, baseOp uint32, ops []*ast.Operand, n int) ([]byte, error) { if len(ops) != n { return nil, fmt.Errorf("%s expects %d operands, got %d", mnem, n, len(ops)) } vs := make([]a64Vec, n) for i, op := range ops { v, ok := arm64VecOf(op) if !ok || v.hasIdx { return nil, fmt.Errorf("invalid register operand in %s", mnem) } if v.arr != "" && a64ArrIndex(v.arr) != a64CryptoArr[mnem] { return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, v.arr) } vs[i] = v } if n == 2 { return a64wordLE(baseOp | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil } return a64wordLE(baseOp | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil } // encodeARM64MoveWide encodes a standalone MOVK: ($value, Rd) with the value // sitting in one 16-bit chunk, the chunk's position becoming the hw field. func encodeARM64MoveWide(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects $value, Rd", mnem) } rd := arm64RegNum(operandRegName(ops[1])) if rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } // The opcode (MOVN 0, MOVZ 2, MOVK 3) and the width ride the table // base, so MOVZ and MOVN come along for free. opc := baseOp >> 29 & 3 sf := baseOp >> 31 & 1 // The toolchain's optab case 33, shared by the whole family in both // widths: the immediate is one unsigned 64-bit pattern (a high-lane // constant such as $(40000<<48) arrives negative through int64 // folding), it must occupy exactly one 16-bit lane, zero is rejected, // and the W forms cannot reach the top half. u := uint64(arm64Imm64(ops[0])) if u == 0 { return nil, fmt.Errorf("%s: zero immediate cannot be handled", mnem) } hw := -1 for lane := range 4 { if u&^(uint64(0xFFFF)<<(lane*16)) == 0 { hw = lane break } } if hw < 0 { return nil, fmt.Errorf("%s: immediate %#x does not fit one 16-bit chunk", mnem, u) } if sf == 0 && hw > 1 { return nil, fmt.Errorf("%s: immediate %#x out of range for the 32-bit form", mnem, u) } return a64wordLE(a64MoveWide(sf, opc, uint32(hw), uint32(u>>uint(hw*16)&0xFFFF), uint32(rd))), nil } // ---- Bitfield/EXTR encoding ---- // encodeARM64Bitfield encodes a bitfield instruction. // BFI/BFXIL/SBFM/UBFM $immr, Rn, $imms, Rd → 4 operands func encodeARM64Bitfield(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { // BFI/BFXIL/SBFM/UBFM: 4 operands ($immr, Rn, $imms, Rd) if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } immr := arm64Imm64(ops[0]) rn := arm64RegNum(operandRegName(ops[1])) imms := arm64Imm64(ops[2]) rd := arm64RegNum(operandRegName(ops[3])) if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } // The toolchain rejects bit numbers at or above the operand width, which // sf (bit 31 of the base) selects: 64 when set, 32 otherwise. width := uint32(32) << (baseOp >> 31 & 1) if immr < 0 || uint32(immr) >= width || imms < 0 || uint32(imms) >= width { return nil, fmt.Errorf("%s: bit number out of range (immr=%d imms=%d, width=%d)", mnem, immr, imms, width) } return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil } // encodeARM64Extr encodes an EXTR instruction. // EXTR $lsb, Rm, Rn, Rd → 4 operands func encodeARM64Extr(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } lsb := arm64Imm64(ops[0]) rm := arm64RegNum(operandRegName(ops[1])) rn := arm64RegNum(operandRegName(ops[2])) rd := arm64RegNum(operandRegName(ops[3])) if rm < 0 || rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } // The imms field is 6 bits and must stay below the operand width, which // sf (bit 31 of the base) selects: 64 when set, 32 otherwise. width := int64(32) << (baseOp >> 31 & 1) if lsb < 0 || lsb >= width { return nil, fmt.Errorf("%s: bit number %d out of range (width=%d)", mnem, lsb, width) } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(lsb)<<10 | uint32(rn)<<5 | uint32(rd)), nil } // ---- SIMD/NEON encoding ---- // arm64VecOf parses a vector operand. The element suffix of V13.S[0] does // not survive into the symbol name, so the verbatim operand text is tried // first and the register name second. func arm64VecOf(op *ast.Operand) (a64Vec, bool) { if v, ok := a64VecReg(op.Raw); ok { return v, true } return a64VecReg(operandRegName(op)) } // arm64SimdHasElement reports whether any operand carries a lane index such // as V13.S[0]. func arm64SimdHasElement(ops []*ast.Operand) bool { for _, op := range ops { if v, ok := arm64VecOf(op); ok && v.hasIdx { return true } } return false } // arm64SimdArrs validates that a SIMD operand run spells one arrangement, // that it is the same on every operand that spells one, and that the table // admits it. It returns the arrangement's index, with a64Arr8B for a bare // V/F spelling. func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) { sel := a64Arr8B for _, a := range arrs { if a == "" { continue } i := a64ArrIndex(a) if i < 0 || specBit(i)&allowed == 0 { return 0, fmt.Errorf("%s: invalid arrangement %q", mnem, a) } if sel != a64Arr8B && sel != i { return 0, fmt.Errorf("%s: mixed arrangements", mnem) } sel = i } return sel, nil } // specBit returns the a64SimdVSpec bitmask bit for an arrangement index. func specBit(i int) uint16 { return 1 << uint(i) } // arm64SimdZeroImm reports whether the first operand of a SIMD compare is // the zero immediate: $0 for the integer compares, $(0.0) for the FP ones // (the toolchain accepts the FP zero only as a spelled float or integer 0). func arm64SimdZeroImm(mnem string, op *ast.Operand) bool { if v, ok := arm64ImmOperandValue(op); ok && v == 0 { return true } if !strings.HasPrefix(mnem, "VFCM") { return false } s := strings.Join(strings.Fields(op.Raw), "") s = strings.TrimPrefix(s, "$") s = strings.Trim(s, "()") return s == "0" || s == "0.0" } // encodeARM64SimdV encodes an arrangement-aware three-register SIMD // instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD // compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take // their compare-against-zero form instead, and the polynomial multiplies read // the arrangement from their source operands alone, the result spelling // (H8, Q1) riding no encoding bits. func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) { if mnem == "VPMULL" || mnem == "VPMULL2" { if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } vs := make([]a64Vec, 2) arrs := make([]string, 2) for i, op := range ops[:2] { v, ok := arm64VecOf(op) if !ok || v.hasIdx { return nil, fmt.Errorf("invalid register operand in %s", mnem) } vs[i], arrs[i] = v, v.arr } if _, ok := arm64VecOf(ops[2]); !ok { return nil, fmt.Errorf("invalid register operand in %s", mnem) } arr, err := arm64SimdArrs(mnem, arrs, spec.arrs) if err != nil { return nil, err } rd, _ := arm64VecOf(ops[2]) return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil } if len(ops) == 3 && isImmOperand(ops[0]) { base, ok := a64SimdVZero[mnem] if !ok { return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem) } if !arm64SimdZeroImm(mnem, ops[0]) { return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem) } vn, ok1 := arm64VecOf(ops[1]) vd, ok2 := arm64VecOf(ops[2]) if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx { return nil, fmt.Errorf("invalid register operand in %s", mnem) } allowed := uint16(0x7f) if strings.HasPrefix(mnem, "VFCM") { allowed = 1< 63 { return nil, fmt.Errorf("%s: rotation %d out of range (0..63)", mnem, rot) } vs := make([]a64Vec, 3) arrs := make([]string, 3) for i, op := range ops[1:] { v, ok := arm64VecOf(op) if !ok || v.hasIdx { return nil, fmt.Errorf("invalid register operand in %s", mnem) } vs[i], arrs[i] = v, v.arr } if _, err := arm64SimdArrs(mnem, arrs, 1< int64(max) { return nil, fmt.Errorf("%s: index %d out of range (0..%d)", mnem, idx, max) } return a64wordLE(base | b16 | uint32(idx)<<11 | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil } return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) } // encodeARM64VTBL encodes VTBL Vidx.arr, [Vt1.arr, ...], Vdest.arr: the // index register rides bits 19:16, the first table register bits 9:5, the // destination bits 4:0 and the table length (registers minus one) bits // 14:13. The table registers must be consecutive. func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) { if len(ops) < 3 { return nil, fmt.Errorf("VTBL expects index, table list and destination") } vi, ok := arm64VecOf(ops[0]) if !ok || vi.hasIdx { return nil, fmt.Errorf("invalid index register in VTBL") } ts, end, ok := a64VecListOf(ops, 1) if !ok || len(ts) < 1 || len(ts) > 4 { return nil, fmt.Errorf("VTBL expects a table of one to four registers") } vd, ok := a64VecReg(operandRegName(ops[end+1])) if !ok || end+2 != len(ops) || vd.hasIdx { return nil, fmt.Errorf("invalid destination register in VTBL") } for i, t := range ts { if t.hasIdx || t.reg != ts[0].reg+i { return nil, fmt.Errorf("VTBL table registers must be consecutive") } } q := uint32(0) switch vi.arr { case "B16": q = 1 << 30 case "B8", "": default: return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr) } base := uint32(0x0e000000) if mnem == "VTBX" { base |= 1 << 12 } return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil } // encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.: a // general register into an arranged vector, the spelling asm7.go's case 82 // calls vmov/vdup Rn, Vd.. ok is false for anything that is not that // shape, so the caller falls through to the arrangement and element paths; // the toolchain rejects the bare spellings outright, and the reverse // Vd., Rs with them. func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) { if len(ops) != 2 { return nil, false, nil } if ops[0].Addr.Base != "" || isImmOperand(ops[0]) { return nil, false, nil } rs := arm64RegNum(operandRegName(ops[0])) if rs < 0 { return nil, false, nil } dst, ok := arm64VecOf(ops[1]) if !ok || dst.hasIdx || dst.arr == "" { return nil, false, nil } b, err := a64GPVecWhole(mnem, rs, dst) return b, true, err } // a64GPVecWhole lays down the general-register-into-a-whole-vector move: // word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the // lane width and Q the vector length. Both VMOV and VDUP take this form // (asm7.go case 82); INS-into-one-lane is encoded elsewhere. func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) { var imm5, q uint32 switch dst.arr { case "B8": imm5, q = 1, 0 case "B16": imm5, q = 1, 1<<30 case "H4": imm5, q = 2, 0 case "H8": imm5, q = 2, 1<<30 case "S2": imm5, q = 4, 0 case "S4": imm5, q = 4, 1<<30 case "D2": imm5, q = 8, 1<<30 default: // D1 rides no case-82 row: the toolchain rejects the one-doubleword // spelling for this form, so the encoder refuses it too. return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr) } return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil } // encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with // lane indices: // // Vn.[i], Rd UMOV, element to general register // Vn.[i], Vd.arr DUP, element across a vector // Vn.[i], Vd.[j] INS, element to element // Rs, Vd.[i] INS, general register into an element func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } dst, dstVec := arm64VecOf(ops[1]) dstGP := false if !dstVec { // A general-register spelling as destination (UMOV forms). if rd := arm64RegNum(operandRegName(ops[1])); rd >= 0 { dst, dstGP, dstVec = a64Vec{reg: rd}, true, true } } if !dstVec { return nil, fmt.Errorf("%s: invalid destination operand", mnem) } src, ok1 := arm64VecOf(ops[0]) if (!ok1 || !src.hasIdx) && !dst.hasIdx { return nil, fmt.Errorf("%s expects an element operand Vn.[i]", mnem) } if !ok1 || !src.hasIdx { // General register into a vector. An arranged destination without a // lane index duplicates the register across every lane (DUP Vd.T, // Rn); an indexed one is an INS into that single lane. rs := arm64RegNum(operandRegName(ops[0])) if rs < 0 { return nil, fmt.Errorf("%s: source must be a general register", mnem) } if !dst.hasIdx { // Duplicates the register across every lane (DUP Vd.T, Rn). return a64GPVecWhole(mnem, rs, dst) } f, ok := a64ElemField(dst.arr, dst.idx) if !ok { return nil, fmt.Errorf("%s: invalid element operand", mnem) } return a64wordLE(0x4e001c00 | f<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil } sf, ok := a64ElemField(src.arr, src.idx) if !ok { return nil, fmt.Errorf("%s: invalid element operand", mnem) } if dst.hasIdx { // Element to element. df, ok := a64ElemField(dst.arr, dst.idx) if !ok { return nil, fmt.Errorf("%s: invalid element operand", mnem) } return a64wordLE(0x6e000400 | df<<16 | sf>>1<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil } if dstGP { // Element to a general register: UMOV, with the D form setting bit // 30. base := uint32(0x0e003c00) if src.arr == "D" { base = 0x4e003c00 } return a64wordLE(base | sf<<16 | uint32(src.reg)<<5 | uint32(dst.reg)), nil } if dst.arr == "" { // Element across a bare V register. return a64wordLE(0x5e000400 | sf<<16 | uint32(src.reg)<<5 | uint32(dst.reg)), nil } // Element across an arranged vector; the 128-bit arrangements set bit // 30. q := uint32(0) switch dst.arr { case "B16", "H8", "S4", "D2": q = 1 << 30 case "B8", "H4", "S2", "D1": default: return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr) } return a64wordLE(0x0e000400 | q | sf<<16 | uint32(src.reg)<<5 | uint32(dst.reg)), nil } // encodeARM64VLDST encodes the SIMD structure loads and stores: // // VLD1 (Rn), [Vt.arr, ...] VST1 [Vt.arr, ...], (Rn) // VLD1.P off(Rn), [Vt.arr, ...] VST1.P [Vt.arr, ...], off(Rn) // VLD1.P off(Rn), Vt.T[i] VST1.P Vt.T[i], off(Rn) (one lane) // VLD1R (Rn), [Vt.arr] VLD4R (Rn), [Vt.arr, Vt+1, Vt+2, Vt+3] // // The post-index forms set the post bit and Rm = 11111. A spelled offset // rides along (the encoding ignores it; the toolchain only checks that it // matches the access size), and a multi-register post-index list takes no // offset at all, the increment following from the list. func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, error) { load := strings.HasPrefix(mnem, "VLD") // One-lane forms spell a single Vt.T[i] operand, not a bracketed list. laneIdx := 1 if !load { laneIdx = 0 } if len(ops) > laneIdx { if v, ok := arm64VecOf(ops[laneIdx]); ok && v.hasIdx { return encodeARM64VLDSTLane(mnem, post, ops, load, laneIdx, v) } } listStart, memAt := 0, 1 if load { // Every load spells the memory operand first. listStart, memAt = 1, 0 } vs, end, ok := a64VecListOf(ops, listStart) if !ok { return nil, fmt.Errorf("%s: invalid register list", mnem) } memIdx := end + 1 if memAt == 0 { memIdx = 0 } if memIdx >= len(ops) { return nil, fmt.Errorf("%s expects a (Rn) memory operand", mnem) } rn, off := arm64MemWithFrame(ops[memIdx], arm64FrameInfo{}) if rn < 0 { return nil, fmt.Errorf("%s: invalid memory operand", mnem) } if off != 0 && post == 0 { return nil, fmt.Errorf("%s: offset %d not supported, plain and list accesses take a plain (Rn) operand", mnem, off) } // VLD1R loads one register and replicates; VLD4R loads four. if strings.HasPrefix(mnem, "VLD1R") || strings.HasPrefix(mnem, "VLD4R") { want := 1 base := uint32(0x0d40c000) if strings.HasPrefix(mnem, "VLD4R") { want, base = 4, 0x0d60e000 } if len(vs) != want { return nil, fmt.Errorf("%s expects a list of %d registers", mnem, want) } size, q, ok := a64ArrSizeQ(vs[0].arr) if !ok { return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr) } w := base | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg) if post != 0 { w |= 1<<23 | 0x1f<<16 } return a64wordLE(w), nil } if len(vs) < 1 || len(vs) > 4 { return nil, fmt.Errorf("%s expects a list of one to four registers", mnem) } for i, v := range vs { if v.hasIdx || v.reg != vs[0].reg+i { return nil, fmt.Errorf("%s: register list must be consecutive", mnem) } _, _, okArr := a64ArrSizeQ(v.arr) if !okArr || (i > 0 && v.arr != vs[0].arr) { return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, v.arr) } } size, q, ok := a64ArrSizeQ(vs[0].arr) if !ok { return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr) } base := a64VLD1Base[len(vs)] if !load { base = a64VST1Base[len(vs)] } postBits := uint32(0) if post != 0 { postBits = 0x9f0000 } return a64wordLE(base | q<<30 | size<<10 | postBits | uint32(rn)<<5 | uint32(vs[0].reg)), nil } // encodeARM64VLDSTLane encodes the one-lane structure forms: // VLD1 off(Rn), Vt.T[i] (post-index adds the post bit and Rm=11111) and // VST1.P Vt.T[i], off(Rn); the plain VST1 lane form does not exist in the // toolchain's table and is rejected. func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load bool, laneIdx int, v a64Vec) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects a memory operand and one Vt.T[i] lane operand", mnem) } memIdx := laneIdx ^ 1 rn, _ := arm64MemWithFrame(ops[memIdx], arm64FrameInfo{}) if rn < 0 { return nil, fmt.Errorf("%s: invalid memory operand", mnem) } if !load && post == 0 { return nil, fmt.Errorf("%s: the toolchain only spells a post-index single-lane store", mnem) } w := uint32(0x0d400000) switch strings.ToUpper(v.arr) { case "B": // Index at bits 12:10 (the size field doubles as the low index bits). w |= uint32(v.idx) << 10 case "H": // Index<2> at bit 30, index<1> at bit 12, index<0> at bit 11. w |= 1<<14 | uint32(v.idx&1)<<11 | uint32(v.idx>>1&1)<<12 | uint32(v.idx>>2&1)<<30 case "S": // Index<0> at bit 12, index<1> at bit 30. w |= 4<<13 | uint32(v.idx&1)<<12 | uint32(v.idx>>1&1)<<30 case "D": // Index<0> at bit 30, fixed size field 01. w |= 4<<13 | 1<<10 | uint32(v.idx&1)<<30 default: return nil, fmt.Errorf("%s: invalid lane arrangement %q", mnem, v.arr) } // The base carries bit 22 (L) set; a store clears it. The post-index // forms add bit 23 and Rm = 11111. if !load { w &^= 1 << 22 } if post != 0 { w |= 1<<23 | 0x1f<<16 } return a64wordLE(w | uint32(rn)<<5 | uint32(v.reg)), nil } // a64ArrSizeQ maps an arrangement to its size code (bits 11:10) and 128-bit // flag for the structure load/store words. func a64ArrSizeQ(arr string) (size, q uint32, ok bool) { switch arr { case "B8": return 0, 0, true case "B16": return 0, 1, true case "H4": return 1, 0, true case "H8": return 1, 1, true case "S2": return 2, 0, true case "S4": return 2, 1, true case "D1": return 3, 0, true case "D2": return 3, 1, true } return 0, 0, false } // encodeARM64ShiftImm encodes a SIMD shift by immediate: // word = base | Q<<30 | immh:immb<<16 | Rn<<5 | Rd, where immh:immb is the // element size plus the shift for a left shift (VSHL) and twice the element // size minus the shift for right shifts (VUSHR, VSRI). func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 3 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects 3 operands ($shift, Vn.arr, Vd.arr)", mnem) } sh := arm64Imm64(ops[0]) vn, ok1 := arm64VecOf(ops[1]) vd, ok2 := arm64VecOf(ops[2]) if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx || vn.arr != vd.arr { return nil, fmt.Errorf("%s: operands must share one arrangement", mnem) } var esize int64 q := uint32(0) switch vn.arr { case "B8", "B": esize = 8 case "B16": esize, q = 8, 1 case "H4", "H": esize = 16 case "H8": esize, q = 16, 1 case "S2", "S": esize = 32 case "S4": esize, q = 32, 1 case "D2": esize, q = 64, 1 default: return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vn.arr) } var immval int64 switch mnem { case "VSHL", "VSLI", "VSQSHL", "VUQSHL", "VSQSHLU": if sh < 0 || sh >= esize { return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1) } immval = esize + sh default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR if sh < 1 || sh > esize { return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize) } immval = 2*esize - sh } return a64wordLE(base | q<<30 | uint32(immval)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil } // encodeARM64MoviLit loads a large vector constant the way the toolchain // does: ADRP R27 and ADD materialise the literal's address, then FMOVS, // FMOVD or the 128-bit FMOVQ form loads it, with R_ADDRARM64 relocations // against a read-only literal the file assembler lays out. func encodeARM64MoviLit(mnem string, ldr uint32, ops []*ast.Operand, relocs *[]Reloc, lits *arm64Literals) ([]byte, error) { var vd a64Vec var ok bool var data []byte switch mnem { case "VMOVS", "VMOVD": if len(ops) != 2 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects $value, Vd", mnem) } v := arm64Imm64(ops[0]) if mnem == "VMOVS" { if v < -2147483648 || v > 0xFFFFFFFF { return nil, fmt.Errorf("%s: constant does not fit 32 bits", mnem) } data = a64wordLE(uint32(v)) } else { data = a64WordsLE(uint32(v), uint32(v>>32)) } vd, ok = arm64VecOf(ops[1]) if !ok || vd.hasIdx || vd.arr != "" { return nil, fmt.Errorf("%s: destination must be a bare V register", mnem) } case "VMOVQ": if len(ops) != 3 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) { return nil, fmt.Errorf("VMOVQ expects $lo, $hi, Vd") } lo, hi := arm64Imm64(ops[0]), arm64Imm64(ops[1]) data = a64WordsLE(uint32(lo), uint32(lo>>32), uint32(hi), uint32(hi>>32)) vd, ok = arm64VecOf(ops[2]) if !ok || vd.hasIdx || vd.arr != "" { return nil, fmt.Errorf("VMOVQ: destination must be a bare V register") } default: return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) } name := lits.add(moviLitName(mnem, data), data) if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: name, Kind: RelArm64Addr}, Reloc{Off: 4, After: 4, Name: name, Kind: RelArm64Addr}, ) } return a64WordsLE( a64ADR(1, 0, 0, 27), a64AddSub(1, 0, 0, 0, 0, 27, 27), ldr|0<<10|27<<5|uint32(vd.reg), ), nil } // moviLitName mirrors the toolchain's literal naming: $i32/$i64/$i128 // followed by the constant's value in hex. func moviLitName(mnem string, data []byte) string { switch mnem { case "VMOVS": v := uint32(data[0]) | uint32(data[1])<<8 | uint32(data[2])<<16 | uint32(data[3])<<24 return "$i32." + strconv.FormatUint(uint64(v), 16) case "VMOVD": v := uint64(data[0]) | uint64(data[1])<<8 | uint64(data[2])<<16 | uint64(data[3])<<24 | uint64(data[4])<<32 | uint64(data[5])<<40 | uint64(data[6])<<48 | uint64(data[7])<<56 return "$i64." + strconv.FormatUint(v, 16) default: hi := uint64(data[8]) | uint64(data[9])<<8 | uint64(data[10])<<16 | uint64(data[11])<<24 | uint64(data[12])<<32 | uint64(data[13])<<40 | uint64(data[14])<<48 | uint64(data[15])<<56 lo := uint64(data[0]) | uint64(data[1])<<8 | uint64(data[2])<<16 | uint64(data[3])<<24 | uint64(data[4])<<32 | uint64(data[5])<<40 | uint64(data[6])<<48 | uint64(data[7])<<56 return "$i128." + strings.Repeat("0", max(0, 16-len(strconv.FormatUint(hi, 16)))) + strconv.FormatUint(hi, 16) + strings.Repeat("0", max(0, 16-len(strconv.FormatUint(lo, 16)))) + strconv.FormatUint(lo, 16) } } // arm64Literals collects the read-only constants the VMOVS/VMOVD/VMOVQ // loads refer to. Names follow the toolchain's $i32/$i64/$i128 spellings so // equal constants deduplicate to one literal. type arm64Literals struct { order []Arm64Literal seen map[string]bool } // Arm64Literal is one pooled vector constant. type Arm64Literal struct { Name string Data []byte } // add registers a literal under its name and returns it. func (l *arm64Literals) add(name string, data []byte) string { if l.seen == nil { l.seen = map[string]bool{} } if !l.seen[name] { l.seen[name] = true l.order = append(l.order, Arm64Literal{Name: name, Data: data}) } return name } // list returns the literals in first-use order. func (l *arm64Literals) list() []Arm64Literal { return l.order } // AssembleFileARM64 assembles every TEXT function of a parsed arm64 file // and lays out its static symbols (GLOBL/DATA) in a data section behind the // code. SB references in the code are encoded as ADRP pairs with zero // immediates; the object-file emitters record R_ADDRARM64 relocations for // the linker. func AssembleFileARM64(f *ast.File) (*Image, error) { // Resolve the file's own simple #define aliases (RARG0 → R0, NR → R9, // TEB_error → 0x68, B0 → V0) the way the toolchain's preprocessor does // textually. Parameterised macros and multi-line bodies are beyond // token substitution and stay untouched. arm64ResolveAliases(f) dataSyms, err := collectData(f) if err != nil { return nil, err } img := &Image{Symbols: map[string]int{}, SourcePath: f.Path} var pendingLits []Arm64Literal litSeen := map[string]bool{} for _, d := range f.Decls { t, ok := d.(*ast.Text) if !ok { continue } code, labels, relocs, lines, spadj, lits, err := assembleARM64(t) if err != nil { return nil, fmt.Errorf("%s: %w", t.Name.Name, err) } // The literals this function's constant loads refer to join the // data section once, deduplicated by name. for _, lit := range lits { if _, seen := litSeen[lit.Name]; seen { continue } litSeen[lit.Name] = true pendingLits = append(pendingLits, lit) } fl := FuncLayout{ Name: t.Name.Name, Pkg: t.Name.Pkg, Static: t.Name.Static, Offset: len(img.Code), Size: len(code), Frame: frameSize(t), Args: argsSize(t), Line: t.Pos().Line, Labels: labels, Lines: lines, Spadj: spadj, Relocs: relocs, } for _, f := range t.Flags { switch f { case "NOSPLIT": fl.NoSplit = true case "SPWRITE": fl.SPWrite = true } } img.Funcs = append(img.Funcs, fl) img.Code = append(img.Code, code...) } // Lay out the data section behind the code, 16-aligned. dataStart := len(img.Code) for _, d := range dataSyms { pos := dataStart + len(img.Data) for pos%16 != 0 { img.Data = append(img.Data, 0) pos++ } img.Symbols[d.name] = pos img.Data = append(img.Data, d.buf...) img.DataSyms = append(img.DataSyms, DataSymbol{ Name: d.name, Pkg: d.pkg, Offset: len(img.Data) - len(d.buf), Size: d.size, Static: d.static, Rodata: d.rodata, Dupok: d.dupok, }) } // The read-only literals the VMOVS/VMOVD/VMOVQ constant loads refer to // follow the declared data, deduplicated across the file. for _, lit := range pendingLits { pos := dataStart + len(img.Data) for pos%16 != 0 { img.Data = append(img.Data, 0) pos++ } img.Symbols[lit.Name] = pos img.Data = append(img.Data, lit.Data...) img.DataSyms = append(img.DataSyms, DataSymbol{ Name: lit.Name, Offset: len(img.Data) - len(lit.Data), Size: len(lit.Data), Rodata: true, Dupok: true, }) } markExternals(img, dataSyms) return img, nil } // arm64ResolveAliases applies the file's own simple #define aliases to every // instruction operand, the way the toolchain's preprocessor substitutes them // textually. Only single-line, non-parameterised bodies whose value is a // register name or an integer constant are resolved: anything else // (parameterised macros, multi-instruction bodies, header-supplied names) // stays as written and surfaces as a normal operand error. func arm64ResolveAliases(f *ast.File) { type alias struct { raw string // replacement text reg bool // the body is a register name value int64 // the body as an integer (when !reg) isInt bool // the body parsed as an integer mem bool // the body is a frame-relative memory reference sym *ast.Symbol // the parsed frame-relative reference (when mem) } aliases := map[string]alias{} raws := map[string]string{} for _, d := range f.Decls { pre, ok := d.(*ast.Preproc) if !ok { continue } fields := strings.Fields(pre.Raw) if len(fields) < 3 || fields[0] != "define" { continue } name, body := fields[1], strings.Join(fields[2:], " ") // A parameterised macro spells its parameter list right after the // name; a multi-instruction body needs statement expansion. if strings.ContainsAny(name, "(") || body == "" || strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") { continue } raws[name] = body } // Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute // transitively until nothing changes, bounded against cycles. for range 8 { changed := false for name, body := range raws { if next, ok := raws[body]; ok && next != body { raws[name] = next changed = true } } if !changed { break } } for name, body := range raws { isReg := func(s string) bool { if arm64RegNum(s) >= 0 { return true } if v, ok := a64VecReg(s); ok && !v.hasIdx { return true } return false } if isReg(body) { aliases[name] = alias{raw: body, reg: true} continue } if sym, ok := arm64FrameAliasBody(body); ok { aliases[name] = alias{raw: body, mem: true, sym: sym} continue } v, err := strconv.ParseInt(body, 0, 64) if err != nil { continue } aliases[name] = alias{raw: body, value: v, isInt: true} } if len(aliases) == 0 { return } // replaceToken rewrites an operand whose whole text is one alias use // possibly followed by syntax (POLY.D[0]): the alias must be a prefix // ending at a non-identifier character. replaceToken := func(s string) (string, bool) { for name, a := range aliases { if s == name { return a.raw, true } if strings.HasPrefix(s, name) { rest := s[len(name):] if rest != "" && !isAliasWordByte(rest[0]) { return a.raw + rest, true } } } return s, false } // replaceScan rewrites alias uses inside a composite operand (a // parenthesised memory operand or a bracketed register list): every // identifier run of word and dot characters is matched against the alias // names, everything else copies verbatim. The whitespace-split replace // above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their // punctuation attached. replaceScan := func(s string) string { var b strings.Builder for i := 0; i < len(s); { if isAliasWordByte(s[i]) || s[i] == '.' { j := i for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') { j++ } if nn, ok := replaceToken(s[i:j]); ok { b.WriteString(nn) } else { b.WriteString(s[i:j]) } i = j continue } b.WriteByte(s[i]) i++ } return b.String() } for _, d := range f.Decls { t, ok := d.(*ast.Text) if !ok { continue } for _, stmt := range t.Body { in, ok := stmt.(*ast.Instr) if !ok { continue } for _, op := range in.Operands { // Immediate aliases: $CLOCK_REALTIME → $0. The parser may // leave the unevaluable name in Raw alone or carry it as an // unevaluated symbol immediate; both shapes resolve here. if op.Kind == ast.OpImmediate && !op.Imm.HasVal { name := "" if op.Imm.Sym != nil && op.Imm.Sym.Pseudo == "" { name = op.Imm.Sym.Name } else if op.Addr.Sym == nil && op.Addr.Base == "" { name = strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(op.Raw), "$")) } if a, ok := aliases[name]; ok && a.isInt { op.Imm.Val, op.Imm.HasVal = a.value, true op.Imm.Sym = nil op.Addr = ast.Address{} op.Raw = "$" + a.raw continue } } // Memory operand whose displacement is an alias: // TEB_error(R18_PLATFORM) with TEB_error → 0x68. if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base != "" { if a, ok := aliases[op.Addr.Sym.Name]; ok && a.isInt { op.Addr.Offset, op.Addr.HasOff = a.value, true op.Addr.Sym = nil op.Raw = a.raw + "(" + op.Addr.Base + ")" continue } } // Register alias as a memory base. if op.Addr.Base != "" { if a, ok := aliases[op.Addr.Base]; ok && a.reg { op.Addr.Base = a.raw } } // Bare and suffixed symbol tokens: registers, vector // registers with arrangement or lane, branch labels. if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" { if nn, changed := replaceToken(op.Addr.Sym.Name); changed { // A frame-relative body (ret+24(FP)) rebuilds the // operand as a full memory reference. if a, ok := aliases[op.Addr.Sym.Name]; ok && a.mem { op.Addr.Sym, op.Addr.Base, op.Addr.Shift = a.sym, "", "" op.Raw = a.raw continue } op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn // The span shape depends on what trailed the // name: an element or arrangement selector // (POLY.D[0], POLY.B16) rides in Shift and folds // back onto the rewritten token; a shift // operator stays in Shift while the span carries // the bare register; a split list keeps its // closing bracket, so the rewrite goes through // the scan. sfx := strings.Join(strings.Fields(op.Addr.Shift), "") switch { case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]": // Element or arrangement selectors and the // closing bracket of a split list belong to // the token text. op.Raw = nn + sfx op.Addr.Shift = "" case op.Addr.Shift != "": op.Raw = nn default: op.Raw = replaceScan(op.Raw) } continue } } // Bracketed groups and lists travel in Raw: (RARG0, R1) and // [V0.B16, V1.B16] with aliased members. if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") || strings.HasPrefix(strings.TrimSpace(op.Raw), "[") { op.Raw = replaceScan(op.Raw) } } } } } // isAliasWordByte reports whether b can appear inside an identifier, so a // substitution ending here would have merged two tokens. func isAliasWordByte(b byte) bool { return b == '_' || b >= '0' && b <= '9' || b >= 'a' && b <= 'z' || b >= 'A' && b <= 'Z' } // arm64FrameAliasBody parses an alias body of the shape NAME, NAME+off or // NAME+off(PSEUDO) with PSEUDO one of FP/SP: the frame-relative memory // references the runtime headers alias wholesale (LOCAL_RETVALID // → ret+24(FP)). ok is false for anything else. func arm64FrameAliasBody(body string) (*ast.Symbol, bool) { s := strings.Join(strings.Fields(body), "") i := strings.LastIndexByte(s, '(') if i < 0 || !strings.HasSuffix(s, ")") { return nil, false } pseudo := s[i+1 : len(s)-1] if pseudo != "FP" && pseudo != "SP" { return nil, false } head := s[:i] name, offStr := head, "" if j := strings.LastIndexByte(head, '+'); j >= 0 { name, offStr = head[:j], head[j+1:] } if name == "" { return nil, false } off, hasOff := int64(0), false if offStr != "" { v, err := strconv.ParseInt(offStr, 0, 64) if err != nil { return nil, false } off, hasOff = v, true } return &ast.Symbol{Name: name, Pseudo: pseudo, Offset: off, HasOff: hasOff, Raw: s}, true }