// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm // arm64 frame mapping, matching the Go toolchain's arm64 backend. // // Go's arm64 functions use R29 as the frame pointer (FP) and R30 as the link // register (LR). R31 is the stack pointer (SP). FP and SP in the source // are synthetic pseudo-registers resolved against the hardware SP and the // frame size. // // The autosize is the real stack adjustment: the declared local frame plus // 8 bytes for the saved link register, rounded up to a 16-byte multiple. // The toolchain adds an "extrasize" to align: if autosize%16 == 8, add 8; // if autosize%16 == 0, add 16. // // Prologue (autosize > 0, small frame ≤ 0xf0): // // MOVD.W LR, -autosize(SP) // pre-index: SP -= autosize, store LR at SP // MOVD FP, -8(SP) // store FP at SP-8 // SUB $8, SP, FP // FP = SP - 8 // // Prologue (autosize > 0, large frame > 0xf0): // // SUB $autosize, SP, R20 // R20 = SP - autosize // STP (FP, LR), -8(R20) // store FP,LR at R20-8 // MOVD R20, SP // SP = R20 // SUB $8, SP, FP // FP = SP - 8 // // Epilogue (non-leaf, small frame): // // ADD $autosize-8, SP, FP // restore FP // ADD $autosize, SP, SP // deallocate frame // MOVD -8(SP), FP // (actually the reverse of prologue) // Actually: // MOVD -8(SP), FP // load FP from SP-8 // MOVD.P autosize(SP), LR // post-index: load LR, SP += autosize // // Epilogue (non-leaf, large frame): // ADD $autosize-8, SP, FP // ADD $autosize, SP, SP // Actually: // LDP -8(SP), (FP, LR) // load FP,LR // ADD $autosize, SP, SP // deallocate frame // // Epilogue (leaf with frame): // ADD $autosize-8, SP, FP // ADD $autosize, SP, SP // // RET always emits as BR LR (0xd65f03c0). import ( "strings" "sourcedock.dev/petrbalvin/gasm-sdk/ast" ) // arm64FrameInfo holds the frame layout derived from a TEXT directive. type arm64FrameInfo struct { autosize int // the real SP adjustment (locals + saved LR + alignment) frame int // the declared $framesize args int // the declared -argsize noSplit bool // the NOSPLIT flag leaf bool // no call instructions in the body // Stack-split guard state: needSplit mirrors the toolchain, which skips // the check for NOSPLIT functions and auto-marks leaf functions with an // autosize below StackSmall as NOSPLIT. needSplit bool splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig } // arm64ComputeFrame derives the frame layout for a TEXT function. func arm64ComputeFrame(t *ast.Text) arm64FrameInfo { fi := arm64FrameInfo{ frame: frameSize(t), args: argsSize(t), } for _, f := range t.Flags { if f == "NOSPLIT" { fi.noSplit = true } } fi.leaf = arm64IsLeaf(t) if fi.frame != 0 || !fi.leaf { fi.autosize = fi.frame + 8 // space for the saved LR // The toolchain always adds an extrasize: 8 when the total leaves a // 16-byte alignment gap, another 16 when already aligned. switch fi.autosize % 16 { case 8: fi.autosize += 8 case 0: fi.autosize += 16 default: // The toolchain rejects unaligned frames; round up so such // sources still assemble. fi.autosize += 16 - (fi.autosize % 16) } } switch { case fi.noSplit: case fi.autosize < stackSmall && fi.leaf: // Auto-NOSPLIT, as the toolchain's leaf mark concludes. default: fi.needSplit = true switch { case fi.autosize <= stackSmall: fi.splitClass = 0 case fi.autosize <= stackBig: fi.splitClass = 1 default: fi.splitClass = 2 } } return fi } // arm64GuardLen returns the byte length of the stack-split guard prefix // (zero when the function needs no guard). The big class materialises // framesize-StackSmall into REGTMP, whose MOVZ/MOVK sequence length varies. func arm64GuardLen(fi arm64FrameInfo) int { if !fi.needSplit { return 0 } switch fi.splitClass { case 0: return 12 case 1: return 16 default: n, err := arm64LoadImmLen(int64(fi.autosize - stackSmall)) if err != nil { return 0 } return 4 + n + 4 + 4 + 4 + 4 } } // arm64LoadImmLen returns the byte length of the MOVZ/MOVK sequence that // loads v into a register. func arm64LoadImmLen(v int64) (int, error) { b, err := encodeARM64LoadImm(27, v, "MOVD") if err != nil { return 0, err } return len(b), nil } // arm64IsLeaf reports whether a function contains no call instructions // (BL/CALL), matching the toolchain's LEAF mark. func arm64IsLeaf(t *ast.Text) bool { for _, stmt := range t.Body { in, ok := stmt.(*ast.Instr) if !ok { continue } switch strings.ToUpper(in.Mnemonic.Text) { case "BL", "CALL": return false } } return true } // arm64Prologue returns the prologue bytes for an arm64 function. func arm64Prologue(fi arm64FrameInfo) []byte { if fi.autosize == 0 { return nil } if fi.autosize <= 0xf0 { // Small frame: MOVD.W LR, -autosize(SP); MOVD FP, -8(SP); SUB $8, SP, FP return a64WordsLE( arm64PreStoreImm(3, 0, int32(-fi.autosize), 31, 30), // STR.W LR, -autosize(SP) (pre-index store) arm64UnscaledStore(3, 0, -8, 31, 29), // STUR FP, [SP, #-8] a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB) ) } // Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP ws := arm64SubImmWords(uint32(fi.autosize), 20) ws = append(ws, a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair) a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP) a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB) ) return a64WordsLE(ws...) } // arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into // two imm12 instructions instead of materialising it into REGTMP // (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits // unsigned and be neither encodable as one imm12 (checked by the callers // first), nor loadable into a register in a single MOVZ/MOVN word, nor a // logical immediate, because conclass tests all three before C_ADDCON2. func arm64SplitImm12(imm uint32) bool { if imm > 0xFFFFFF { return false } if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok { return false } return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0 } // arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an // ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the // immediate form when the value fits imm12 (plain, or shifted left by 12 // when it is a multiple of 4096); a value with a single 16-bit chunk, a // logical immediate, or one wider than 24 bits is materialised into REGTMP // (R27) and subtracted in the extended-register form; everything else up to // 0xFFFFFF is split into two imm12 instructions: // // SUB $(imm&0xfff), SP, Rd // SUB $((imm&0xfff000)>>12)<<12, Rd, Rd func arm64SubImmWords(imm uint32, rd uint32) []uint32 { if imm <= 0xFFF { return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)} } if imm <= 4095<<12 && imm&0xFFF == 0 { return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)} } if !arm64SplitImm12(imm) { mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") if err != nil { mov = nil } return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd)) } return []uint32{ a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd), a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd), } } // arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12, // split and REGTMP ladder as arm64SubImmWords. func arm64AddImmWords(imm uint32, rd uint32) []uint32 { if imm <= 0xFFF { return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)} } if imm <= 4095<<12 && imm&0xFFF == 0 { return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)} } if !arm64SplitImm12(imm) { mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") if err != nil { mov = nil } return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd)) } return []uint32{ a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd), a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd), } } // arm64RetAddWords emits the frame deallocation of a non-leaf RET with a // large frame. The toolchain adds the frame back with a single instruction: // a plain imm12 ADD when autosize fits 12 bits, otherwise the value is // materialised into REGTMP and added as a register, so the epilogue never // leaves a partially deallocated frame (obj7.go ARET, issue 73259). The // shifted-imm12 and split-imm12 forms are therefore never used here, unlike // the leaf epilogue's plain ADD instructions. func arm64RetAddWords(autosize uint32) []uint32 { if autosize < 1<<12 { return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)} } mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD") if err != nil { mov = nil } return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31)) } // arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and // deallocate the frame when present) followed by RET (BR LR). func arm64Return(fi arm64FrameInfo) []byte { var ws []uint32 if fi.autosize != 0 { if fi.leaf { // Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP ws = append(ws, arm64AddImmWords(uint32(fi.autosize-8), 29)...) ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...) } else if fi.autosize <= 0xf0 { // Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize ws = append(ws, arm64UnscaledLoad(3, 0, -8, 31, 29), // LDR FP, [SP, #-8] arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize ) } else { // Large frame: LDP -8(SP), (FP, LR), then deallocate. ws = append(ws, a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair) ) ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...) } } // RET: BR LR (0xd65f03c0) ws = append(ws, a64UncondBranch(2, 30, 0)) // opc=2(RET), Rn=LR(30), Rd=0 return a64WordsLE(ws...) } // arm64PrologueSpadjPC returns the function-relative byte offset where the // prologue has finished decrementing SP (the delta becomes autosize). func arm64PrologueSpadjPC(fi arm64FrameInfo) int { if fi.autosize == 0 { return 0 } if fi.autosize <= 0xf0 { return 4 // MOVD.W instruction decrements SP } // Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose // position depends on how many words the SUB itself took (immediate, // shifted immediate, the two-word imm12 split, or a materialised REGTMP // sequence). return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1) } // arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to // (but not including) the final RET instruction. The lengths are read from // the same word-emitting helpers the epilogue uses rather than assumed: the // leaf path shares the prologue's immediate ladder, and a materialised // autosize costs its MOV words plus the ADD itself. func arm64ReturnEpilogueLen(fi arm64FrameInfo) int { if fi.autosize == 0 { return 0 } if fi.leaf { return 4 * (len(arm64AddImmWords(uint32(fi.autosize-8), 29)) + len(arm64AddImmWords(uint32(fi.autosize), 31))) } if fi.autosize <= 0xf0 { return 8 // LDR + LDR.P } // LDP + the deallocation emitted by arm64RetAddWords, so the length // tracks whatever the MOVD ladder needs. return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize))) } // arm64ResolvePseudo translates a pseudo-register memory reference into a // hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP); // x+N(SP) → (N + frame + 8)(SP). Returns base = -1 for an unresolvable // reference (SB: static data, handled by the relocation path). // // The Go toolchain resolves all pseudo-register references against the // hardware stack pointer (R31/SP): FP references add autosize+8 (the // distance from SP after the prologue to the caller's argument area), // SP references add frame+8 (the distance to the local area). func arm64ResolvePseudo(sym *ast.Symbol, fi arm64FrameInfo) (base int, off int32) { if sym == nil { return -1, 0 } switch sym.Pseudo { case "FP": return 31, int32(sym.Offset) + int32(fi.autosize) + 8 case "SP": return 31, int32(sym.Offset) + int32(fi.frame) + 8 case "SB": return -1, int32(sym.Offset) } return -1, 0 } // arm64PreStoreImm encodes a pre-index store (STR with writeback): // size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt. func arm64PreStoreImm(size, V int, imm9 int32, rn, rt int) uint32 { return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 | 3<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) } // arm64UnscaledStore encodes an unscaled store (STUR): // size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt. func arm64UnscaledStore(size, V int, imm9 int32, rn, rt int) uint32 { return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) } // arm64UnscaledLoad encodes an unscaled load (LDUR): // size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt. func arm64UnscaledLoad(size, V int, imm9 int32, rn, rt int) uint32 { return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) } // arm64PostLoad encodes a post-index load (LDR with post-increment): // size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt. func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 { return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 | 1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) } // Data-processing (shifted register) base opcodes for the guard blocks. const ( arm64OpAdd = 1<<31 | 0<<30 | 0<<29 | 0x0b<<24 arm64OpSub = 1<<31 | 1<<30 | 0<<29 | 0x0b<<24 arm64OpSubs = 1<<31 | 1<<30 | 1<<29 | 0x0b<<24 ) // arm64DPSRWords builds one data-processing (shifted register) word: // OP Rm, Rn, Rd in the Go assembler's operand order. func arm64DPSRWords(base uint32, rm, rn, rd uint32) uint32 { return base | rm<<16 | rn<<5 | rd } // arm64DPExtWords builds one data-processing (extended register) word, the // form the toolchain picks when a large immediate was materialised into // REGTMP before the operation: base | 1<<21 | Rm<<16 | UXTX<<13 | Rn<<5 | Rd. func arm64DPExtWords(base, rm, rn, rd uint32) uint32 { return base | 1<<21 | rm<<16 | 3<<13 | rn<<5 | rd } // wordsOf converts little-endian instruction bytes back to words. func wordsOf(b []byte) []uint32 { ws := make([]uint32, 0, len(b)/4) for i := 0; i+4 <= len(b); i += 4 { ws = append(ws, uint32(b[i])|uint32(b[i+1])<<8|uint32(b[i+2])<<16|uint32(b[i+3])<<24) } return ws } // arm64GuardBytes emits the stack-split guard prefix; blockStart is the // function-relative byte address of the morestack block the branches target. func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte { // MOVD 16(R28), R16 (g.stackguard0) ws := []uint32{a64LSU(3, 0, 1, 2, 28, 16)} br := func(from int, cond uint32) uint32 { return a64BranchCond(int32((blockStart-from)>>2), cond) } switch fi.splitClass { case 0: // CMP R16, RSP in the exact encoding go tool asm emits for it. ws = append(ws, 0xeb3063ff) ws = append(ws, br(8, a64CondLS)) case 1: ws = append(ws, a64AddSub(1, 1, 0, 0, uint32(fi.autosize-stackSmall), 31, 17)) ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17 ws = append(ws, br(12, a64CondLS)) default: mov, err := encodeARM64LoadImm(27, int64(fi.autosize-stackSmall), "MOVD") if err != nil { mov = nil } ws = append(ws, wordsOf(mov)...) ml := len(mov) / 4 ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27 // The branches sit at fixed byte offsets in the guard prefix: after // the LDR (4), the ml MOV words (4*ml) and the SUBS (4) for B.LO, // then a further B.LO word and the CMP for B.LS. ws = append(ws, br(8+4*ml, a64CondLO)) ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17 ws = append(ws, br(16+4*ml, a64CondLS)) } return a64WordsLE(ws...) } // arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR), // BL runtime.morestack_noctxt, B back to the function start. The BL carries // the R_CALLARM64 relocation. func arm64MoreStackBlock(blockStart int) ([]byte, Reloc) { ws := []uint32{ 1<<31 | 1<<29 | 0x0a<<24 | 30<<16 | 31<<5 | 3, // MOVD R30, R3 a64Branch(1, 0), // BL, patched by the linker } bPC := blockStart + 8 ws = append(ws, a64Branch(0, int32(-bPC>>2))) // B back to the entry reloc := Reloc{ Off: blockStart + 4, After: blockStart + 8, Name: "runtime\u00b7morestack_noctxt", Kind: RelArm64Branch, } return a64WordsLE(ws...), reloc }