421 lines
14 KiB
Go
421 lines
14 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
package asm
|
|
|
|
// arm64 frame mapping, matching the Go toolchain's arm64 backend.
|
|
//
|
|
// Go's arm64 functions use R29 as the frame pointer (FP) and R30 as the link
|
|
// register (LR). R31 is the stack pointer (SP). FP and SP in the source
|
|
// are synthetic pseudo-registers resolved against the hardware SP and the
|
|
// frame size.
|
|
//
|
|
// The autosize is the real stack adjustment: the declared local frame plus
|
|
// 8 bytes for the saved link register, rounded up to a 16-byte multiple.
|
|
// The toolchain adds an "extrasize" to align: if autosize%16 == 8, add 8;
|
|
// if autosize%16 == 0, add 16.
|
|
//
|
|
// Prologue (autosize > 0, small frame ≤ 0xf0):
|
|
//
|
|
// MOVD.W LR, -autosize(SP) // pre-index: SP -= autosize, store LR at SP
|
|
// MOVD FP, -8(SP) // store FP at SP-8
|
|
// SUB $8, SP, FP // FP = SP - 8
|
|
//
|
|
// Prologue (autosize > 0, large frame > 0xf0):
|
|
//
|
|
// SUB $autosize, SP, R20 // R20 = SP - autosize
|
|
// STP (FP, LR), -8(R20) // store FP,LR at R20-8
|
|
// MOVD R20, SP // SP = R20
|
|
// SUB $8, SP, FP // FP = SP - 8
|
|
//
|
|
// Epilogue (non-leaf, small frame):
|
|
//
|
|
// ADD $autosize-8, SP, FP // restore FP
|
|
// ADD $autosize, SP, SP // deallocate frame
|
|
// MOVD -8(SP), FP // (actually the reverse of prologue)
|
|
// Actually:
|
|
// MOVD -8(SP), FP // load FP from SP-8
|
|
// MOVD.P autosize(SP), LR // post-index: load LR, SP += autosize
|
|
//
|
|
// Epilogue (non-leaf, large frame):
|
|
// ADD $autosize-8, SP, FP
|
|
// ADD $autosize, SP, SP
|
|
// Actually:
|
|
// LDP -8(SP), (FP, LR) // load FP,LR
|
|
// ADD $autosize, SP, SP // deallocate frame
|
|
//
|
|
// Epilogue (leaf with frame):
|
|
// ADD $autosize-8, SP, FP
|
|
// ADD $autosize, SP, SP
|
|
//
|
|
// RET always emits as BR LR (0xd65f03c0).
|
|
|
|
import (
|
|
"strings"
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
|
)
|
|
|
|
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
|
|
type arm64FrameInfo struct {
|
|
autosize int // the real SP adjustment (locals + saved LR + alignment)
|
|
frame int // the declared $framesize
|
|
args int // the declared -argsize
|
|
noSplit bool // the NOSPLIT flag
|
|
leaf bool // no call instructions in the body
|
|
|
|
// Stack-split guard state: needSplit mirrors the toolchain, which skips
|
|
// the check for NOSPLIT functions and auto-marks leaf functions with an
|
|
// autosize below StackSmall as NOSPLIT.
|
|
needSplit bool
|
|
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
|
|
}
|
|
|
|
// arm64ComputeFrame derives the frame layout for a TEXT function.
|
|
func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
|
|
fi := arm64FrameInfo{
|
|
frame: frameSize(t),
|
|
args: argsSize(t),
|
|
}
|
|
for _, f := range t.Flags {
|
|
if f == "NOSPLIT" {
|
|
fi.noSplit = true
|
|
}
|
|
}
|
|
fi.leaf = arm64IsLeaf(t)
|
|
|
|
if fi.frame != 0 || !fi.leaf {
|
|
fi.autosize = fi.frame + 8 // space for the saved LR
|
|
// The toolchain always adds an extrasize: 8 when the total leaves a
|
|
// 16-byte alignment gap, another 16 when already aligned.
|
|
switch fi.autosize % 16 {
|
|
case 8:
|
|
fi.autosize += 8
|
|
case 0:
|
|
fi.autosize += 16
|
|
default:
|
|
// The toolchain rejects unaligned frames; round up so such
|
|
// sources still assemble.
|
|
fi.autosize += 16 - (fi.autosize % 16)
|
|
}
|
|
}
|
|
switch {
|
|
case fi.noSplit:
|
|
case fi.autosize < stackSmall && fi.leaf:
|
|
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
|
|
default:
|
|
fi.needSplit = true
|
|
switch {
|
|
case fi.autosize <= stackSmall:
|
|
fi.splitClass = 0
|
|
case fi.autosize <= stackBig:
|
|
fi.splitClass = 1
|
|
default:
|
|
fi.splitClass = 2
|
|
}
|
|
}
|
|
return fi
|
|
}
|
|
|
|
// arm64GuardLen returns the byte length of the stack-split guard prefix
|
|
// (zero when the function needs no guard). The big class materialises
|
|
// framesize-StackSmall into REGTMP, whose MOVZ/MOVK sequence length varies.
|
|
func arm64GuardLen(fi arm64FrameInfo) int {
|
|
if !fi.needSplit {
|
|
return 0
|
|
}
|
|
switch fi.splitClass {
|
|
case 0:
|
|
return 12
|
|
case 1:
|
|
return 16
|
|
default:
|
|
n, err := arm64LoadImmLen(int64(fi.autosize - stackSmall))
|
|
if err != nil {
|
|
return 0
|
|
}
|
|
return 4 + n + 4 + 4 + 4 + 4
|
|
}
|
|
}
|
|
|
|
// arm64LoadImmLen returns the byte length of the MOVZ/MOVK sequence that
|
|
// loads v into a register.
|
|
func arm64LoadImmLen(v int64) (int, error) {
|
|
b, err := encodeARM64LoadImm(27, v, "MOVD")
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
return len(b), nil
|
|
}
|
|
|
|
// arm64IsLeaf reports whether a function contains no call instructions
|
|
// (BL/CALL), matching the toolchain's LEAF mark.
|
|
func arm64IsLeaf(t *ast.Text) bool {
|
|
for _, stmt := range t.Body {
|
|
in, ok := stmt.(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
switch strings.ToUpper(in.Mnemonic.Text) {
|
|
case "BL", "CALL":
|
|
return false
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
|
|
// arm64Prologue returns the prologue bytes for an arm64 function.
|
|
func arm64Prologue(fi arm64FrameInfo) []byte {
|
|
if fi.autosize == 0 {
|
|
return nil
|
|
}
|
|
if fi.autosize <= 0xf0 {
|
|
// Small frame: MOVD.W LR, -autosize(SP); MOVD FP, -8(SP); SUB $8, SP, FP
|
|
return a64WordsLE(
|
|
arm64PreStoreImm(3, 0, int32(-fi.autosize), 31, 30), // STR.W LR, -autosize(SP) (pre-index store)
|
|
arm64UnscaledStore(3, 0, -8, 31, 29), // STUR FP, [SP, #-8]
|
|
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
|
|
)
|
|
}
|
|
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
|
|
ws := arm64SubImmWords(uint32(fi.autosize), 20)
|
|
ws = append(ws,
|
|
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
|
|
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
|
|
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
|
|
)
|
|
return a64WordsLE(ws...)
|
|
}
|
|
|
|
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value
|
|
// fits the imm12 field (plain, or shifted left by 12 when it is a multiple
|
|
// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and
|
|
// subtracts the register in the extended-register form.
|
|
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
|
if imm <= 0xFFF {
|
|
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
|
}
|
|
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
|
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
|
|
}
|
|
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
|
if err != nil {
|
|
mov = nil
|
|
}
|
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
|
}
|
|
|
|
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12
|
|
// and REGTMP fallback ladder.
|
|
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
|
if imm <= 0xFFF {
|
|
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
|
}
|
|
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
|
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
|
|
}
|
|
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
|
if err != nil {
|
|
mov = nil
|
|
}
|
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
|
|
}
|
|
|
|
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
|
// deallocate the frame when present) followed by RET (BR LR).
|
|
func arm64Return(fi arm64FrameInfo) []byte {
|
|
var ws []uint32
|
|
if fi.autosize != 0 {
|
|
if fi.leaf {
|
|
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
|
|
ws = append(ws, arm64AddImmWords(uint32(fi.autosize-8), 29)...)
|
|
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
|
|
} else if fi.autosize <= 0xf0 {
|
|
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
|
|
ws = append(ws,
|
|
arm64UnscaledLoad(3, 0, -8, 31, 29), // LDR FP, [SP, #-8]
|
|
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
|
)
|
|
} else {
|
|
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
|
|
ws = append(ws,
|
|
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
|
)
|
|
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
|
|
}
|
|
}
|
|
// RET: BR LR (0xd65f03c0)
|
|
ws = append(ws, a64UncondBranch(2, 30, 0)) // opc=2(RET), Rn=LR(30), Rd=0
|
|
return a64WordsLE(ws...)
|
|
}
|
|
|
|
// arm64PrologueSpadjPC returns the function-relative byte offset where the
|
|
// prologue has finished decrementing SP (the delta becomes autosize).
|
|
func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
|
|
if fi.autosize == 0 {
|
|
return 0
|
|
}
|
|
if fi.autosize <= 0xf0 {
|
|
return 4 // MOVD.W instruction decrements SP
|
|
}
|
|
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
|
|
// position depends on how many words the SUB itself took (immediate,
|
|
// shifted immediate, or a materialised REGTMP sequence).
|
|
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
|
|
}
|
|
|
|
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
|
// (but not including) the final RET instruction. The ADD sequences share the
|
|
// prologue's immediate ladder, so their length is read from the same helper
|
|
// rather than assumed: a materialised autosize costs its MOV words plus the
|
|
// ADD itself.
|
|
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
|
if fi.autosize == 0 {
|
|
return 0
|
|
}
|
|
if fi.leaf {
|
|
return 4 * (len(arm64AddImmWords(uint32(fi.autosize-8), 29)) +
|
|
len(arm64AddImmWords(uint32(fi.autosize), 31)))
|
|
}
|
|
if fi.autosize <= 0xf0 {
|
|
return 8 // LDR + LDR.P
|
|
}
|
|
// LDP + the ADD ladder that deallocates the frame.
|
|
return 4 + 4*len(arm64AddImmWords(uint32(fi.autosize), 31))
|
|
}
|
|
|
|
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
|
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
|
|
// x+N(SP) → (N + frame + 8)(SP). Returns base = -1 for an unresolvable
|
|
// reference (SB: static data, handled by the relocation path).
|
|
//
|
|
// The Go toolchain resolves all pseudo-register references against the
|
|
// hardware stack pointer (R31/SP): FP references add autosize+8 (the
|
|
// distance from SP after the prologue to the caller's argument area),
|
|
// SP references add frame+8 (the distance to the local area).
|
|
func arm64ResolvePseudo(sym *ast.Symbol, fi arm64FrameInfo) (base int, off int32) {
|
|
if sym == nil {
|
|
return -1, 0
|
|
}
|
|
switch sym.Pseudo {
|
|
case "FP":
|
|
return 31, int32(sym.Offset) + int32(fi.autosize) + 8
|
|
case "SP":
|
|
return 31, int32(sym.Offset) + int32(fi.frame) + 8
|
|
case "SB":
|
|
return -1, int32(sym.Offset)
|
|
}
|
|
return -1, 0
|
|
}
|
|
|
|
// arm64PreStoreImm encodes a pre-index store (STR with writeback):
|
|
// size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
|
|
func arm64PreStoreImm(size, V int, imm9 int32, rn, rt int) uint32 {
|
|
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
|
|
3<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
|
}
|
|
|
|
// arm64UnscaledStore encodes an unscaled store (STUR):
|
|
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
|
|
func arm64UnscaledStore(size, V int, imm9 int32, rn, rt int) uint32 {
|
|
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
|
|
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
|
}
|
|
|
|
// arm64UnscaledLoad encodes an unscaled load (LDUR):
|
|
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
|
|
func arm64UnscaledLoad(size, V int, imm9 int32, rn, rt int) uint32 {
|
|
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
|
|
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
|
}
|
|
|
|
// arm64PostLoad encodes a post-index load (LDR with post-increment):
|
|
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
|
|
func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
|
|
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
|
|
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
|
}
|
|
|
|
// Data-processing (shifted register) base opcodes for the guard blocks.
|
|
const (
|
|
arm64OpAdd = 1<<31 | 0<<30 | 0<<29 | 0x0b<<24
|
|
arm64OpSub = 1<<31 | 1<<30 | 0<<29 | 0x0b<<24
|
|
arm64OpSubs = 1<<31 | 1<<30 | 1<<29 | 0x0b<<24
|
|
)
|
|
|
|
// arm64DPSRWords builds one data-processing (shifted register) word:
|
|
// OP Rm, Rn, Rd in the Go assembler's operand order.
|
|
func arm64DPSRWords(base uint32, rm, rn, rd uint32) uint32 {
|
|
return base | rm<<16 | rn<<5 | rd
|
|
}
|
|
|
|
// arm64DPExtWords builds one data-processing (extended register) word, the
|
|
// form the toolchain picks when a large immediate was materialised into
|
|
// REGTMP before the operation: base | 1<<21 | Rm<<16 | UXTX<<13 | Rn<<5 | Rd.
|
|
func arm64DPExtWords(base, rm, rn, rd uint32) uint32 {
|
|
return base | 1<<21 | rm<<16 | 3<<13 | rn<<5 | rd
|
|
}
|
|
|
|
// wordsOf converts little-endian instruction bytes back to words.
|
|
func wordsOf(b []byte) []uint32 {
|
|
ws := make([]uint32, 0, len(b)/4)
|
|
for i := 0; i+4 <= len(b); i += 4 {
|
|
ws = append(ws, uint32(b[i])|uint32(b[i+1])<<8|uint32(b[i+2])<<16|uint32(b[i+3])<<24)
|
|
}
|
|
return ws
|
|
}
|
|
|
|
// arm64GuardBytes emits the stack-split guard prefix; blockStart is the
|
|
// function-relative byte address of the morestack block the branches target.
|
|
func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
|
|
// MOVD 16(R28), R16 (g.stackguard0)
|
|
ws := []uint32{a64LSU(3, 0, 1, 2, 28, 16)}
|
|
br := func(from int, cond uint32) uint32 {
|
|
return a64BranchCond(int32((blockStart-from)>>2), cond)
|
|
}
|
|
switch fi.splitClass {
|
|
case 0:
|
|
// CMP R16, RSP in the exact encoding go tool asm emits for it.
|
|
ws = append(ws, 0xeb3063ff)
|
|
ws = append(ws, br(8, a64CondLS))
|
|
case 1:
|
|
ws = append(ws, a64AddSub(1, 1, 0, 0, uint32(fi.autosize-stackSmall), 31, 17))
|
|
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
|
|
ws = append(ws, br(12, a64CondLS))
|
|
default:
|
|
mov, err := encodeARM64LoadImm(27, int64(fi.autosize-stackSmall), "MOVD")
|
|
if err != nil {
|
|
mov = nil
|
|
}
|
|
ws = append(ws, wordsOf(mov)...)
|
|
ml := len(mov) / 4
|
|
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
|
|
// The branches sit at fixed byte offsets in the guard prefix: after
|
|
// the LDR (4), the ml MOV words (4*ml) and the SUBS (4) for B.LO,
|
|
// then a further B.LO word and the CMP for B.LS.
|
|
ws = append(ws, br(8+4*ml, a64CondLO))
|
|
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
|
|
ws = append(ws, br(16+4*ml, a64CondLS))
|
|
}
|
|
return a64WordsLE(ws...)
|
|
}
|
|
|
|
// arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR),
|
|
// BL runtime.morestack_noctxt, B back to the function start. The BL carries
|
|
// the R_CALLARM64 relocation.
|
|
func arm64MoreStackBlock(blockStart int) ([]byte, Reloc) {
|
|
ws := []uint32{
|
|
1<<31 | 1<<29 | 0x0a<<24 | 30<<16 | 31<<5 | 3, // MOVD R30, R3
|
|
a64Branch(1, 0), // BL, patched by the linker
|
|
}
|
|
bPC := blockStart + 8
|
|
ws = append(ws, a64Branch(0, int32(-bPC>>2))) // B back to the entry
|
|
reloc := Reloc{
|
|
Off: blockStart + 4,
|
|
After: blockStart + 8,
|
|
Name: "runtime\u00b7morestack_noctxt",
|
|
Kind: RelArm64Branch,
|
|
}
|
|
return a64WordsLE(ws...), reloc
|
|
}
|