From 8dc1e98ca1665c75a03100c7dff9d78713128a56 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Mon, 14 Sep 2026 20:49:03 +0200 Subject: [PATCH] feat(asm): emit the arm64 stack-split guard and morestack block --- asm/arm64_assemble.go | 35 ++++++-- asm/arm64_frame.go | 172 +++++++++++++++++++++++++++++++++++++--- asm/arm64_reloc_test.go | 19 +++-- asm/goobj.go | 2 +- asm/guard_test.go | 43 ++++++++++ cmd/gasm/unidiff.go | 6 +- 6 files changed, 248 insertions(+), 29 deletions(-) diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index 5f8c42d..f2643e6 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -26,6 +26,7 @@ import ( func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { fi := arm64ComputeFrame(t) prologue := arm64Prologue(fi) + guardLen := arm64GuardLen(fi) chain := arm64JumpChain(t) resolve := func(name string) string { if r, ok := chain[name]; ok { @@ -38,14 +39,14 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ var spadj []SpadjStep // The prologue (3 instructions when a small frame, 4 for large) - // raises the SP delta by autosize. + // raises the SP delta by autosize. The guard prefix shifts its PC. if fi.autosize != 0 { - spadj = append(spadj, SpadjStep{PC: arm64PrologueSpadjPC(fi), Value: fi.autosize}) + spadj = append(spadj, SpadjStep{PC: guardLen + arm64PrologueSpadjPC(fi), Value: fi.autosize}) } // Pass 1: label offsets from the instruction sizes. offsets := map[string]int{} - pos := len(prologue) + pos := guardLen + len(prologue) for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: @@ -55,9 +56,25 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ } } - // Pass 2: encode. Relocation offsets are recorded function-relative. - out := append([]byte(nil), prologue...) - pc := len(prologue) + // Pass 2: encode. The guard prefix precedes the prologue; its branches + // target the morestack block at the end of the function, whose position + // the first pass has settled. + bodyLen := 0 + { + p := guardLen + len(prologue) + for _, stmt := range t.Body { + if in, ok := stmt.(*ast.Instr); ok { + p += arm64InstrSize(in, fi) + } + } + bodyLen = p - (guardLen + len(prologue)) + } + var out []byte + if fi.needSplit { + out = append(out, arm64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...) + } + out = append(out, prologue...) + pc := guardLen + len(prologue) preCount := len(relocs) var lines []LineEntry for _, stmt := range t.Body { @@ -87,6 +104,12 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ out = append(out, code...) pc += len(code) } + if fi.needSplit { + block, blReloc := arm64MoreStackBlock(pc) + out = append(out, block...) + relocs = append(relocs, blReloc) + pc += len(block) + } return out, offsets, relocs, lines, spadj, nil } diff --git a/asm/arm64_frame.go b/asm/arm64_frame.go index 32bc882..6252751 100644 --- a/asm/arm64_frame.go +++ b/asm/arm64_frame.go @@ -63,6 +63,12 @@ type arm64FrameInfo struct { args int // the declared -argsize noSplit bool // the NOSPLIT flag leaf bool // no call instructions in the body + + // Stack-split guard state: needSplit mirrors the toolchain, which skips + // the check for NOSPLIT functions and auto-marks leaf functions with an + // autosize below StackSmall as NOSPLIT. + needSplit bool + splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig } // arm64ComputeFrame derives the frame layout for a TEXT function. @@ -86,9 +92,55 @@ func arm64ComputeFrame(t *ast.Text) arm64FrameInfo { fi.autosize += 16 - (fi.autosize % 16) } } + switch { + case fi.noSplit: + case fi.autosize < stackSmall && fi.leaf: + // Auto-NOSPLIT, as the toolchain's leaf mark concludes. + default: + fi.needSplit = true + switch { + case fi.autosize <= stackSmall: + fi.splitClass = 0 + case fi.autosize <= stackBig: + fi.splitClass = 1 + default: + fi.splitClass = 2 + } + } return fi } +// arm64GuardLen returns the byte length of the stack-split guard prefix +// (zero when the function needs no guard). The big class materialises +// framesize-StackSmall into REGTMP, whose MOVZ/MOVK sequence length varies. +func arm64GuardLen(fi arm64FrameInfo) int { + if !fi.needSplit { + return 0 + } + switch fi.splitClass { + case 0: + return 12 + case 1: + return 16 + default: + n, err := arm64LoadImmLen(int64(fi.autosize - stackSmall)) + if err != nil { + return 0 + } + return 4 + n + 4 + 4 + 4 + 4 + } +} + +// arm64LoadImmLen returns the byte length of the MOVZ/MOVK sequence that +// loads v into a register. +func arm64LoadImmLen(v int64) (int, error) { + b, err := encodeARM64LoadImm(27, v, "MOVD") + if err != nil { + return 0, err + } + return len(b), nil +} + // arm64IsLeaf reports whether a function contains no call instructions // (BL/CALL), matching the toolchain's LEAF mark. func arm64IsLeaf(t *ast.Text) bool { @@ -119,12 +171,39 @@ func arm64Prologue(fi arm64FrameInfo) []byte { ) } // Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP - return a64WordsLE( - a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20 - a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair) - a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP) - a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB) + ws := arm64SubImmWords(uint32(fi.autosize), 20) + ws = append(ws, + a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair) + a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP) + a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB) ) + return a64WordsLE(ws...) +} + +// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value +// fits the imm12 field, otherwise the toolchain materialises it into REGTMP +// (R27) and subtracts the register. +func arm64SubImmWords(imm uint32, rd uint32) []uint32 { + if imm <= 0xFFF { + return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)} + } + mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") + if err != nil { + mov = nil + } + return append(wordsOf(mov), arm64DPSRWords(arm64OpSub, 27, 31, rd)) +} + +// arm64AddImmWords emits ADD $imm, SP, Rd with the same REGTMP fallback. +func arm64AddImmWords(imm uint32, rd uint32) []uint32 { + if imm <= 0xFFF { + return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)} + } + mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") + if err != nil { + mov = nil + } + return append(wordsOf(mov), arm64DPSRWords(arm64OpAdd, 27, 31, rd)) } // arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and @@ -134,10 +213,8 @@ func arm64Return(fi arm64FrameInfo) []byte { if fi.autosize != 0 { if fi.leaf { // Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP - ws = append(ws, - a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP - a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP - ) + ws = append(ws, arm64AddImmWords(uint32(fi.autosize-8), 29)...) + ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...) } else if fi.autosize <= 0xf0 { // Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize ws = append(ws, @@ -147,9 +224,9 @@ func arm64Return(fi arm64FrameInfo) []byte { } else { // Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP ws = append(ws, - a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair) - a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP + a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair) ) + ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...) } } // RET: BR LR (0xd65f03c0) @@ -235,3 +312,76 @@ func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 { return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 | 1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) } + +// Data-processing (shifted register) base opcodes for the guard blocks. +const ( + arm64OpAdd = 1<<31 | 0<<30 | 0<<29 | 0x0b<<24 + arm64OpSub = 1<<31 | 1<<30 | 0<<29 | 0x0b<<24 + arm64OpSubs = 1<<31 | 1<<30 | 1<<29 | 0x0b<<24 +) + +// arm64DPSRWords builds one data-processing (shifted register) word: +// OP Rm, Rn, Rd in the Go assembler's operand order. +func arm64DPSRWords(base uint32, rm, rn, rd uint32) uint32 { + return base | rm<<16 | rn<<5 | rd +} + +// wordsOf converts little-endian instruction bytes back to words. +func wordsOf(b []byte) []uint32 { + ws := make([]uint32, 0, len(b)/4) + for i := 0; i+4 <= len(b); i += 4 { + ws = append(ws, uint32(b[i])|uint32(b[i+1])<<8|uint32(b[i+2])<<16|uint32(b[i+3])<<24) + } + return ws +} + +// arm64GuardBytes emits the stack-split guard prefix; blockStart is the +// function-relative byte address of the morestack block the branches target. +func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte { + // MOVD 16(R28), R16 (g.stackguard0) + ws := []uint32{a64LSU(3, 0, 1, 2, 28, 16)} + br := func(from int, cond uint32) uint32 { + return a64BranchCond(int32((blockStart-from)>>2), cond) + } + switch fi.splitClass { + case 0: + // CMP R16, RSP in the exact encoding go tool asm emits for it. + ws = append(ws, 0xeb3063ff) + ws = append(ws, br(8, a64CondLS)) + case 1: + ws = append(ws, a64AddSub(1, 1, 0, 0, uint32(fi.autosize-stackSmall), 31, 17)) + ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17 + ws = append(ws, br(12, a64CondLS)) + default: + mov, err := encodeARM64LoadImm(27, int64(fi.autosize-stackSmall), "MOVD") + if err != nil { + mov = nil + } + ws = append(ws, wordsOf(mov)...) + ml := len(mov) / 4 + ws = append(ws, arm64DPSRWords(arm64OpSubs, 27, 31, 17)) // SUBS R27, RSP, R17 + ws = append(ws, br(8+ml, a64CondLO)) + ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17 + ws = append(ws, br(8+ml+8, a64CondLS)) + } + return a64WordsLE(ws...) +} + +// arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR), +// BL runtime.morestack_noctxt, B back to the function start. The BL carries +// the R_CALLARM64 relocation. +func arm64MoreStackBlock(blockStart int) ([]byte, Reloc) { + ws := []uint32{ + 1<<31 | 1<<29 | 0x0a<<24 | 30<<16 | 31<<5 | 3, // MOVD R30, R3 + a64Branch(1, 0), // BL, patched by the linker + } + bPC := blockStart + 8 + ws = append(ws, a64Branch(0, int32(-bPC>>2))) // B back to the entry + reloc := Reloc{ + Off: blockStart + 4, + After: blockStart + 8, + Name: "runtime\u00b7morestack_noctxt", + Kind: RelArm64Branch, + } + return a64WordsLE(ws...), reloc +} diff --git a/asm/arm64_reloc_test.go b/asm/arm64_reloc_test.go index f994207..5a57e44 100644 --- a/asm/arm64_reloc_test.go +++ b/asm/arm64_reloc_test.go @@ -27,6 +27,8 @@ func parseArm64File(t *testing.T, src string) *Image { // TestArm64RelocOffsetsIncludePrologue pins the function-relative relocation // offsets of a framed function: the offsets used to exclude the prologue, so // every relocation landed on a prologue instruction in the GOOBJ/ELF output. +// The function calls an external, so it is a non-leaf and carries the +// stack-split guard (12 bytes, small class) before the prologue. func TestArm64RelocOffsetsIncludePrologue(t *testing.T) { img := parseArm64File(t, "TEXT \u00b7f(SB), $16-0\n"+ "\tBL ext\u00b7foo(SB)\n"+ @@ -36,8 +38,8 @@ func TestArm64RelocOffsetsIncludePrologue(t *testing.T) { "GLOBL gdata(SB), $8\n") fn := img.Funcs[0] - // Layout: 12-byte prologue, BL (12), ADRP+ADD (16, 20), ADRP+ADD (24, 28), - // 12-byte epilogue with RET. + // Layout: 12-byte guard, 12-byte prologue, BL (24), ADRP+ADD (28, 32), + // ADRP+ADD (36, 40), 12-byte epilogue with RET, 12-byte morestack block. want := []struct { off int after int @@ -45,11 +47,12 @@ func TestArm64RelocOffsetsIncludePrologue(t *testing.T) { kind RelocKind external bool }{ - {12, 16, "foo", RelArm64Branch, true}, - {16, 16, "gdata", RelArm64Addr, false}, - {20, 20, "gdata", RelArm64Addr, false}, - {24, 24, "extsym", RelArm64Addr, true}, - {28, 28, "extsym", RelArm64Addr, true}, + {24, 28, "foo", RelArm64Branch, true}, + {28, 28, "gdata", RelArm64Addr, false}, + {32, 32, "gdata", RelArm64Addr, false}, + {36, 36, "extsym", RelArm64Addr, true}, + {40, 40, "extsym", RelArm64Addr, true}, + {60, 64, "runtime\u00b7morestack_noctxt", RelArm64Branch, true}, } if len(fn.Relocs) != len(want) { t.Fatalf("relocs = %d, want %d", len(fn.Relocs), len(want)) @@ -64,7 +67,7 @@ func TestArm64RelocOffsetsIncludePrologue(t *testing.T) { // The BL with a zero offset sits exactly at the first reloc site. code := img.Code[fn.Offset : fn.Offset+fn.Size] - if w := binary.LittleEndian.Uint32(code[12:16]); w != 0x94000000 { + if w := binary.LittleEndian.Uint32(code[24:28]); w != 0x94000000 { t.Errorf("BL word = %08x, want 94000000", w) } } diff --git a/asm/goobj.go b/asm/goobj.go index f0be041..b470652 100644 --- a/asm/goobj.go +++ b/asm/goobj.go @@ -393,7 +393,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r symRelocs[si] = append(symRelocs[si], rec[:]...) continue } - if r.External && r.Kind == RelCall && r.Name == goobjBuiltinMorestack { + if r.External && r.Name == goobjBuiltinMorestack { // The stack-guard morestack call uses the toolchain's // builtin reference. var rec [23]byte diff --git a/asm/guard_test.go b/asm/guard_test.go index 33cd08e..84fbd9c 100644 --- a/asm/guard_test.go +++ b/asm/guard_test.go @@ -100,3 +100,46 @@ func TestStackGuardGOObj(t *testing.T) { t.Fatal("object lacks the GOOBJ magic") } } + +// The arm64 stack-split guard, pinned from `go tool asm` (Go 1.27, arm64): +// the guard classes, the auto-NOSPLIT leaf behaviour and the morestack +// block. Relocation fields are masked: the toolchain's object leaves them +// zero for the linker, the gasm image resolves file-internal references. +func TestStackGuardBytesARM64(t *testing.T) { + for _, tt := range []struct { + name string + src string + want string + }{ + {"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n", + "fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"}, + {"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n", + "900b40f9f14302d13f0210eb09010054f44304d19dfa3fa99f020091fd2300d1fd230491ff430491c0035fd6e3031eaa00000000f3ffff17"}, + {"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n", + "900b40f91bf283d2f1031beba30100543f0210eb690100541b0284d2f4031bcb9dfa3fa99f020091fd2300d11b0184d2fd031b8b1b0284d2ff031b8bc0035fd6e3031eaa00000000eeffff17"}, + {"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n", + "900b40f9ff6330eb09010054fe0f1ef8fd831ff8fd2300d100000000fd835ff8fe0742f8c0035fd6e3031eaa00000000f4ffff17"}, + {"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n", + "fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"}, + } { + f, errs := parser.Parse("g_arm64.s", tt.src) + if len(errs) > 0 { + t.Fatalf("%s: parse: %v", tt.name, errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("%s: assemble: %v", tt.name, err) + } + fn := img.Funcs[0] + code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...) + for _, r := range fn.Relocs { + for j := r.Off; j < r.Off+4 && j < len(code); j++ { + code[j] = 0 + } + } + got := hex.EncodeToString(code) + if got != tt.want { + t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want) + } + } +} diff --git a/cmd/gasm/unidiff.go b/cmd/gasm/unidiff.go index 9277fe5..e0a1710 100644 --- a/cmd/gasm/unidiff.go +++ b/cmd/gasm/unidiff.go @@ -41,9 +41,9 @@ func unifiedDiff(name string, a, b []string) string { // Walk the LCS once, assigning every op its absolute position in both // files (1-based, the position an insertion sits before). type op struct { - kind byte // ' ', '-' or '+' - aLine, bLine int - text string + kind byte // ' ', '-' or '+' + aLine, bLine int + text string } var ops []op aPos, bPos := 0, 0