feat(asm): emit the arm64 stack-split guard and morestack block
This commit is contained in:
+29
-6
@@ -26,6 +26,7 @@ import (
|
||||
func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
|
||||
fi := arm64ComputeFrame(t)
|
||||
prologue := arm64Prologue(fi)
|
||||
guardLen := arm64GuardLen(fi)
|
||||
chain := arm64JumpChain(t)
|
||||
resolve := func(name string) string {
|
||||
if r, ok := chain[name]; ok {
|
||||
@@ -38,14 +39,14 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
|
||||
var spadj []SpadjStep
|
||||
|
||||
// The prologue (3 instructions when a small frame, 4 for large)
|
||||
// raises the SP delta by autosize.
|
||||
// raises the SP delta by autosize. The guard prefix shifts its PC.
|
||||
if fi.autosize != 0 {
|
||||
spadj = append(spadj, SpadjStep{PC: arm64PrologueSpadjPC(fi), Value: fi.autosize})
|
||||
spadj = append(spadj, SpadjStep{PC: guardLen + arm64PrologueSpadjPC(fi), Value: fi.autosize})
|
||||
}
|
||||
|
||||
// Pass 1: label offsets from the instruction sizes.
|
||||
offsets := map[string]int{}
|
||||
pos := len(prologue)
|
||||
pos := guardLen + len(prologue)
|
||||
for _, stmt := range t.Body {
|
||||
switch s := stmt.(type) {
|
||||
case *ast.Label:
|
||||
@@ -55,9 +56,25 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2: encode. Relocation offsets are recorded function-relative.
|
||||
out := append([]byte(nil), prologue...)
|
||||
pc := len(prologue)
|
||||
// Pass 2: encode. The guard prefix precedes the prologue; its branches
|
||||
// target the morestack block at the end of the function, whose position
|
||||
// the first pass has settled.
|
||||
bodyLen := 0
|
||||
{
|
||||
p := guardLen + len(prologue)
|
||||
for _, stmt := range t.Body {
|
||||
if in, ok := stmt.(*ast.Instr); ok {
|
||||
p += arm64InstrSize(in, fi)
|
||||
}
|
||||
}
|
||||
bodyLen = p - (guardLen + len(prologue))
|
||||
}
|
||||
var out []byte
|
||||
if fi.needSplit {
|
||||
out = append(out, arm64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...)
|
||||
}
|
||||
out = append(out, prologue...)
|
||||
pc := guardLen + len(prologue)
|
||||
preCount := len(relocs)
|
||||
var lines []LineEntry
|
||||
for _, stmt := range t.Body {
|
||||
@@ -87,6 +104,12 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
|
||||
out = append(out, code...)
|
||||
pc += len(code)
|
||||
}
|
||||
if fi.needSplit {
|
||||
block, blReloc := arm64MoreStackBlock(pc)
|
||||
out = append(out, block...)
|
||||
relocs = append(relocs, blReloc)
|
||||
pc += len(block)
|
||||
}
|
||||
return out, offsets, relocs, lines, spadj, nil
|
||||
}
|
||||
|
||||
|
||||
+161
-11
@@ -63,6 +63,12 @@ type arm64FrameInfo struct {
|
||||
args int // the declared -argsize
|
||||
noSplit bool // the NOSPLIT flag
|
||||
leaf bool // no call instructions in the body
|
||||
|
||||
// Stack-split guard state: needSplit mirrors the toolchain, which skips
|
||||
// the check for NOSPLIT functions and auto-marks leaf functions with an
|
||||
// autosize below StackSmall as NOSPLIT.
|
||||
needSplit bool
|
||||
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
|
||||
}
|
||||
|
||||
// arm64ComputeFrame derives the frame layout for a TEXT function.
|
||||
@@ -86,9 +92,55 @@ func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
|
||||
fi.autosize += 16 - (fi.autosize % 16)
|
||||
}
|
||||
}
|
||||
switch {
|
||||
case fi.noSplit:
|
||||
case fi.autosize < stackSmall && fi.leaf:
|
||||
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
|
||||
default:
|
||||
fi.needSplit = true
|
||||
switch {
|
||||
case fi.autosize <= stackSmall:
|
||||
fi.splitClass = 0
|
||||
case fi.autosize <= stackBig:
|
||||
fi.splitClass = 1
|
||||
default:
|
||||
fi.splitClass = 2
|
||||
}
|
||||
}
|
||||
return fi
|
||||
}
|
||||
|
||||
// arm64GuardLen returns the byte length of the stack-split guard prefix
|
||||
// (zero when the function needs no guard). The big class materialises
|
||||
// framesize-StackSmall into REGTMP, whose MOVZ/MOVK sequence length varies.
|
||||
func arm64GuardLen(fi arm64FrameInfo) int {
|
||||
if !fi.needSplit {
|
||||
return 0
|
||||
}
|
||||
switch fi.splitClass {
|
||||
case 0:
|
||||
return 12
|
||||
case 1:
|
||||
return 16
|
||||
default:
|
||||
n, err := arm64LoadImmLen(int64(fi.autosize - stackSmall))
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
return 4 + n + 4 + 4 + 4 + 4
|
||||
}
|
||||
}
|
||||
|
||||
// arm64LoadImmLen returns the byte length of the MOVZ/MOVK sequence that
|
||||
// loads v into a register.
|
||||
func arm64LoadImmLen(v int64) (int, error) {
|
||||
b, err := encodeARM64LoadImm(27, v, "MOVD")
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(b), nil
|
||||
}
|
||||
|
||||
// arm64IsLeaf reports whether a function contains no call instructions
|
||||
// (BL/CALL), matching the toolchain's LEAF mark.
|
||||
func arm64IsLeaf(t *ast.Text) bool {
|
||||
@@ -119,12 +171,39 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
|
||||
)
|
||||
}
|
||||
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20
|
||||
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
|
||||
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
|
||||
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
|
||||
ws := arm64SubImmWords(uint32(fi.autosize), 20)
|
||||
ws = append(ws,
|
||||
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
|
||||
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
|
||||
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
|
||||
)
|
||||
return a64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value
|
||||
// fits the imm12 field, otherwise the toolchain materialises it into REGTMP
|
||||
// (R27) and subtracts the register.
|
||||
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
||||
if imm <= 0xFFF {
|
||||
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
||||
}
|
||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
}
|
||||
return append(wordsOf(mov), arm64DPSRWords(arm64OpSub, 27, 31, rd))
|
||||
}
|
||||
|
||||
// arm64AddImmWords emits ADD $imm, SP, Rd with the same REGTMP fallback.
|
||||
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
||||
if imm <= 0xFFF {
|
||||
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
||||
}
|
||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
}
|
||||
return append(wordsOf(mov), arm64DPSRWords(arm64OpAdd, 27, 31, rd))
|
||||
}
|
||||
|
||||
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
||||
@@ -134,10 +213,8 @@ func arm64Return(fi arm64FrameInfo) []byte {
|
||||
if fi.autosize != 0 {
|
||||
if fi.leaf {
|
||||
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
|
||||
ws = append(ws,
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
|
||||
)
|
||||
ws = append(ws, arm64AddImmWords(uint32(fi.autosize-8), 29)...)
|
||||
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
|
||||
} else if fi.autosize <= 0xf0 {
|
||||
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
|
||||
ws = append(ws,
|
||||
@@ -147,9 +224,9 @@ func arm64Return(fi arm64FrameInfo) []byte {
|
||||
} else {
|
||||
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
|
||||
ws = append(ws,
|
||||
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
|
||||
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
||||
)
|
||||
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
|
||||
}
|
||||
}
|
||||
// RET: BR LR (0xd65f03c0)
|
||||
@@ -235,3 +312,76 @@ func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
|
||||
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
|
||||
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
||||
}
|
||||
|
||||
// Data-processing (shifted register) base opcodes for the guard blocks.
|
||||
const (
|
||||
arm64OpAdd = 1<<31 | 0<<30 | 0<<29 | 0x0b<<24
|
||||
arm64OpSub = 1<<31 | 1<<30 | 0<<29 | 0x0b<<24
|
||||
arm64OpSubs = 1<<31 | 1<<30 | 1<<29 | 0x0b<<24
|
||||
)
|
||||
|
||||
// arm64DPSRWords builds one data-processing (shifted register) word:
|
||||
// OP Rm, Rn, Rd in the Go assembler's operand order.
|
||||
func arm64DPSRWords(base uint32, rm, rn, rd uint32) uint32 {
|
||||
return base | rm<<16 | rn<<5 | rd
|
||||
}
|
||||
|
||||
// wordsOf converts little-endian instruction bytes back to words.
|
||||
func wordsOf(b []byte) []uint32 {
|
||||
ws := make([]uint32, 0, len(b)/4)
|
||||
for i := 0; i+4 <= len(b); i += 4 {
|
||||
ws = append(ws, uint32(b[i])|uint32(b[i+1])<<8|uint32(b[i+2])<<16|uint32(b[i+3])<<24)
|
||||
}
|
||||
return ws
|
||||
}
|
||||
|
||||
// arm64GuardBytes emits the stack-split guard prefix; blockStart is the
|
||||
// function-relative byte address of the morestack block the branches target.
|
||||
func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
|
||||
// MOVD 16(R28), R16 (g.stackguard0)
|
||||
ws := []uint32{a64LSU(3, 0, 1, 2, 28, 16)}
|
||||
br := func(from int, cond uint32) uint32 {
|
||||
return a64BranchCond(int32((blockStart-from)>>2), cond)
|
||||
}
|
||||
switch fi.splitClass {
|
||||
case 0:
|
||||
// CMP R16, RSP in the exact encoding go tool asm emits for it.
|
||||
ws = append(ws, 0xeb3063ff)
|
||||
ws = append(ws, br(8, a64CondLS))
|
||||
case 1:
|
||||
ws = append(ws, a64AddSub(1, 1, 0, 0, uint32(fi.autosize-stackSmall), 31, 17))
|
||||
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
|
||||
ws = append(ws, br(12, a64CondLS))
|
||||
default:
|
||||
mov, err := encodeARM64LoadImm(27, int64(fi.autosize-stackSmall), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
}
|
||||
ws = append(ws, wordsOf(mov)...)
|
||||
ml := len(mov) / 4
|
||||
ws = append(ws, arm64DPSRWords(arm64OpSubs, 27, 31, 17)) // SUBS R27, RSP, R17
|
||||
ws = append(ws, br(8+ml, a64CondLO))
|
||||
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
|
||||
ws = append(ws, br(8+ml+8, a64CondLS))
|
||||
}
|
||||
return a64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR),
|
||||
// BL runtime.morestack_noctxt, B back to the function start. The BL carries
|
||||
// the R_CALLARM64 relocation.
|
||||
func arm64MoreStackBlock(blockStart int) ([]byte, Reloc) {
|
||||
ws := []uint32{
|
||||
1<<31 | 1<<29 | 0x0a<<24 | 30<<16 | 31<<5 | 3, // MOVD R30, R3
|
||||
a64Branch(1, 0), // BL, patched by the linker
|
||||
}
|
||||
bPC := blockStart + 8
|
||||
ws = append(ws, a64Branch(0, int32(-bPC>>2))) // B back to the entry
|
||||
reloc := Reloc{
|
||||
Off: blockStart + 4,
|
||||
After: blockStart + 8,
|
||||
Name: "runtime\u00b7morestack_noctxt",
|
||||
Kind: RelArm64Branch,
|
||||
}
|
||||
return a64WordsLE(ws...), reloc
|
||||
}
|
||||
|
||||
+11
-8
@@ -27,6 +27,8 @@ func parseArm64File(t *testing.T, src string) *Image {
|
||||
// TestArm64RelocOffsetsIncludePrologue pins the function-relative relocation
|
||||
// offsets of a framed function: the offsets used to exclude the prologue, so
|
||||
// every relocation landed on a prologue instruction in the GOOBJ/ELF output.
|
||||
// The function calls an external, so it is a non-leaf and carries the
|
||||
// stack-split guard (12 bytes, small class) before the prologue.
|
||||
func TestArm64RelocOffsetsIncludePrologue(t *testing.T) {
|
||||
img := parseArm64File(t, "TEXT \u00b7f(SB), $16-0\n"+
|
||||
"\tBL ext\u00b7foo(SB)\n"+
|
||||
@@ -36,8 +38,8 @@ func TestArm64RelocOffsetsIncludePrologue(t *testing.T) {
|
||||
"GLOBL gdata(SB), $8\n")
|
||||
fn := img.Funcs[0]
|
||||
|
||||
// Layout: 12-byte prologue, BL (12), ADRP+ADD (16, 20), ADRP+ADD (24, 28),
|
||||
// 12-byte epilogue with RET.
|
||||
// Layout: 12-byte guard, 12-byte prologue, BL (24), ADRP+ADD (28, 32),
|
||||
// ADRP+ADD (36, 40), 12-byte epilogue with RET, 12-byte morestack block.
|
||||
want := []struct {
|
||||
off int
|
||||
after int
|
||||
@@ -45,11 +47,12 @@ func TestArm64RelocOffsetsIncludePrologue(t *testing.T) {
|
||||
kind RelocKind
|
||||
external bool
|
||||
}{
|
||||
{12, 16, "foo", RelArm64Branch, true},
|
||||
{16, 16, "gdata", RelArm64Addr, false},
|
||||
{20, 20, "gdata", RelArm64Addr, false},
|
||||
{24, 24, "extsym", RelArm64Addr, true},
|
||||
{28, 28, "extsym", RelArm64Addr, true},
|
||||
{24, 28, "foo", RelArm64Branch, true},
|
||||
{28, 28, "gdata", RelArm64Addr, false},
|
||||
{32, 32, "gdata", RelArm64Addr, false},
|
||||
{36, 36, "extsym", RelArm64Addr, true},
|
||||
{40, 40, "extsym", RelArm64Addr, true},
|
||||
{60, 64, "runtime\u00b7morestack_noctxt", RelArm64Branch, true},
|
||||
}
|
||||
if len(fn.Relocs) != len(want) {
|
||||
t.Fatalf("relocs = %d, want %d", len(fn.Relocs), len(want))
|
||||
@@ -64,7 +67,7 @@ func TestArm64RelocOffsetsIncludePrologue(t *testing.T) {
|
||||
|
||||
// The BL with a zero offset sits exactly at the first reloc site.
|
||||
code := img.Code[fn.Offset : fn.Offset+fn.Size]
|
||||
if w := binary.LittleEndian.Uint32(code[12:16]); w != 0x94000000 {
|
||||
if w := binary.LittleEndian.Uint32(code[24:28]); w != 0x94000000 {
|
||||
t.Errorf("BL word = %08x, want 94000000", w)
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -393,7 +393,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||
continue
|
||||
}
|
||||
if r.External && r.Kind == RelCall && r.Name == goobjBuiltinMorestack {
|
||||
if r.External && r.Name == goobjBuiltinMorestack {
|
||||
// The stack-guard morestack call uses the toolchain's
|
||||
// builtin reference.
|
||||
var rec [23]byte
|
||||
|
||||
@@ -100,3 +100,46 @@ func TestStackGuardGOObj(t *testing.T) {
|
||||
t.Fatal("object lacks the GOOBJ magic")
|
||||
}
|
||||
}
|
||||
|
||||
// The arm64 stack-split guard, pinned from `go tool asm` (Go 1.27, arm64):
|
||||
// the guard classes, the auto-NOSPLIT leaf behaviour and the morestack
|
||||
// block. Relocation fields are masked: the toolchain's object leaves them
|
||||
// zero for the linker, the gasm image resolves file-internal references.
|
||||
func TestStackGuardBytesARM64(t *testing.T) {
|
||||
for _, tt := range []struct {
|
||||
name string
|
||||
src string
|
||||
want string
|
||||
}{
|
||||
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
|
||||
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
|
||||
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
|
||||
"900b40f9f14302d13f0210eb09010054f44304d19dfa3fa99f020091fd2300d1fd230491ff430491c0035fd6e3031eaa00000000f3ffff17"},
|
||||
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
||||
"900b40f91bf283d2f1031beba30100543f0210eb690100541b0284d2f4031bcb9dfa3fa99f020091fd2300d11b0184d2fd031b8b1b0284d2ff031b8bc0035fd6e3031eaa00000000eeffff17"},
|
||||
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
||||
"900b40f9ff6330eb09010054fe0f1ef8fd831ff8fd2300d100000000fd835ff8fe0742f8c0035fd6e3031eaa00000000f4ffff17"},
|
||||
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
||||
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
|
||||
} {
|
||||
f, errs := parser.Parse("g_arm64.s", tt.src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("%s: parse: %v", tt.name, errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: assemble: %v", tt.name, err)
|
||||
}
|
||||
fn := img.Funcs[0]
|
||||
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
|
||||
for _, r := range fn.Relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||
code[j] = 0
|
||||
}
|
||||
}
|
||||
got := hex.EncodeToString(code)
|
||||
if got != tt.want {
|
||||
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+3
-3
@@ -41,9 +41,9 @@ func unifiedDiff(name string, a, b []string) string {
|
||||
// Walk the LCS once, assigning every op its absolute position in both
|
||||
// files (1-based, the position an insertion sits before).
|
||||
type op struct {
|
||||
kind byte // ' ', '-' or '+'
|
||||
aLine, bLine int
|
||||
text string
|
||||
kind byte // ' ', '-' or '+'
|
||||
aLine, bLine int
|
||||
text string
|
||||
}
|
||||
var ops []op
|
||||
aPos, bPos := 0, 0
|
||||
|
||||
Reference in New Issue
Block a user