From 2e2c0b82a088972ad7133216281e5fd227de70f4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Mon, 14 Sep 2026 21:08:00 +0200 Subject: [PATCH] feat(asm): emit the riscv64 stack-split guard and fix large-frame addressing --- asm/guard_test.go | 42 ++++++++++ asm/riscv_assemble.go | 25 ++++-- asm/riscv_frame.go | 181 ++++++++++++++++++++++++++++++++++++++++-- 3 files changed, 236 insertions(+), 12 deletions(-) diff --git a/asm/guard_test.go b/asm/guard_test.go index 84fbd9c..e1fd463 100644 --- a/asm/guard_test.go +++ b/asm/guard_test.go @@ -143,3 +143,45 @@ func TestStackGuardBytesARM64(t *testing.T) { } } } + +// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27, +// riscv64): the morestack call sits between the guard and the body, and the +// guard branches forward over it. Relocation fields are masked. +func TestStackGuardBytesRISCV64(t *testing.T) { + for _, tt := range []struct { + name string + src string + want string + }{ + {"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n", + "03b30d0163662300000000006ff05fff233411fe211106e08260610167800000"}, + {"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n", + "03b30d01930381f763667300000000006ff01fff233c11ee130181ef06e082601301811067800000"}, + {"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n", + "03b30d0189639b8383f863697100f97f9b8f8f07b303f10163667300000000006ff01ffef97f8a9f23bc1ffef97fe13f7e9106e08260896fa12f7e9167800000"}, + {"frameless", "TEXT \u00b7frameless(SB), $0-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n", + "03b30d0163662300000000006ff05fff233c11fe611106e0000000008260210167800000"}, + {"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n", + "233411fe211106e08260610167800000"}, + } { + f, errs := parser.Parse("g_riscv64.s", tt.src) + if len(errs) > 0 { + t.Fatalf("%s: parse: %v", tt.name, errs) + } + img, err := AssembleFileRISCV(f) + if err != nil { + t.Fatalf("%s: assemble: %v", tt.name, err) + } + fn := img.Funcs[0] + code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...) + for _, r := range fn.Relocs { + for j := r.Off; j < r.Off+4 && j < len(code); j++ { + code[j] = 0 + } + } + got := hex.EncodeToString(code) + if got != tt.want { + t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want) + } + } +} diff --git a/asm/riscv_assemble.go b/asm/riscv_assemble.go index 99074bc..5ee6fc2 100644 --- a/asm/riscv_assemble.go +++ b/asm/riscv_assemble.go @@ -15,14 +15,16 @@ import ( func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { fi := riscvComputeFrame(t) prologue := riscvPrologue(fi) + guardLen := riscvGuardLen(fi) var relocs []Reloc var spadj []SpadjStep // The prologue raises the SP delta by autosize; the boundary is reported // at the pc just past its ADDI, exactly as the toolchain's pctospadj does. + // The guard prefix shifts its PC. if fi.autosize != 0 { - spadj = append(spadj, SpadjStep{PC: riscvPrologueSpadjPC(fi), Value: fi.autosize}) + spadj = append(spadj, SpadjStep{PC: guardLen + riscvPrologueSpadjPC(fi), Value: fi.autosize}) } // Pass 1: collect instructions and compute label offsets assuming 4 bytes @@ -34,7 +36,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ } var recs []instrRec offsets := map[string]int{} - pos := len(prologue) + pos := guardLen + len(prologue) for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: @@ -66,7 +68,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ // Pass 4: recompute offsets with actual sizes. offsets = map[string]int{} - pos = len(prologue) + pos = guardLen + len(prologue) for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: @@ -82,9 +84,17 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ } // Pass 5: re-encode branches with corrected offsets. Record relocations - // during this final pass (relocation offsets are relative to instruction start). - out := append([]byte(nil), prologue...) - pc = len(prologue) + // during this final pass (relocation offsets are relative to instruction + // start). The guard prefix precedes the prologue; its branches target + // the morestack block at the end of the function, which the previous + // passes have sized. + var out []byte + guardBytes, guardReloc := riscvGuard(fi) + if fi.needSplit { + out = append(out, guardBytes...) + } + out = append(out, prologue...) + pc = guardLen + len(prologue) preCount := len(relocs) var lines []LineEntry for _, r := range recs { @@ -119,6 +129,9 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ pc += len(code) } } + if fi.needSplit { + relocs = append(relocs, guardReloc) + } return out, offsets, relocs, lines, spadj, nil } diff --git a/asm/riscv_frame.go b/asm/riscv_frame.go index e6ade22..1b94e65 100644 --- a/asm/riscv_frame.go +++ b/asm/riscv_frame.go @@ -32,6 +32,13 @@ import ( // riscvFrameInfo holds the frame layout derived from a TEXT directive. type riscvFrameInfo struct { autosize int // the real SP adjustment (locals + saved LR) + + // Stack-split guard state: the toolchain emits the check for every + // non-NOSPLIT function whose autosize is nonzero (a zero autosize is + // "effectively NOSPLIT"); unlike amd64 and arm64 there is no leaf + // auto-NOSPLIT. + needSplit bool + splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig } // riscvComputeFrame derives the frame layout for a TEXT function. @@ -40,11 +47,34 @@ func riscvComputeFrame(t *ast.Text) riscvFrameInfo { if frame != 0 || !riscvIsLeaf(t) { // FixedFrameSize = 8: space for the saved link register. A // zero-frame non-leaf function still opens an 8-byte frame for LR. - return riscvFrameInfo{autosize: frame + 8} + autosize := frame + 8 + fi := riscvFrameInfo{autosize: autosize} + if !hasNoSplitFlag(t) { + fi.needSplit = true + switch { + case autosize <= stackSmall: + fi.splitClass = 0 + case autosize <= stackBig: + fi.splitClass = 1 + default: + fi.splitClass = 2 + } + } + return fi } return riscvFrameInfo{} } +// hasNoSplitFlag reports whether the TEXT directive carries NOSPLIT. +func hasNoSplitFlag(t *ast.Text) bool { + for _, f := range t.Flags { + if strings.EqualFold(f, "NOSPLIT") { + return true + } + } + return false +} + // riscvIsLeaf reports whether a function contains no call instructions. // CALL always links; JAL/JALR link only when their destination register is // the link register (X1), matching cmd/internal/obj/riscv's containsCall. @@ -85,16 +115,80 @@ func riscvPrologue(fi riscvFrameInfo) []byte { } var out []byte // MOV LR, -autosize(SP), SD X1, -autosize(X2). The negative offset is - // not compressible to C.SDSP (unsigned), so it stays 4 bytes. - out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...) - // ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits). - out = append(out, riscvSPAdjust(int32(-fi.autosize))...) + // not compressible to C.SDSP (unsigned), so it stays 4 bytes. Beyond + // the imm12 range the toolchain materialises the address in X31. + if fits12(int32(-fi.autosize)) { + out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...) + } else { + out = append(out, riscvAddressInX31(int32(-fi.autosize))...) + lo := int32(-fi.autosize) - (splitHi(int32(-fi.autosize)) << 12) + out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 1, lo))...) + } + // ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits; X31 + // materialisation beyond imm12). + if fits12(int32(-fi.autosize)) { + out = append(out, riscvSPAdjust(int32(-fi.autosize))...) + } else { + out = append(out, riscvAddToSP(int32(-fi.autosize))...) + } // MOV LR, 0(SP), SD X1, 0(X2) → C.SDSP X1, 0. c := rvcSSP(0x7, 1, 0) out = append(out, byte(c), byte(c>>8)) return out } +func fits12(v int32) bool { return v >= -2048 && v <= 2047 } + +// splitHi returns the LUI half of the hi/lo split of v (what remains is the +// sign-extended 12-bit low part). +func splitHi(v int32) int32 { + _, high := splitRISCV32Imm(v) + return high +} + +// riscvAddressInX31 materialises hi(v) into X31 and leaves the caller to add +// the low part, matching the toolchain's large-frame addressing: C.LUI (or +// LUI) X31, hi; C.ADD (or ADD) X31, SP. +func riscvAddressInX31(v int32) []byte { + hi := splitHi(v) + var out []byte + if hi >= -32 && hi <= 31 { + c := rvcCI(0x3, 31, uint32(hi)&0x3F) + out = append(out, byte(c), byte(c>>8)) + } else { + out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...) + } + if hi >= -32 && hi <= 31 { + c := rvcCR(0x9, 31, 2) + out = append(out, byte(c), byte(c>>8)) + } else { + out = append(out, wordLE(riscvRType(riscvEnc{0x33, 0x0, 0x00}, 31, 2, 31))...) + } + return out +} + +// riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry: +// C.LUI X31, hi; C.ADDIW X31, lo; C.ADD SP, X31 (the toolchain's form). +func riscvAddToSP(v int32) []byte { + hi := splitHi(v) + lo := v - (hi << 12) + var out []byte + if hi >= -32 && hi <= 31 { + c := rvcCI(0x3, 31, uint32(hi)&0x3F) + out = append(out, byte(c), byte(c>>8)) + } else { + out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...) + } + if lo >= -32 && lo <= 31 { + c := rvcCI(0x1, 31, uint32(lo)&0x3F) + out = append(out, byte(c), byte(c>>8)) + } else { + out = append(out, wordLE(riscvIType(riscvEnc{0x1b, 0x0, 0x00}, 31, 31, lo))...) + } + c := rvcCR(0x9, 2, 31) + return append(out, byte(c), byte(c>>8)) +} + // riscvReturn returns the bytes for a RET: the epilogue (restore LR and // deallocate the frame when present) followed by the uncompressed JALR X0, // 0(X1) the toolchain emits for RET (it never compresses RET to C.JR). @@ -105,7 +199,11 @@ func riscvReturn(fi riscvFrameInfo) []byte { c := rvcLSP(0x3, 1, 0) out = append(out, byte(c), byte(c>>8)) // ADDI $autosize, SP, SP, close the frame (C.ADDI when it fits). - out = append(out, riscvSPAdjust(int32(fi.autosize))...) + if fits12(int32(fi.autosize)) { + out = append(out, riscvSPAdjust(int32(fi.autosize))...) + } else { + out = append(out, riscvAddToSP(int32(fi.autosize))...) + } } // JALR X0, 0(X1). return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...) @@ -180,3 +278,74 @@ func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32 } return -1, 0 } + +// riscvGuardLen returns the byte length of the stack-split guard prefix +// including the inline morestack call (zero when the function needs no +// guard). Unlike amd64 and arm64, the toolchain places the morestack call +// between the guard and the body: the guard branches forward over it. +func riscvGuardLen(fi riscvFrameInfo) int { + _, reloc := riscvGuard(fi) + _ = reloc + return len(riscvGuardBytes(fi)) +} + +// riscvGuard emits the stack-split guard prefix with the inline morestack +// call: the branch skips forward over JAL X5 and JAL X0 straight into the +// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are +// relative to the guard itself, which sits at function offset 0. +func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) { + if !fi.needSplit { + return nil, Reloc{} + } + // MOV 16(g), X6 (g.stackguard0), g = X27. + out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16)) + jalBack := func() []byte { + // JAL X0 back to the function start: it sits right after the JAL X5, + // so its displacement is minus the current offset. + return wordLE(riscvJType(0, int32(-len(out)))) + } + var reloc Reloc + switch fi.splitClass { + case 0: + // BLTU X6, SP, done (+8: over the CALL and the JMP back) + out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...) + call := len(out) + reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal} + out = append(out, wordLE(riscvJType(5, 0))...) + out = append(out, jalBack()...) + case 1: + // ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+8) + off := int32(fi.autosize - stackSmall) + out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...) + out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...) + call := len(out) + reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal} + out = append(out, wordLE(riscvJType(5, 0))...) + out = append(out, jalBack()...) + default: + // MOV $(framesize-StackSmall), X7; BLTU SP, X7, call; + // ADD $-(framesize-StackSmall), SP, X7; BLTU X6, X7, call + off := int32(fi.autosize - stackSmall) + mov := encodeRISCVLoadImm(7, off) + out = append(out, mov...) + addiLen := riscvItypeImmediateSize("ADDI", -off) + out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...) + addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off) + if err != nil { + addi = nil + } + out = append(out, addi...) + out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...) + call := len(out) + reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal} + out = append(out, wordLE(riscvJType(5, 0))...) + out = append(out, jalBack()...) + } + return out, reloc +} + +// riscvGuardBytes emits the guard prefix bytes alone (sizing helper). +func riscvGuardBytes(fi riscvFrameInfo) []byte { + g, _ := riscvGuard(fi) + return g +}