Files
gasm-sdk/asm/riscv_frame.go
T

362 lines
12 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
//
// Go's riscv64 functions have no hardware frame pointer: FP and SP are
// synthetic registers resolved against the hardware stack pointer (X2) and
// the frame size. The return address lives in the link register (X1, RA/LR).
//
// The autosize is the real stack adjustment: the declared local frame plus
// the 8 bytes for the saved link register (the toolchain's FixedFrameSize).
// A leaf function with a zero frame gets no prologue at all.
//
// Prologue (autosize > 0), byte-identical to the toolchain:
//
// MOV LR, -autosize(SP) // save LR below the new SP (traceback-safe)
// ADDI $-autosize, SP, SP // open the frame
// MOV LR, 0(SP) // save LR again at SP (signal-safety)
//
// Epilogue (autosize > 0): MOV 0(SP), LR; ADDI $autosize, SP, SP; the RET's
// uncompressed JALR X0, 0(X1) follows. The toolchain restores LR on every
// frame, leaf or not.
// riscvFrameInfo holds the frame layout derived from a TEXT directive.
type riscvFrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR)
// Stack-split guard state: the toolchain emits the check for every
// non-NOSPLIT function whose autosize is nonzero (a zero autosize is
// "effectively NOSPLIT"); unlike amd64 and arm64 there is no leaf
// auto-NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
}
// riscvComputeFrame derives the frame layout for a TEXT function.
func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
frame := frameSize(t)
if frame != 0 || !riscvIsLeaf(t) {
// FixedFrameSize = 8: space for the saved link register. A
// zero-frame non-leaf function still opens an 8-byte frame for LR.
autosize := frame + 8
fi := riscvFrameInfo{autosize: autosize}
if !hasNoSplitFlag(t) {
fi.needSplit = true
switch {
case autosize <= stackSmall:
fi.splitClass = 0
case autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
return riscvFrameInfo{}
}
// hasNoSplitFlag reports whether the TEXT directive carries NOSPLIT.
func hasNoSplitFlag(t *ast.Text) bool {
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
return true
}
}
return false
}
// riscvIsLeaf reports whether a function contains no call instructions.
// CALL always links; JAL/JALR link only when their destination register is
// the link register (X1), matching cmd/internal/obj/riscv's containsCall.
func riscvIsLeaf(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "CALL":
return false
case "JAL":
// JAL rd, target, a call only when rd is the link register.
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
return false
}
case "JALR":
// JALR rd, offset(rs1) links when the destination register (the
// first operand) is X1; JALR rs1, rd links when the second
// register is X1; JALR offset(rs1) always links to X1.
if len(in.Operands) == 1 {
return false
}
if isMemOperand(in.Operands[1]) {
if regFromOperand(in.Operands[0]) == 1 {
return false
}
continue
}
if regFromOperand(in.Operands[1]) == 1 {
return false
}
}
}
return true
}
// riscvPrologue returns the prologue bytes for a RISC-V function, matching
// the toolchain's compression: the SP adjustment compresses to C.ADDI when
// the immediate fits, and the second LR save compresses to C.SDSP.
func riscvPrologue(fi riscvFrameInfo) []byte {
if fi.autosize == 0 {
return nil
}
var out []byte
// MOV LR, -autosize(SP), SD X1, -autosize(X2). The negative offset is
// not compressible to C.SDSP (unsigned), so it stays 4 bytes. Beyond
// the imm12 range the toolchain materialises the address in X31.
if fits12(int32(-fi.autosize)) {
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
} else {
out = append(out, riscvAddressInX31(int32(-fi.autosize))...)
lo := int32(-fi.autosize) - (splitHi(int32(-fi.autosize)) << 12)
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 1, lo))...)
}
// ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits; X31
// materialisation beyond imm12).
if fits12(int32(-fi.autosize)) {
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
} else {
out = append(out, riscvAddToSP(int32(-fi.autosize))...)
}
// MOV LR, 0(SP), SD X1, 0(X2) → C.SDSP X1, 0.
c := rvcSSP(0x7, 1, 0)
out = append(out, byte(c), byte(c>>8))
return out
}
func fits12(v int32) bool { return v >= -2048 && v <= 2047 }
// splitHi returns the LUI half of the hi/lo split of v (what remains is the
// sign-extended 12-bit low part).
func splitHi(v int32) int32 {
_, high := splitRISCV32Imm(v)
return high
}
// riscvAddressInX31 materialises hi(v) into X31 against the stack pointer,
// matching the toolchain's large-frame addressing: C.LUI (or LUI) X31, hi;
// C.ADD (or ADD) X31, SP.
func riscvAddressInX31(v int32) []byte {
return riscvAddressInX31WithBase(v, 2)
}
// riscvAddressInX31WithBase materialises hi(v) into X31 against an arbitrary
// base register: LUI (or C.LUI) X31, hi; C.ADD X31, rs1. The CR rs2 field
// carries the full 5-bit register, so the compressed form is always
// available.
func riscvAddressInX31WithBase(v int32, rs1 int) []byte {
hi := splitHi(v)
var out []byte
if hi >= -32 && hi <= 31 {
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
}
c := rvcCR(0x9, 31, uint32(rs1))
return append(out, byte(c), byte(c>>8))
}
// riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry:
// C.LUI X31, hi; C.ADDIW X31, lo; C.ADD SP, X31 (the toolchain's form).
func riscvAddToSP(v int32) []byte {
hi := splitHi(v)
lo := v - (hi << 12)
var out []byte
if hi >= -32 && hi <= 31 {
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
}
if lo >= -32 && lo <= 31 {
c := rvcCI(0x1, 31, uint32(lo)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1b, 0x0, 0x00}, 31, 31, lo))...)
}
c := rvcCR(0x9, 2, 31)
return append(out, byte(c), byte(c>>8))
}
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and
// deallocate the frame when present) followed by the uncompressed JALR X0,
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
func riscvReturn(fi riscvFrameInfo) []byte {
var out []byte
if fi.autosize != 0 {
// MOV 0(SP), LR, LD X1, 0(X2) → C.LDSP X1, 0.
c := rvcLSP(0x3, 1, 0)
out = append(out, byte(c), byte(c>>8))
// ADDI $autosize, SP, SP, close the frame (C.ADDI when it fits).
if fits12(int32(fi.autosize)) {
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
} else {
out = append(out, riscvAddToSP(int32(fi.autosize))...)
}
}
// JALR X0, 0(X1).
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
}
// riscvSPAdjust emits an ADDI rd, imm, rd for the stack pointer (rd = rs1 =
// X2), compressed to C.ADDI16SP when the immediate is a nonzero 16-byte
// multiple, else C.ADDI when it fits 6-bit signed.
func riscvSPAdjust(imm int32) []byte {
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
c := rvcADDI16SP(2, imm)
return []byte{byte(c), byte(c >> 8)}
}
if riscvFitsCAddi(imm) {
c := rvcCI(0x0, 2, uint32(imm)&0x3F)
return []byte{byte(c), byte(c >> 8)}
}
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 2, 2, imm))
}
// riscvFitsCAddi reports whether imm compresses to C.ADDI (a nonzero 6-bit
// signed immediate).
func riscvFitsCAddi(imm int32) bool {
return imm != 0 && imm >= -32 && imm <= 31
}
// riscvPrologueSpadjPC returns the function-relative byte offset where the
// prologue has finished decrementing SP (the delta becomes autosize).
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 4 + riscvSPAdjustLen(int32(-fi.autosize))
}
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final JALR, the point where SP is restored.
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 2 + riscvSPAdjustLen(int32(fi.autosize))
}
func riscvSPAdjustLen(imm int32) int {
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
return 2
}
if riscvFitsCAddi(imm) {
return 2
}
return 4
}
// riscvResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x+N(SP) → (N + autosize)(SP). Returns base = -1 for an unresolvable
// reference (SB: static data, handled by the relocation path).
func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32) {
if sym == nil {
return -1, 0
}
switch sym.Pseudo {
case "FP":
return 2, int32(sym.Offset) + int32(fi.autosize) + 8
case "SP":
return 2, int32(fi.autosize) + int32(sym.Offset)
case "SB":
return -1, int32(sym.Offset)
}
return -1, 0
}
// riscvGuardLen returns the byte length of the stack-split guard prefix
// including the inline morestack call (zero when the function needs no
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
// between the guard and the body: the guard branches forward over it.
func riscvGuardLen(fi riscvFrameInfo) int {
_, reloc := riscvGuard(fi)
_ = reloc
return len(riscvGuardBytes(fi))
}
// riscvGuard emits the stack-split guard prefix with the inline morestack
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
// relative to the guard itself, which sits at function offset 0.
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
if !fi.needSplit {
return nil, Reloc{}
}
// MOV 16(g), X6 (g.stackguard0), g = X27.
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
jalBack := func() []byte {
// JAL X0 back to the function start: it sits right after the JAL X5,
// so its displacement is minus the current offset.
return wordLE(riscvJType(0, int32(-len(out))))
}
var reloc Reloc
switch fi.splitClass {
case 0:
// BLTU X6, SP, done (+8: over the CALL and the JMP back)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
case 1:
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+8)
off := int32(fi.autosize - stackSmall)
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
default:
// MOV $(framesize-StackSmall), X7; BLTU SP, X7, call;
// ADD $-(framesize-StackSmall), SP, X7; BLTU X6, X7, call
off := int32(fi.autosize - stackSmall)
mov := encodeRISCVLoadImm(7, off)
out = append(out, mov...)
addiLen := riscvItypeImmediateSize("ADDI", -off)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
if err != nil {
addi = nil
}
out = append(out, addi...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
}
return out, reloc
}
// riscvGuardBytes emits the guard prefix bytes alone (sizing helper).
func riscvGuardBytes(fi riscvFrameInfo) []byte {
g, _ := riscvGuard(fi)
return g
}