// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" ) // RISC-V frame mapping, matching the Go toolchain's riscv64 backend. // // Go's riscv64 functions have no hardware frame pointer: FP and SP are // synthetic registers resolved against the hardware stack pointer (X2) and // the frame size. The return address lives in the link register (X1, RA/LR). // // The autosize is the real stack adjustment: the declared local frame plus // the 8 bytes for the saved link register (the toolchain's FixedFrameSize). // A leaf function with a zero frame gets no prologue at all. // // Prologue (autosize > 0), byte-identical to the toolchain: // // MOV LR, -autosize(SP) // save LR below the new SP (traceback-safe) // ADDI $-autosize, SP, SP // open the frame // MOV LR, 0(SP) // save LR again at SP (signal-safety) // // Epilogue (autosize > 0): MOV 0(SP), LR; ADDI $autosize, SP, SP; the RET's // uncompressed JALR X0, 0(X1) follows. The toolchain restores LR on every // frame, leaf or not. // riscvFrameInfo holds the frame layout derived from a TEXT directive. type riscvFrameInfo struct { autosize int // the real SP adjustment (locals + saved LR) // Stack-split guard state: the toolchain emits the check for every // non-NOSPLIT function whose autosize is nonzero (a zero autosize is // "effectively NOSPLIT"); unlike amd64 and arm64 there is no leaf // auto-NOSPLIT. needSplit bool splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig } // riscvComputeFrame derives the frame layout for a TEXT function. func riscvComputeFrame(t *ast.Text) riscvFrameInfo { frame := frameSize(t) if frame != 0 || !riscvIsLeaf(t) { // FixedFrameSize = 8: space for the saved link register. A // zero-frame non-leaf function still opens an 8-byte frame for LR. autosize := frame + 8 fi := riscvFrameInfo{autosize: autosize} if !hasNoSplitFlag(t) { fi.needSplit = true switch { case autosize <= stackSmall: fi.splitClass = 0 case autosize <= stackBig: fi.splitClass = 1 default: fi.splitClass = 2 } } return fi } return riscvFrameInfo{} } // hasNoSplitFlag reports whether the TEXT directive carries NOSPLIT. func hasNoSplitFlag(t *ast.Text) bool { for _, f := range t.Flags { if strings.EqualFold(f, "NOSPLIT") { return true } } return false } // riscvIsLeaf reports whether a function contains no call instructions. // CALL always links; JAL/JALR link only when their destination register is // the link register (X1), matching cmd/internal/obj/riscv's containsCall. func riscvIsLeaf(t *ast.Text) bool { for _, stmt := range t.Body { in, ok := stmt.(*ast.Instr) if !ok { continue } switch strings.ToUpper(in.Mnemonic.Text) { case "CALL": return false case "JAL": // JAL rd, target, a call only when rd is the link register. if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 { return false } case "JALR": // JALR rs1, rd, a call when rd is X1; JALR offset(rs1) always // links to X1. if len(in.Operands) == 1 { return false } if len(in.Operands) >= 2 && regFromOperand(in.Operands[1]) == 1 { return false } } } return true } // riscvPrologue returns the prologue bytes for a RISC-V function, matching // the toolchain's compression: the SP adjustment compresses to C.ADDI when // the immediate fits, and the second LR save compresses to C.SDSP. func riscvPrologue(fi riscvFrameInfo) []byte { if fi.autosize == 0 { return nil } var out []byte // MOV LR, -autosize(SP), SD X1, -autosize(X2). The negative offset is // not compressible to C.SDSP (unsigned), so it stays 4 bytes. Beyond // the imm12 range the toolchain materialises the address in X31. if fits12(int32(-fi.autosize)) { out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...) } else { out = append(out, riscvAddressInX31(int32(-fi.autosize))...) lo := int32(-fi.autosize) - (splitHi(int32(-fi.autosize)) << 12) out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 1, lo))...) } // ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits; X31 // materialisation beyond imm12). if fits12(int32(-fi.autosize)) { out = append(out, riscvSPAdjust(int32(-fi.autosize))...) } else { out = append(out, riscvAddToSP(int32(-fi.autosize))...) } // MOV LR, 0(SP), SD X1, 0(X2) → C.SDSP X1, 0. c := rvcSSP(0x7, 1, 0) out = append(out, byte(c), byte(c>>8)) return out } func fits12(v int32) bool { return v >= -2048 && v <= 2047 } // splitHi returns the LUI half of the hi/lo split of v (what remains is the // sign-extended 12-bit low part). func splitHi(v int32) int32 { _, high := splitRISCV32Imm(v) return high } // riscvAddressInX31 materialises hi(v) into X31 against the stack pointer, // matching the toolchain's large-frame addressing: C.LUI (or LUI) X31, hi; // C.ADD (or ADD) X31, SP. func riscvAddressInX31(v int32) []byte { return riscvAddressInX31WithBase(v, 2) } // riscvAddressInX31WithBase materialises hi(v) into X31 against an arbitrary // base register: LUI (or C.LUI) X31, hi; C.ADD X31, rs1. The CR rs2 field // carries the full 5-bit register, so the compressed form is always // available. func riscvAddressInX31WithBase(v int32, rs1 int) []byte { hi := splitHi(v) var out []byte if hi >= -32 && hi <= 31 { c := rvcCI(0x3, 31, uint32(hi)&0x3F) out = append(out, byte(c), byte(c>>8)) } else { out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...) } c := rvcCR(0x9, 31, uint32(rs1)) return append(out, byte(c), byte(c>>8)) } // riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry: // C.LUI X31, hi; C.ADDIW X31, lo; C.ADD SP, X31 (the toolchain's form). func riscvAddToSP(v int32) []byte { hi := splitHi(v) lo := v - (hi << 12) var out []byte if hi >= -32 && hi <= 31 { c := rvcCI(0x3, 31, uint32(hi)&0x3F) out = append(out, byte(c), byte(c>>8)) } else { out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...) } if lo >= -32 && lo <= 31 { c := rvcCI(0x1, 31, uint32(lo)&0x3F) out = append(out, byte(c), byte(c>>8)) } else { out = append(out, wordLE(riscvIType(riscvEnc{0x1b, 0x0, 0x00}, 31, 31, lo))...) } c := rvcCR(0x9, 2, 31) return append(out, byte(c), byte(c>>8)) } // riscvReturn returns the bytes for a RET: the epilogue (restore LR and // deallocate the frame when present) followed by the uncompressed JALR X0, // 0(X1) the toolchain emits for RET (it never compresses RET to C.JR). func riscvReturn(fi riscvFrameInfo) []byte { var out []byte if fi.autosize != 0 { // MOV 0(SP), LR, LD X1, 0(X2) → C.LDSP X1, 0. c := rvcLSP(0x3, 1, 0) out = append(out, byte(c), byte(c>>8)) // ADDI $autosize, SP, SP, close the frame (C.ADDI when it fits). if fits12(int32(fi.autosize)) { out = append(out, riscvSPAdjust(int32(fi.autosize))...) } else { out = append(out, riscvAddToSP(int32(fi.autosize))...) } } // JALR X0, 0(X1). return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...) } // riscvSPAdjust emits an ADDI rd, imm, rd for the stack pointer (rd = rs1 = // X2), compressed to C.ADDI16SP when the immediate is a nonzero 16-byte // multiple, else C.ADDI when it fits 6-bit signed. func riscvSPAdjust(imm int32) []byte { if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 { c := rvcADDI16SP(2, imm) return []byte{byte(c), byte(c >> 8)} } if riscvFitsCAddi(imm) { c := rvcCI(0x0, 2, uint32(imm)&0x3F) return []byte{byte(c), byte(c >> 8)} } return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 2, 2, imm)) } // riscvFitsCAddi reports whether imm compresses to C.ADDI (a nonzero 6-bit // signed immediate). func riscvFitsCAddi(imm int32) bool { return imm != 0 && imm >= -32 && imm <= 31 } // riscvPrologueSpadjPC returns the function-relative byte offset where the // prologue has finished decrementing SP (the delta becomes autosize). func riscvPrologueSpadjPC(fi riscvFrameInfo) int { if fi.autosize == 0 { return 0 } // SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes). return 4 + riscvSPAdjustLen(int32(-fi.autosize)) } // riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to // (but not including) the final JALR, the point where SP is restored. func riscvReturnEpilogueLen(fi riscvFrameInfo) int { if fi.autosize == 0 { return 0 } // C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes). return 2 + riscvSPAdjustLen(int32(fi.autosize)) } func riscvSPAdjustLen(imm int32) int { if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 { return 2 } if riscvFitsCAddi(imm) { return 2 } return 4 } // riscvResolvePseudo translates a pseudo-register memory reference into a // hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP); // x+N(SP) → (N + autosize)(SP). Returns base = -1 for an unresolvable // reference (SB: static data, handled by the relocation path). func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32) { if sym == nil { return -1, 0 } switch sym.Pseudo { case "FP": return 2, int32(sym.Offset) + int32(fi.autosize) + 8 case "SP": return 2, int32(fi.autosize) + int32(sym.Offset) case "SB": return -1, int32(sym.Offset) } return -1, 0 } // riscvGuardLen returns the byte length of the stack-split guard prefix // including the inline morestack call (zero when the function needs no // guard). Unlike amd64 and arm64, the toolchain places the morestack call // between the guard and the body: the guard branches forward over it. func riscvGuardLen(fi riscvFrameInfo) int { _, reloc := riscvGuard(fi) _ = reloc return len(riscvGuardBytes(fi)) } // riscvGuard emits the stack-split guard prefix with the inline morestack // call: the branch skips forward over JAL X5 and JAL X0 straight into the // body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are // relative to the guard itself, which sits at function offset 0. func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) { if !fi.needSplit { return nil, Reloc{} } // MOV 16(g), X6 (g.stackguard0), g = X27. out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16)) jalBack := func() []byte { // JAL X0 back to the function start: it sits right after the JAL X5, // so its displacement is minus the current offset. return wordLE(riscvJType(0, int32(-len(out)))) } var reloc Reloc switch fi.splitClass { case 0: // BLTU X6, SP, done (+8: over the CALL and the JMP back) out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...) call := len(out) reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal} out = append(out, wordLE(riscvJType(5, 0))...) out = append(out, jalBack()...) case 1: // ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+8) off := int32(fi.autosize - stackSmall) out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...) out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...) call := len(out) reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal} out = append(out, wordLE(riscvJType(5, 0))...) out = append(out, jalBack()...) default: // MOV $(framesize-StackSmall), X7; BLTU SP, X7, call; // ADD $-(framesize-StackSmall), SP, X7; BLTU X6, X7, call off := int32(fi.autosize - stackSmall) mov := encodeRISCVLoadImm(7, off) out = append(out, mov...) addiLen := riscvItypeImmediateSize("ADDI", -off) out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...) addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off) if err != nil { addi = nil } out = append(out, addi...) out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...) call := len(out) reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal} out = append(out, wordLE(riscvJType(5, 0))...) out = append(out, jalBack()...) } return out, reloc } // riscvGuardBytes emits the guard prefix bytes alone (sizing helper). func riscvGuardBytes(fi riscvFrameInfo) []byte { g, _ := riscvGuard(fi) return g }