feat(asm): emit the riscv64 stack-split guard and fix large-frame addressing
This commit is contained in:
+175
-6
@@ -32,6 +32,13 @@ import (
|
||||
// riscvFrameInfo holds the frame layout derived from a TEXT directive.
|
||||
type riscvFrameInfo struct {
|
||||
autosize int // the real SP adjustment (locals + saved LR)
|
||||
|
||||
// Stack-split guard state: the toolchain emits the check for every
|
||||
// non-NOSPLIT function whose autosize is nonzero (a zero autosize is
|
||||
// "effectively NOSPLIT"); unlike amd64 and arm64 there is no leaf
|
||||
// auto-NOSPLIT.
|
||||
needSplit bool
|
||||
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
|
||||
}
|
||||
|
||||
// riscvComputeFrame derives the frame layout for a TEXT function.
|
||||
@@ -40,11 +47,34 @@ func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
|
||||
if frame != 0 || !riscvIsLeaf(t) {
|
||||
// FixedFrameSize = 8: space for the saved link register. A
|
||||
// zero-frame non-leaf function still opens an 8-byte frame for LR.
|
||||
return riscvFrameInfo{autosize: frame + 8}
|
||||
autosize := frame + 8
|
||||
fi := riscvFrameInfo{autosize: autosize}
|
||||
if !hasNoSplitFlag(t) {
|
||||
fi.needSplit = true
|
||||
switch {
|
||||
case autosize <= stackSmall:
|
||||
fi.splitClass = 0
|
||||
case autosize <= stackBig:
|
||||
fi.splitClass = 1
|
||||
default:
|
||||
fi.splitClass = 2
|
||||
}
|
||||
}
|
||||
return fi
|
||||
}
|
||||
return riscvFrameInfo{}
|
||||
}
|
||||
|
||||
// hasNoSplitFlag reports whether the TEXT directive carries NOSPLIT.
|
||||
func hasNoSplitFlag(t *ast.Text) bool {
|
||||
for _, f := range t.Flags {
|
||||
if strings.EqualFold(f, "NOSPLIT") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// riscvIsLeaf reports whether a function contains no call instructions.
|
||||
// CALL always links; JAL/JALR link only when their destination register is
|
||||
// the link register (X1), matching cmd/internal/obj/riscv's containsCall.
|
||||
@@ -85,16 +115,80 @@ func riscvPrologue(fi riscvFrameInfo) []byte {
|
||||
}
|
||||
var out []byte
|
||||
// MOV LR, -autosize(SP), SD X1, -autosize(X2). The negative offset is
|
||||
// not compressible to C.SDSP (unsigned), so it stays 4 bytes.
|
||||
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
|
||||
// ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits).
|
||||
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
|
||||
// not compressible to C.SDSP (unsigned), so it stays 4 bytes. Beyond
|
||||
// the imm12 range the toolchain materialises the address in X31.
|
||||
if fits12(int32(-fi.autosize)) {
|
||||
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
|
||||
} else {
|
||||
out = append(out, riscvAddressInX31(int32(-fi.autosize))...)
|
||||
lo := int32(-fi.autosize) - (splitHi(int32(-fi.autosize)) << 12)
|
||||
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 1, lo))...)
|
||||
}
|
||||
// ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits; X31
|
||||
// materialisation beyond imm12).
|
||||
if fits12(int32(-fi.autosize)) {
|
||||
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
|
||||
} else {
|
||||
out = append(out, riscvAddToSP(int32(-fi.autosize))...)
|
||||
}
|
||||
// MOV LR, 0(SP), SD X1, 0(X2) → C.SDSP X1, 0.
|
||||
c := rvcSSP(0x7, 1, 0)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
return out
|
||||
}
|
||||
|
||||
func fits12(v int32) bool { return v >= -2048 && v <= 2047 }
|
||||
|
||||
// splitHi returns the LUI half of the hi/lo split of v (what remains is the
|
||||
// sign-extended 12-bit low part).
|
||||
func splitHi(v int32) int32 {
|
||||
_, high := splitRISCV32Imm(v)
|
||||
return high
|
||||
}
|
||||
|
||||
// riscvAddressInX31 materialises hi(v) into X31 and leaves the caller to add
|
||||
// the low part, matching the toolchain's large-frame addressing: C.LUI (or
|
||||
// LUI) X31, hi; C.ADD (or ADD) X31, SP.
|
||||
func riscvAddressInX31(v int32) []byte {
|
||||
hi := splitHi(v)
|
||||
var out []byte
|
||||
if hi >= -32 && hi <= 31 {
|
||||
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
} else {
|
||||
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
|
||||
}
|
||||
if hi >= -32 && hi <= 31 {
|
||||
c := rvcCR(0x9, 31, 2)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
} else {
|
||||
out = append(out, wordLE(riscvRType(riscvEnc{0x33, 0x0, 0x00}, 31, 2, 31))...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry:
|
||||
// C.LUI X31, hi; C.ADDIW X31, lo; C.ADD SP, X31 (the toolchain's form).
|
||||
func riscvAddToSP(v int32) []byte {
|
||||
hi := splitHi(v)
|
||||
lo := v - (hi << 12)
|
||||
var out []byte
|
||||
if hi >= -32 && hi <= 31 {
|
||||
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
} else {
|
||||
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
|
||||
}
|
||||
if lo >= -32 && lo <= 31 {
|
||||
c := rvcCI(0x1, 31, uint32(lo)&0x3F)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
} else {
|
||||
out = append(out, wordLE(riscvIType(riscvEnc{0x1b, 0x0, 0x00}, 31, 31, lo))...)
|
||||
}
|
||||
c := rvcCR(0x9, 2, 31)
|
||||
return append(out, byte(c), byte(c>>8))
|
||||
}
|
||||
|
||||
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and
|
||||
// deallocate the frame when present) followed by the uncompressed JALR X0,
|
||||
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
|
||||
@@ -105,7 +199,11 @@ func riscvReturn(fi riscvFrameInfo) []byte {
|
||||
c := rvcLSP(0x3, 1, 0)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
// ADDI $autosize, SP, SP, close the frame (C.ADDI when it fits).
|
||||
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
|
||||
if fits12(int32(fi.autosize)) {
|
||||
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
|
||||
} else {
|
||||
out = append(out, riscvAddToSP(int32(fi.autosize))...)
|
||||
}
|
||||
}
|
||||
// JALR X0, 0(X1).
|
||||
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
|
||||
@@ -180,3 +278,74 @@ func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32
|
||||
}
|
||||
return -1, 0
|
||||
}
|
||||
|
||||
// riscvGuardLen returns the byte length of the stack-split guard prefix
|
||||
// including the inline morestack call (zero when the function needs no
|
||||
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
|
||||
// between the guard and the body: the guard branches forward over it.
|
||||
func riscvGuardLen(fi riscvFrameInfo) int {
|
||||
_, reloc := riscvGuard(fi)
|
||||
_ = reloc
|
||||
return len(riscvGuardBytes(fi))
|
||||
}
|
||||
|
||||
// riscvGuard emits the stack-split guard prefix with the inline morestack
|
||||
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
|
||||
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
|
||||
// relative to the guard itself, which sits at function offset 0.
|
||||
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) {
|
||||
if !fi.needSplit {
|
||||
return nil, Reloc{}
|
||||
}
|
||||
// MOV 16(g), X6 (g.stackguard0), g = X27.
|
||||
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
|
||||
jalBack := func() []byte {
|
||||
// JAL X0 back to the function start: it sits right after the JAL X5,
|
||||
// so its displacement is minus the current offset.
|
||||
return wordLE(riscvJType(0, int32(-len(out))))
|
||||
}
|
||||
var reloc Reloc
|
||||
switch fi.splitClass {
|
||||
case 0:
|
||||
// BLTU X6, SP, done (+8: over the CALL and the JMP back)
|
||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
|
||||
call := len(out)
|
||||
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
|
||||
out = append(out, wordLE(riscvJType(5, 0))...)
|
||||
out = append(out, jalBack()...)
|
||||
case 1:
|
||||
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+8)
|
||||
off := int32(fi.autosize - stackSmall)
|
||||
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
|
||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
||||
call := len(out)
|
||||
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
|
||||
out = append(out, wordLE(riscvJType(5, 0))...)
|
||||
out = append(out, jalBack()...)
|
||||
default:
|
||||
// MOV $(framesize-StackSmall), X7; BLTU SP, X7, call;
|
||||
// ADD $-(framesize-StackSmall), SP, X7; BLTU X6, X7, call
|
||||
off := int32(fi.autosize - stackSmall)
|
||||
mov := encodeRISCVLoadImm(7, off)
|
||||
out = append(out, mov...)
|
||||
addiLen := riscvItypeImmediateSize("ADDI", -off)
|
||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
|
||||
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
|
||||
if err != nil {
|
||||
addi = nil
|
||||
}
|
||||
out = append(out, addi...)
|
||||
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
|
||||
call := len(out)
|
||||
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
|
||||
out = append(out, wordLE(riscvJType(5, 0))...)
|
||||
out = append(out, jalBack()...)
|
||||
}
|
||||
return out, reloc
|
||||
}
|
||||
|
||||
// riscvGuardBytes emits the guard prefix bytes alone (sizing helper).
|
||||
func riscvGuardBytes(fi riscvFrameInfo) []byte {
|
||||
g, _ := riscvGuard(fi)
|
||||
return g
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user