feat(asm): emit the loong64 stack-split guard for big frames
This commit is contained in:
+141
-36
@@ -20,14 +20,20 @@ import (
|
||||
// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`).
|
||||
// A leaf function (no calls) with a zero frame gets no prologue at all.
|
||||
//
|
||||
// Prologue (autosize > 0), byte-identical to the toolchain:
|
||||
// Prologue (autosize > 0, small), byte-identical to the toolchain:
|
||||
//
|
||||
// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe)
|
||||
// ADDV $-autosize, R3 // open the frame
|
||||
// MOVV R1, 0(R3) // save LR again at SP (signal-safety)
|
||||
//
|
||||
// Large frames (autosize past the 12-bit offset or immediate ranges) expand
|
||||
// the store and the adjust through REGTMP (R30) exactly as the toolchain's
|
||||
// assembler does: the store via the rounding LU12IW split, the adjust via
|
||||
// the floor LU12IW/ORI split.
|
||||
//
|
||||
// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR
|
||||
// restore); the RET's jirl r0, r1, 0 follows.
|
||||
// restore; the adjust materialised when the immediate does not fit); the
|
||||
// RET's jirl r0, r1, 0 follows.
|
||||
|
||||
// loong64FrameInfo holds the frame layout derived from a TEXT directive.
|
||||
type loong64FrameInfo struct {
|
||||
@@ -84,62 +90,108 @@ func loong64ComputeFrame(t *ast.Text) loong64FrameInfo {
|
||||
|
||||
// loong64GuardLen returns the byte length of the stack-split guard prefix
|
||||
// (zero when the function needs no guard). The big class materialises two
|
||||
// constants through R30.
|
||||
// constants through R30; each materialisation shrinks by one word when the
|
||||
// constant's low 12 bits are zero.
|
||||
func loong64GuardLen(fi loong64FrameInfo) int {
|
||||
if !fi.needSplit {
|
||||
return 0
|
||||
}
|
||||
off := int64(fi.autosize - stackSmall)
|
||||
switch fi.splitClass {
|
||||
case 0:
|
||||
return 12
|
||||
case 1:
|
||||
return 16
|
||||
if off <= 2048 {
|
||||
return 16 // ADDV $-off fits the signed 12-bit immediate
|
||||
}
|
||||
return 24 // MOVV + LU12IW + ORI + ADDV + SGTU + BEQ
|
||||
default:
|
||||
return 40 // MOVV + [LU12IW+ORI] + SGTU + BNE + [LU12IW+ORI] + ADDV + SGTU + BEQ
|
||||
// MOVV + [mat] + SGTU + BNE + [mat] + ADDV + SGTU + BEQ
|
||||
return (6 + loong64MatLen(off) + loong64MatLen(-off)) * 4
|
||||
}
|
||||
}
|
||||
|
||||
// loong64Lu12iOri materialises the 32-bit constant v in rd with the
|
||||
// toolchain's LU12IW/ORI pair (the ORI reads and writes rd itself).
|
||||
func loong64Lu12iOri(rd int, v int64) []uint32 {
|
||||
hi := int32(v >> 12)
|
||||
lo := int32(v & 0xFFF)
|
||||
return []uint32{
|
||||
0x0a<<25 | uint32(hi&0xFFFFF)<<5 | uint32(rd),
|
||||
0x0e<<22 | uint32(lo)<<10 | uint32(rd)<<5 | uint32(rd),
|
||||
// loong64MatLen reports the word count of materialising v in R30: a value
|
||||
// with a zero high part needs only the ORI (the toolchain's MOVW $v, R30),
|
||||
// one with a zero low part only the LU12IW.
|
||||
func loong64MatLen(v int64) int {
|
||||
if v>>12 == 0 || v&0xFFF == 0 {
|
||||
return 1
|
||||
}
|
||||
return 2
|
||||
}
|
||||
|
||||
// loong64MatWords appends the words that materialise v in R30, splitting it
|
||||
// as v>>12 plus the zero-extended low 12 bits.
|
||||
func loong64MatWords(ws []uint32, v int64) []uint32 {
|
||||
hi := v >> 12
|
||||
lo := v & 0xFFF
|
||||
if hi == 0 {
|
||||
return append(ws, l64irr(l64OriOp, int(v), 0, 30))
|
||||
}
|
||||
ws = append(ws, l64ir(l64Lu12iwOp, int(hi), 30))
|
||||
if lo != 0 {
|
||||
ws = append(ws, l64irr(l64OriOp, int(lo), 30, 30))
|
||||
}
|
||||
return ws
|
||||
}
|
||||
|
||||
// The LU12IW and ORI opcode bases (2RI20 and 2RI12 formats); the ORI reads
|
||||
// and writes rd itself.
|
||||
const (
|
||||
l64Lu12iwOp = 0x0a << 25
|
||||
l64OriOp = 0x0e << 22
|
||||
)
|
||||
|
||||
// loong64Imm12 reports whether v fits a signed 12-bit immediate.
|
||||
func loong64Imm12(v int64) bool { return v >= -2048 && v <= 2047 }
|
||||
|
||||
// loong64GuardBytes emits the stack-split guard prefix. blockStart is the
|
||||
// function-relative address of the morestack call at the end of the function;
|
||||
// branch displacements are in instructions.
|
||||
// branch displacements are in instructions and are computed from each
|
||||
// branch's own position.
|
||||
func loong64GuardBytes(fi loong64FrameInfo, blockStart int) []byte {
|
||||
// MOVV 16(g), R20 (g.stackguard0), g = R22.
|
||||
ws := []uint32{l64irr(l64loadStoreTable["MOVV"].ld, 16, 22, 20)}
|
||||
off := int64(fi.autosize - stackSmall)
|
||||
// beq appends BEQ R20, blockStart from the branch's own position.
|
||||
beq := func() {
|
||||
ws = append(ws, loong64Beqz(20, int32((blockStart-len(ws)*4)>>2)))
|
||||
}
|
||||
switch fi.splitClass {
|
||||
case 0:
|
||||
// SGTU SP, R20, R20; BEQ R20, more
|
||||
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 3, 20, 20))
|
||||
ws = append(ws, loong64Beqz(20, int32((blockStart-8)>>2)))
|
||||
beq()
|
||||
case 1:
|
||||
off := int32(fi.autosize - stackSmall)
|
||||
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, int(-off), 3, 24))
|
||||
ws = append(ws, loong64MediumWords(off)...)
|
||||
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
|
||||
ws = append(ws, loong64Beqz(20, int32((blockStart-12)>>2)))
|
||||
beq()
|
||||
default:
|
||||
off := int64(fi.autosize - stackSmall)
|
||||
movLen := 8 // LU12IW + ORI
|
||||
ws = append(ws, loong64Lu12iOri(30, off)...)
|
||||
// SGTU $off, SP, R24 catches the SP underflow a huge frame would
|
||||
// cause; BNE jumps to morestack in that case.
|
||||
ws = append(ws, loong64MatWords(nil, off)...)
|
||||
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 30, 3, 24))
|
||||
ws = append(ws, loong64Bnez(24, int32((blockStart-(8+movLen))>>2)))
|
||||
ws = append(ws, loong64Lu12iOri(30, -off)...)
|
||||
ws = append(ws, loong64Bnez(24, int32((blockStart-len(ws)*4)>>2)))
|
||||
ws = append(ws, loong64MatWords(nil, -off)...)
|
||||
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
|
||||
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
|
||||
ws = append(ws, loong64Beqz(20, int32((blockStart-loong64GuardLen(fi)+12)>>2)))
|
||||
beq()
|
||||
}
|
||||
return l64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// loong64MediumWords emits the medium-class stack check for offset off: the
|
||||
// ADDV immediate when it fits, otherwise the same sequence with the constant
|
||||
// materialised in R30.
|
||||
func loong64MediumWords(off int64) []uint32 {
|
||||
if off <= 2048 {
|
||||
return []uint32{l64irr(l64DualTable["ADDV"].imm, int(-off), 3, 24)}
|
||||
}
|
||||
ws := loong64MatWords(nil, -off)
|
||||
return append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
|
||||
}
|
||||
|
||||
// loong64Beqz/loong64Bnez build the 21-bit conditional branches against R0
|
||||
// that the toolchain emits for its guard compares.
|
||||
func loong64Beqz(rj int, dispInstr int32) uint32 {
|
||||
@@ -150,12 +202,13 @@ func loong64Bnez(rj int, dispInstr int32) uint32 {
|
||||
return l64ir21(l64branch21Table["BNEZ"], int(dispInstr), rj)
|
||||
}
|
||||
|
||||
// loong64MoreStackBlock emits the trailing block: MOVV R1, R31 (save LR),
|
||||
// BL runtime.morestack_noctxt, B back to the function entry.
|
||||
// loong64MoreStackBlock emits the trailing block: MOVV R1, R31 (save LR, the
|
||||
// toolchain's OR R1, R0, R31 expansion), BL runtime.morestack_noctxt, B back
|
||||
// to the function entry.
|
||||
func loong64MoreStackBlock(blockStart int) ([]byte, Reloc) {
|
||||
ws := []uint32{
|
||||
l64rrr(l64DualTable["ADD"].rrr, 0, 1, 31), // MOVV R1, R31 (ADD R1, R0, R31)
|
||||
l64bbl(l64jumpTable["BL"], 0), // BL, patched by the linker
|
||||
l64rrr(l64DualTable["OR"].rrr, 0, 1, 31), // MOVV R1, R31 (OR R1, R0, R31)
|
||||
l64bbl(l64jumpTable["BL"], 0), // BL, patched by the linker
|
||||
}
|
||||
disp := (-(blockStart + 8)) >> 2
|
||||
ws = append(ws, l64bbl(l64jumpTable["B"], int(disp)))
|
||||
@@ -185,17 +238,37 @@ func loong64IsLeaf(t *ast.Text) bool {
|
||||
return true
|
||||
}
|
||||
|
||||
// loong64Prologue returns the prologue bytes for a loong64 function.
|
||||
// loong64Prologue returns the prologue bytes for a loong64 function. When
|
||||
// the LR store offset leaves the toolchain's 12-bit store range ([-2046,
|
||||
// 2045], BIG_12 = 2046) or the SP adjust immediate its 12-bit immediate
|
||||
// range, each switches to the R30 materialisation the assembler expands it
|
||||
// to: the store uses the rounding %hi/%lo split (LU12IW of (v+2048)>>12,
|
||||
// REGTMP += SP, store at the raw offset), the adjust the floor split
|
||||
// (LU12IW, ORI when the low part is non-zero, REGTMP += SP).
|
||||
func loong64Prologue(fi loong64FrameInfo) []byte {
|
||||
if fi.autosize == 0 {
|
||||
return nil
|
||||
}
|
||||
addiD := l64DualTable["ADDV"].imm
|
||||
return l64WordsLE(
|
||||
l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, 3, 1), // MOVV R1, -autosize(R3)
|
||||
l64irr(addiD, -fi.autosize, 3, 3), // ADDV $-autosize, R3
|
||||
l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1), // MOVV R1, 0(R3)
|
||||
)
|
||||
var ws []uint32
|
||||
storeBase := 3
|
||||
if fi.autosize > 2046 {
|
||||
// The store goes through REGTMP: LU12IW of the rounding split,
|
||||
// REGTMP += SP, then the store at REGTMP with the truncated offset.
|
||||
v := -int64(fi.autosize)
|
||||
ws = append(ws, l64ir(l64Lu12iwOp, int((v+2048)>>12), 30))
|
||||
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 3, 30, 30))
|
||||
storeBase = 30
|
||||
}
|
||||
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, storeBase, 1)) // MOVV R1, -autosize(base)
|
||||
if loong64Imm12(-int64(fi.autosize)) {
|
||||
ws = append(ws, l64irr(addiD, -fi.autosize, 3, 3)) // ADDV $-autosize, R3
|
||||
} else {
|
||||
ws = append(ws, loong64MatWords(nil, -int64(fi.autosize))...)
|
||||
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
|
||||
}
|
||||
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1)) // MOVV R1, 0(R3)
|
||||
return l64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// loong64Return returns the bytes for a RET: the epilogue (restore LR and
|
||||
@@ -207,14 +280,46 @@ func loong64Return(fi loong64FrameInfo) []byte {
|
||||
// MOVV 0(R3), R1, restore the link register.
|
||||
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
|
||||
}
|
||||
// ADDV $autosize, R3, close the frame.
|
||||
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
|
||||
// ADDV $autosize, R3, close the frame (materialised when the
|
||||
// immediate does not fit).
|
||||
if loong64Imm12(int64(fi.autosize)) {
|
||||
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
|
||||
} else {
|
||||
ws = append(ws, loong64MatWords(nil, int64(fi.autosize))...)
|
||||
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
|
||||
}
|
||||
}
|
||||
// jirl r0, r1, 0, return.
|
||||
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
|
||||
return l64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// loong64StoreWords reports the prologue word count of the LR store, and
|
||||
// loong64AdjustWords the word count of an SP adjust of v: the immediate
|
||||
// forms when they fit, otherwise the R30 materialisation sequences.
|
||||
func loong64StoreWords(autosize int) int {
|
||||
if autosize > 2046 {
|
||||
return 3
|
||||
}
|
||||
return 1
|
||||
}
|
||||
|
||||
func loong64AdjustWords(v int64) int {
|
||||
if loong64Imm12(v) {
|
||||
return 1
|
||||
}
|
||||
return loong64MatLen(v) + 1
|
||||
}
|
||||
|
||||
// loong64EpilogueWords reports the epilogue word count the RET expands to.
|
||||
func loong64EpilogueWords(fi loong64FrameInfo) int {
|
||||
n := loong64AdjustWords(int64(fi.autosize))
|
||||
if !fi.leaf {
|
||||
n++
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// loong64ResolvePseudo translates a pseudo-register memory reference into a
|
||||
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
|
||||
// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable
|
||||
|
||||
Reference in New Issue
Block a user