From 79a2c16bac6b99ed0c677fa2cc19e4f1474da7c3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sat, 19 Sep 2026 23:49:13 +0200 Subject: [PATCH] fix(riscv64): compressed store offsets, FENCE and branch range checks Assisted-by: GLM 5.3 --- asm/riscv_assemble.go | 154 ++++++++++++++++++++--- asm/riscv_encode.go | 10 +- asm/riscv_encode_test.go | 189 +++++++++++++++++++++++++++++ asm/riscv_frame.go | 65 +++++----- asm/riscv_frame_test.go | 41 +++++++ testdata/verify/rvcstore_riscv64.s | 48 ++++++++ 6 files changed, 453 insertions(+), 54 deletions(-) create mode 100644 testdata/verify/rvcstore_riscv64.s diff --git a/asm/riscv_assemble.go b/asm/riscv_assemble.go index 9197658..4a3fb17 100644 --- a/asm/riscv_assemble.go +++ b/asm/riscv_assemble.go @@ -15,7 +15,10 @@ import ( func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { fi := riscvComputeFrame(t) prologue := riscvPrologue(fi) - guardLen := riscvGuardLen(fi) + guardLen, err := riscvGuardLen(fi) + if err != nil { + return nil, nil, nil, nil, nil, err + } var relocs []Reloc var spadj []SpadjStep @@ -66,20 +69,20 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ } } - // Pass 4: recompute offsets with actual sizes. + // Pass 4: recompute offsets with actual sizes. recs holds the + // instructions in emission order, so an index into it walks t.Body in + // lockstep (the same single pass Pass 1 uses) instead of rescanning the + // whole slice per statement. offsets = map[string]int{} pos = guardLen + len(prologue) + ri := 0 for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: offsets[s.Name.Text] = pos case *ast.Instr: - for _, r := range recs { - if r.instr == s { - pos += len(r.code) - break - } - } + pos += len(recs[ri].code) + ri++ } } @@ -89,7 +92,10 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ // the morestack block at the end of the function, which the previous // passes have sized. var out []byte - guardBytes, guardReloc := riscvGuard(fi) + guardBytes, guardReloc, err := riscvGuard(fi) + if err != nil { + return nil, nil, nil, nil, nil, err + } if fi.needSplit { out = append(out, guardBytes...) } @@ -231,6 +237,25 @@ func isBranchLike(mnem string) bool { return false } +// riscvCheckBranchOffset rejects a B-type displacement outside its signed +// 13-bit span [-4096, 4094]; the encoder masks to 13 bits, so an +// out-of-range offset would otherwise wrap to a wrong target. +func riscvCheckBranchOffset(target string, off int32) error { + if off < -4096 || off > 4094 { + return fmt.Errorf("branch to %q too far (13-bit range)", target) + } + return nil +} + +// riscvCheckJumpOffset rejects a J-type displacement outside its signed +// 21-bit span [-1048576, 1048574]. +func riscvCheckJumpOffset(target string, off int32) error { + if off < -1048576 || off > 1048574 { + return fmt.Errorf("jump to %q too far (21-bit range)", target) + } + return nil +} + // encodeRISCVInstr encodes a single RISC-V instruction. func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) { mnem := instr.Mnemonic.Text @@ -304,6 +329,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } offset := int32(targetOff - pc) + if err := riscvCheckJumpOffset(target, offset); err != nil { + return nil, err + } word = riscvJType(0, offset) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil case "JAL": @@ -320,6 +348,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } offset := int32(targetOff - pc) + if err := riscvCheckJumpOffset(target, offset); err != nil { + return nil, err + } word = riscvJType(rd, offset) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil @@ -365,6 +396,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv case "BGTZ": enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, 0, rs // blt x0, rs } + if err := riscvCheckBranchOffset(target, int32(targetOff-pc)); err != nil { + return nil, err + } word = riscvBType(enc, rs1, rs2, int32(targetOff-pc)) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil @@ -374,7 +408,14 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv if !ok { return nil, fmt.Errorf("unsupported system instruction %q", mnem) } - word = riscvIType(enc, 0, 0, 0) + // The bare FENCE expands to fence iorw, iorw: the predecessor and + // successor fields both carry 0xF in the I-type immediate + // (the toolchain's encodeFenceOperand TYPE_NONE default). + imm := int32(0) + if mnem == "FENCE" { + imm = 0x0FF + } + word = riscvIType(enc, 0, 0, imm) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } @@ -416,7 +457,10 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } csr := immFromOperand(ops[0]) // CSR address (12-bit) - rd := regFromOperand(ops[2]) // destination register + if csr < 0 || csr > 0xFFF { + return nil, fmt.Errorf("%s: CSR address %d out of range 0-0xFFF", mnem, csr) + } + rd := regFromOperand(ops[2]) // destination register if rd < 0 { return nil, fmt.Errorf("invalid destination register in %s", mnem) } @@ -561,9 +605,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv // I-type with immediate: Plan 9 order is INSTR $imm, rs1, rd; the // two-operand form INSTR $imm, rd uses rd as the source. case len(ops) == 3 && isITypeInstr(mnem): - imm := immFromOperand(ops[0]) // immediate - if immNeg { - imm = -imm // SUB $imm arrived through the ADDI alias + imm, err := riscvImm32FromOperand(ops[0], immNeg) // immediate + if err != nil { + return nil, err } rs1 := regFromOperand(ops[1]) // source register rd := regFromOperand(ops[2]) // destination @@ -573,9 +617,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv return encodeRISCVItypeImmediate(mnem, enc, rd, rs1, imm) case len(ops) == 2 && isITypeInstr(mnem): - imm := immFromOperand(ops[0]) - if immNeg { - imm = -imm + imm, err := riscvImm32FromOperand(ops[0], immNeg) + if err != nil { + return nil, err } rd := regFromOperand(ops[1]) if rd < 0 { @@ -614,6 +658,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv if rs1 < 0 || rs2 < 0 { return nil, fmt.Errorf("invalid register in %s", mnem) } + if err := riscvCheckBranchOffset(target, offset); err != nil { + return nil, err + } // The Go assembler never compresses branches to C.BEQZ/C.BNEZ. word = riscvBType(enc, rs1, rs2, offset) @@ -695,7 +742,10 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt if rd < 0 { return nil, fmt.Errorf("MOV $imm: invalid destination register") } - imm := immFromOperand(src) + imm, err := riscvImm32FromOperand(src, false) + if err != nil { + return nil, err + } return encodeRISCVLoadImm(rd, imm), nil } @@ -1081,7 +1131,7 @@ func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) { // tryCompressRVC attempts to compress a RISC-V instruction to its 16-bit // RVC form. It returns the compressed instruction word and true on success. func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) { - mnem := instr.Mnemonic.Text + mnem := riscvCompressMnem(instr) ops := instr.Operands switch mnem { @@ -1331,6 +1381,49 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) { return 0, false } +// riscvCompressMnem maps a MOV-family load or store onto the base mnemonic +// the toolchain lowers it to (MOVW 4(SP), X9 is LW under another name), so +// the width spellings compress exactly like their base forms. Register and +// immediate forms keep their own mnemonic: the C.MV path matches "MOV" and +// nothing else in the switch has a width case. +func riscvCompressMnem(instr *ast.Instr) string { + mnem := instr.Mnemonic.Text + ops := instr.Operands + if !strings.HasPrefix(mnem, "MOV") || len(ops) != 2 { + return mnem + } + load := isMemOperand(ops[0]) && !isMemOperand(ops[1]) + store := !isMemOperand(ops[0]) && isMemOperand(ops[1]) + if !load && !store { + return mnem + } + switch mnem { + case "MOVW": + if load { + return "LW" + } + return "SW" + case "MOVF": + if load { + return "FLW" + } + return "FSW" + case "MOVD": + if load { + return "FLD" + } + return "FSD" + case "MOV": + if load { + return "LD" + } + return "SD" + } + // MOVB/MOVBU/MOVH/MOVHU/MOVWU have no compressed form; their base + // mnemonics (LB/LBU/LH/LHU/LWU, SB/SH) match no case either. + return mnem +} + // extractLDParams extracts rd, rs1, and immediate offset for a load instruction. func extractLDParams(instr *ast.Instr, fi riscvFrameInfo) (rd, rs1 int, imm int32) { ops := instr.Operands @@ -1507,6 +1600,29 @@ func immFromOperand(op *ast.Operand) int32 { return 0 } +// riscvImm32FromOperand reads an immediate for the MOV/I-type paths as a +// signed 32-bit value. The toolchain materialises wider constants through +// its SLLI expansion, which this assembler does not implement, so values +// outside the int32 span are diagnosed instead of silently truncated (MOV +// $0x123456789 must not assemble as $0x3456789). The neg flag carries the +// SUB $imm alias, whose negated value may fit when the written one does not. +func riscvImm32FromOperand(op *ast.Operand, neg bool) (int32, error) { + var v int64 + if op.Imm.HasVal { + v = op.Imm.Val + if op.Imm.Neg { + v = -v + } + } + if neg { + v = -v + } + if int64(int32(v)) != v { + return 0, fmt.Errorf("immediate %d out of range; 64-bit materialisation not supported", v) + } + return int32(v), nil +} + func memFromOperand(op *ast.Operand) (rs1 int, imm int32) { rs1 = riscvRegNum(op.Addr.Base) imm = int32(op.Addr.Offset) diff --git a/asm/riscv_encode.go b/asm/riscv_encode.go index 4ed7c23..abdb408 100644 --- a/asm/riscv_encode.go +++ b/asm/riscv_encode.go @@ -373,7 +373,7 @@ var riscvFmaTable = map[string]riscvFmaEnc{ // riscvFmaType encodes an R4-type fused multiply-add instruction. func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 { return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) | - (uint32(rs1) << 15) | (0x0 << 12) /* rm=dynamic */ | (uint32(rd) << 7) | enc.opcode + (uint32(rs1) << 15) | (0x0 << 12) /* rm=RNE */ | (uint32(rd) << 7) | enc.opcode } // CSR (Control and Status Register) instructions. @@ -522,11 +522,13 @@ func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 { // rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW // (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte -// offset; the immediate bits are extracted per the RISC-V CS format. +// offset; the immediate bits are extracted per the RISC-V CS format, with the +// same five-bit patterns as the load side ({5,4,3,7,6} and {5,4,3,2,6}, +// matching the toolchain's encodeCS). func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 { - pattern := []int{5, 3, 7, 6} + pattern := []int{5, 4, 3, 7, 6} if funct3 == 0x6 { - pattern = []int{5, 3, 2, 6} + pattern = []int{5, 4, 3, 2, 6} } packed := encodeRVCPattern(imm, pattern) return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2)) diff --git a/asm/riscv_encode_test.go b/asm/riscv_encode_test.go index e029d29..0d1b920 100644 --- a/asm/riscv_encode_test.go +++ b/asm/riscv_encode_test.go @@ -5,6 +5,7 @@ package asm import ( "bytes" + "strings" "testing" "sourcedock.dev/petrbalvin/gasm-devkit/ast" @@ -691,6 +692,81 @@ DATA answer<>+0(SB)/8, $42 } } +// TestRISCV_RVC_StorePatterns pins the register-relative compressed store +// encodings for offsets with immediate bits 4 and 5 set, byte-identical to +// the toolchain's encodeCS (patterns {5,4,3,7,6} and {5,4,3,2,6}). +// Regression: the store-side patterns dropped imm[4], so every such store +// silently encoded the wrong address while the loads stayed correct. +func TestRISCV_RVC_StorePatterns(t *testing.T) { + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·csstores(SB), NOSPLIT, $0 + SD X9, 24(X8) + SW X10, 16(X11) + FSD F8, 40(X12) + LD 24(X8), X9 + LW 16(X11), X10 + FLD 40(X12), F8 + RET +`) + code := assembleRISCVHelper(t, fn) + want := []byte{ + 0x04, 0xec, // c.sd x9, 24(x8) + 0x88, 0xc9, // c.sw x10, 16(x11) + 0x00, 0xb6, // c.fsd f8, 40(x12) + 0x04, 0x6c, // c.ld x9, 24(x8) + 0x88, 0x49, // c.lw x10, 16(x11) + 0x00, 0x36, // c.fld f8, 40(x12) + 0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1) + } + if !bytes.Equal(code, want) { + t.Errorf("code = % x\nwant % x", code, want) + } +} + +// TestRISCV_FENCE pins the FENCE encoding: the toolchain expands the bare +// mnemonic to fence iorw, iorw (0x0FF0000F), not fence 0,0. +func TestRISCV_FENCE(t *testing.T) { + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·fence(SB), NOSPLIT, $0 + FENCE + RET +`) + code := assembleRISCVHelper(t, fn) + want := []byte{ + 0x0f, 0x00, 0xf0, 0x0f, // fence iorw, iorw + 0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1) + } + if !bytes.Equal(code, want) { + t.Errorf("code = % x\nwant % x", code, want) + } +} + +// TestRISCV_RVC_WidthSpellings pins the compression of the GOROOT width +// spellings: MOVW and MOVD lower to their base load/store and compress +// exactly like LW/SW/FLD/FSD would (the toolchain compresses these shapes; +// before the normalisation they stayed 4 bytes). +func TestRISCV_RVC_WidthSpellings(t *testing.T) { + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·widths(SB), NOSPLIT, $0-16 + MOVW w+0(FP), X9 + MOVW X9, v+4(FP) + MOVD d+0(FP), F8 + MOVD F8, r+8(FP) + RET +`) + code := assembleRISCVHelper(t, fn) + want := []byte{ + 0xa2, 0x44, // c.lwsp x9, 8 + 0x26, 0xc6, // c.swsp x9, 12 + 0x22, 0x24, // c.fldsp f8, 8 + 0x22, 0xa8, // c.fsdsp f8, 16 + 0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1) + } + if !bytes.Equal(code, want) { + t.Errorf("code = % x\nwant % x", code, want) + } +} + func TestRISCV_system_instrs(t *testing.T) { // Test FENCE, ECALL, EBREAK encoding. fn := firstTextRISCV(t, `#include "textflag.h" @@ -782,3 +858,116 @@ TEXT ·f(SB), NOSPLIT, $0-0 0x00008067, // jalr x0, 1(x0), 0 (RET) ) } + +// encodeOneInstrRISCV encodes a single parsed instruction against a synthetic +// offsets map, the smallest honest harness for the branch-range diagnostics: +// the spans are far larger than any source a test would want to spell out. +func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]int) ([]byte, error) { + t.Helper() + fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src) + instr := fn.Body[0].(*ast.Instr) + return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil) +} + +// TestRISCVBranchJumpRange checks that displacements beyond the B-type span +// [-4096, 4094] and the J-type span [-1048576, 1048574] are diagnosed instead +// of wrapping silently to a wrong target. +func TestRISCVBranchJumpRange(t *testing.T) { + cases := []struct { + name string + src string + off int // the target's function-relative offset (pc 0) + ok bool + }{ + {"branch max", "BEQ X10, X11, tgt\nRET\n", 4094, true}, + {"branch past max", "BEQ X10, X11, tgt\nRET\n", 4096, false}, + {"branch back max", "BEQ X10, X11, tgt\nRET\n", -4096, true}, + {"branch back past max", "BEQ X10, X11, tgt\nRET\n", -4098, false}, + {"branchz past max", "BEQZ X10, tgt\nRET\n", 4096, false}, + {"jump max", "JMP tgt\nRET\n", 1048574, true}, + {"jump past max", "JMP tgt\nRET\n", 1048576, false}, + {"jump back max", "JMP tgt\nRET\n", -1048576, true}, + {"jump back past max", "JMP tgt\nRET\n", -1048578, false}, + {"jal past max", "JAL tgt\nRET\n", 1048576, false}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + _, err := encodeOneInstrRISCV(t, "TEXT ·f(SB), NOSPLIT, $0\n\t"+c.src, 0, map[string]int{"tgt": c.off}) + if c.ok && err != nil { + t.Fatalf("unexpected error: %v", err) + } + if !c.ok && err == nil { + t.Fatal("expected an out-of-range diagnostic, got none") + } + }) + } +} + +// TestRISCVBranchFarBody drives the range check through the full two-pass +// assembler: a forward branch over a body larger than the B-type span must +// error rather than wrap. +func TestRISCVBranchFarBody(t *testing.T) { + var sb strings.Builder + sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n") + for range 1100 { + sb.WriteString("\tADD X10, X11, X12\n") + } + sb.WriteString("done:\n\tRET\n") + fn := firstTextRISCV(t, sb.String()) + if _, _, _, _, _, err := assembleRISCV(fn); err == nil { + t.Error("expected a branch-out-of-range error, got none") + } +} + +// TestRISCV_CSRRange checks the CSR address range: the 12-bit field is +// diagnosed rather than masked, so CSRRW $4096 does not silently address +// CSR 0. +func TestRISCV_CSRRange(t *testing.T) { + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·csrhi(SB), NOSPLIT, $0 + CSRRW $4096, X10, X11 + RET +`) + if _, _, _, _, _, err := assembleRISCV(fn); err == nil { + t.Error("expected an out-of-range error for CSR $4096, got none") + } + fn = firstTextRISCV(t, `#include "textflag.h" +TEXT ·csrmax(SB), NOSPLIT, $0 + CSRRW $4095, X10, X11 + RET +`) + if _, _, _, _, _, err := assembleRISCV(fn); err != nil { + t.Errorf("CSR $4095 must assemble: %v", err) + } +} + +// TestRISCV_Imm64Rejected checks that immediates outside the signed 32-bit +// span are diagnosed instead of silently truncated to their low 32 bits (the +// toolchain materialises such constants via SLLI expansion, which this +// assembler does not implement). +func TestRISCV_Imm64Rejected(t *testing.T) { + cases := []string{ + "MOV $0x123456789, X10", + "ADDI $0x100000000, X10, X11", + "ANDI $-0x800000001, X10, X11", + "SUB $0x100000000, X10, X11", + } + for _, src := range cases { + fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n") + if _, _, _, _, _, err := assembleRISCV(fn); err == nil { + t.Errorf("%s: expected an out-of-range error, got none", src) + } + } + // The full signed 32-bit span still assembles, including the SUB form + // whose negated immediate only just fits. + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·edge(SB), NOSPLIT, $0 + MOV $2147483647, X10 + MOV $-2147483648, X11 + SUB $0x80000000, X12, X13 + RET +`) + if _, _, _, _, _, err := assembleRISCV(fn); err != nil { + t.Errorf("int32-span immediates must assemble: %v", err) + } +} diff --git a/asm/riscv_frame.go b/asm/riscv_frame.go index 10be6dc..aeee94a 100644 --- a/asm/riscv_frame.go +++ b/asm/riscv_frame.go @@ -4,6 +4,7 @@ package asm import ( + "fmt" "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" @@ -241,33 +242,36 @@ func riscvFitsCAddi(imm int32) bool { } // riscvPrologueSpadjPC returns the function-relative byte offset where the -// prologue has finished decrementing SP (the delta becomes autosize). +// prologue has finished decrementing SP (the delta becomes autosize). It is +// computed from the same expansion functions the prologue emits, so the +// large-frame X31 materialisations are counted: C.LUI + C.ADD before the SD, +// C.LUI + ADDIW + C.ADD for the SP adjust. func riscvPrologueSpadjPC(fi riscvFrameInfo) int { if fi.autosize == 0 { return 0 } - // SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes). - return 4 + riscvSPAdjustLen(int32(-fi.autosize)) + adj := int32(-fi.autosize) + if fits12(adj) { + // SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes). + return 4 + len(riscvSPAdjust(adj)) + } + return len(riscvAddressInX31(adj)) + 4 + len(riscvAddToSP(adj)) } // riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to -// (but not including) the final JALR, the point where SP is restored. +// (but not including) the final JALR, the point where SP is restored. The +// small frame closes with C.LDSP + ADDI/C.ADDI; the large frame materialises +// the adjustment through X31 (C.LUI + ADDIW + C.ADD). func riscvReturnEpilogueLen(fi riscvFrameInfo) int { if fi.autosize == 0 { return 0 } - // C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes). - return 2 + riscvSPAdjustLen(int32(fi.autosize)) -} - -func riscvSPAdjustLen(imm int32) int { - if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 { - return 2 + adj := int32(fi.autosize) + if fits12(adj) { + // C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes). + return 2 + len(riscvSPAdjust(adj)) } - if riscvFitsCAddi(imm) { - return 2 - } - return 4 + return 2 + len(riscvAddToSP(adj)) } // riscvResolvePseudo translates a pseudo-register memory reference into a @@ -293,19 +297,21 @@ func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32 // including the inline morestack call (zero when the function needs no // guard). Unlike amd64 and arm64, the toolchain places the morestack call // between the guard and the body: the guard branches forward over it. -func riscvGuardLen(fi riscvFrameInfo) int { - _, reloc := riscvGuard(fi) - _ = reloc - return len(riscvGuardBytes(fi)) +func riscvGuardLen(fi riscvFrameInfo) (int, error) { + g, _, err := riscvGuard(fi) + if err != nil { + return 0, err + } + return len(g), nil } // riscvGuard emits the stack-split guard prefix with the inline morestack // call: the branch skips forward over JAL X5 and JAL X0 straight into the // body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are // relative to the guard itself, which sits at function offset 0. -func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) { +func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc, error) { if !fi.needSplit { - return nil, Reloc{} + return nil, Reloc{}, nil } // MOV 16(g), X6 (g.stackguard0), g = X27. out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16)) @@ -317,14 +323,14 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) { var reloc Reloc switch fi.splitClass { case 0: - // BLTU X6, SP, done (+8: over the CALL and the JMP back) + // BLTU X6, SP, done (+12: over the CALL and the JMP back) out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...) call := len(out) reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal} out = append(out, wordLE(riscvJType(5, 0))...) out = append(out, jalBack()...) case 1: - // ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+8) + // ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+12) off := int32(fi.autosize - stackSmall) out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...) out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...) @@ -342,7 +348,10 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) { out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...) addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off) if err != nil { - addi = nil + // The ADDI expansion failed: the SP adjustment this class + // depends on is not emittable, and silently dropping it would + // corrupt every stack reference in the body. + return nil, Reloc{}, fmt.Errorf("stack-split guard: %w", err) } out = append(out, addi...) out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...) @@ -351,11 +360,5 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc) { out = append(out, wordLE(riscvJType(5, 0))...) out = append(out, jalBack()...) } - return out, reloc -} - -// riscvGuardBytes emits the guard prefix bytes alone (sizing helper). -func riscvGuardBytes(fi riscvFrameInfo) []byte { - g, _ := riscvGuard(fi) - return g + return out, reloc, nil } diff --git a/asm/riscv_frame_test.go b/asm/riscv_frame_test.go index e0f5e20..5a03510 100644 --- a/asm/riscv_frame_test.go +++ b/asm/riscv_frame_test.go @@ -65,6 +65,47 @@ TEXT ·framed(SB), NOSPLIT, $16-16 } } +// TestRISCVFrameSpadjLargeFrame checks the stack-adjustment boundaries of a +// frame past the imm12 range: the prologue materialises the LR-store address +// and the SP adjustment through X31 (C.LUI + C.ADD + SD, then C.LUI + ADDIW + +// C.ADD), so the SP boundary lands at PC 16, and the RET closes with +// C.LDSP plus the same X31 adjustment, 10 bytes. Regression: both helpers +// assumed the small-frame prologue and reported 8 and 6. +func TestRISCVFrameSpadjLargeFrame(t *testing.T) { + f, errs := parser.Parse("bigframe_riscv64.s", `#include "textflag.h" + +TEXT ·big(SB), NOSPLIT, $9000-8 + MOV a+0(FP), X10 + RET +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileRISCV(f) + if err != nil { + t.Fatalf("AssembleFileRISCV: %v", err) + } + fn := img.Funcs[0] + + // autosize = 9008. Prologue: C.LUI X31 + C.ADD X31,SP (4) + SD (4) + + // C.LUI X31 + ADDIW X31 + C.ADD SP,X31 (8) = 16 bytes to the SP boundary; + // C.SDSP X1 (2) follows, so the body starts at 18. + wantSpadj := []SpadjStep{{PC: 16, Value: 9008}, {PC: 36, Value: 0}} + if len(fn.Spadj) != len(wantSpadj) { + t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj) + } + for i := range wantSpadj { + if fn.Spadj[i] != wantSpadj[i] { + t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i]) + } + } + // The FP load materialises its 9016-byte offset through X31 as well + // (8 bytes), then RET's epilogue (C.LDSP + X31 adjust = 10) plus JALR. + if fn.Size != 18+8+14 { + t.Errorf("size = %d, want %d", fn.Size, 18+8+14) + } +} + // TestRISCVRegAliases checks the Go ABI register aliases that the toolchain // defines: LR is the link register (X1) and TMP is the assembler scratch // register (X31/T6). diff --git a/testdata/verify/rvcstore_riscv64.s b/testdata/verify/rvcstore_riscv64.s new file mode 100644 index 0000000..6cbcf39 --- /dev/null +++ b/testdata/verify/rvcstore_riscv64.s @@ -0,0 +1,48 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// rvcstore exercises the register-relative compressed stores with offsets +// that set immediate bits 4 and 5, the bare FENCE, and the GOROOT width +// spellings of the loads and stores, byte-parity-checked against go tool asm. + +#include "textflag.h" + +TEXT ·stores(SB), NOSPLIT, $0 + SD X9, 24(X8) + SW X10, 16(X11) + FSD F8, 40(X12) + LD 24(X8), X9 + LW 16(X11), X10 + FLD 40(X12), F8 + RET + +TEXT ·fence(SB), NOSPLIT, $0 + FENCE + ECALL + RET + +// The GOROOT width spellings: MOVW lowers to LW/SW and compresses exactly +// like the base mnemonic, the byte/half and unsigned forms stay wide. +TEXT ·argwidths(SB), NOSPLIT, $0-24 + MOVW x+0(FP), X9 + MOVW X9, y+4(FP) + MOVWU z+8(FP), X10 + MOVB b+12(FP), X11 + MOVBU c+13(FP), X12 + MOVH h+16(FP), X13 + MOVHU u+18(FP), X14 + RET + +TEXT ·spwidths(SB), NOSPLIT, $32-0 + MOVW 4(SP), X9 + MOVW X9, 8(SP) + MOV 16(SP), X10 + MOV X10, 24(SP) + RET + +TEXT ·fpwidths(SB), NOSPLIT, $0-16 + MOVD d+0(FP), F8 + MOVD F8, r+8(FP) + MOVW w+0(FP), X9 + MOVW X9, v+4(FP) + RET