From f0512a4e1cb170dd4eadf1d53007712684e70f2f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Thu, 13 Aug 2026 15:42:38 +0200 Subject: [PATCH] fix(asm): complete RISC-V compressed loads/stores and word arithmetic Assisted-by: DeepSeek V4 Pro --- CHANGELOG.md | 6 +++ asm/riscv_assemble.go | 59 +++++++++++++++++++++++ asm/riscv_encode.go | 74 ++++++++++++++++++++--------- asm/riscv_encode_test.go | 8 ++-- testdata/verify/loadstore_riscv64.s | 24 ++++++++++ verify/riscv_groundtruth_test.go | 1 + 6 files changed, 146 insertions(+), 26 deletions(-) create mode 100644 testdata/verify/loadstore_riscv64.s diff --git a/CHANGELOG.md b/CHANGELOG.md index e978a0c..f6f78d1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -55,6 +55,12 @@ Unreleased changes on the `development` branch. completed for `C.ADDI16SP`, `C.SLLI`, `C.SRLI`, `C.SRAI`, `C.ANDI`, `C.NOP`, `C.EBREAK`, `C.MV` (from `ADDI`/`ADD`) and the commutative `AND`/`OR`/`XOR` forms; the byte-exact ground-truth test now covers these. +- **RISC-V compressed loads/stores and word arithmetic.** RVC compression + now also covers the register-relative `C.LW`/`C.SW`/`C.LD`/`C.SD`/ + `C.FLD`/`C.FSD` forms (in addition to the stack-relative `C.LWSP`/`C.SWSP`/ + `C.LDSP`/`C.SDSP`), plus `C.ADDI4SPN`, `C.ADDW` and `C.SUBW`. The + byte-exact ground-truth test exercises these against `GOARCH=riscv64 + go tool asm`. - **Debugger watchpoint slots.** `gasm debug`'s `watch` command always used hardware watchpoint slot 0, so a second `watch` call silently overwrote the first. Watchpoint slots are now tracked in the `Session` (DR0–DR3); diff --git a/asm/riscv_assemble.go b/asm/riscv_assemble.go index 3ebf59d..5df261f 100644 --- a/asm/riscv_assemble.go +++ b/asm/riscv_assemble.go @@ -749,6 +749,10 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) { if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcLSP(0x3, uint32(rd), uint32(imm)), true } + // Register-relative C.LD: both in prime regs, 8-byte scaled offset. + if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 { + return rvcCL(0x3, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true + } // MOV reg, mem → store, try C.SDSP. if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && isMemOperand(ops[1]) { rs2, rs1, imm := extractSDParams(instr, fi) @@ -763,6 +767,28 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) { if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcSSP(0x7, uint32(rs2), uint32(imm)), true } + // Register-relative C.SD: base and source in prime regs. + if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 256 && imm%8 == 0 { + return rvcCS(0x7, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true + } + + case "LW": + rd, rs1, imm := extractLDParams(instr, fi) + if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 256 && imm%4 == 0 { + return rvcLSP(0x2, uint32(rd), uint32(imm)), true + } + if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 128 && imm%4 == 0 { + return rvcCL(0x2, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true + } + + case "SW": + rs2, rs1, imm := extractSDParams(instr, fi) + if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 256 && imm%4 == 0 { + return rvcSSP(0x6, uint32(rs2), uint32(imm)), true + } + if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 128 && imm%4 == 0 { + return rvcCS(0x6, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true + } case "ADDI": rd, rs1, imm := extractITypeParams(instr, fi) @@ -777,6 +803,10 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) { // C.ADDI: funct3=0x0, rs1/rd, nzimm[5:0] return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true } + if isRVCIntReg(rd) && rs1 == 2 && imm != 0 && imm >= 0 && imm < 1024 && imm%4 == 0 { + // C.ADDI4SPN: ADDI $imm, SP, rd for a prime rd. + return rvcCIW(0x0, rvcReg3(rd), uint32(imm)), true + } if rs1 == 0 && rd != 0 && imm >= -32 && imm <= 31 { // C.LI: funct3=0x2, rd, imm[5:0] return rvcCI(0x2, uint32(rd), uint32(imm)&0x3F), true @@ -860,12 +890,37 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) { } } + case "ADDW", "SUBW": + // C.ADDW (0x27,1) / C.SUBW (0x27,0) — CA-type, prime regs. + if len(ops) == 3 { + funct2 := uint32(0x0) + if mnem == "ADDW" { + funct2 = 0x1 + } + rs2 := regFromOperand(ops[0]) + rs1 := regFromOperand(ops[1]) + rd := regFromOperand(ops[2]) + if rd != -1 && rs1 != -1 && rs2 != -1 && isRVCIntReg(rd) { + if rd == rs1 && isRVCIntReg(rs2) { + return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs2)), true + } + // ADDW is commutative; SUBW is not. + if mnem == "ADDW" && isRVCIntReg(rs1) && rd == rs2 { + return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs1)), true + } + } + } + case "FLD": // FLD rd, imm(SP) → C.FLDSP (CI-type, funct3=0x1). rd, rs1, imm := extractLDParams(instr, fi) if rs1 == 2 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcLSP(0x1, uint32(rd), uint32(imm)), true } + // Register-relative C.FLD: rd in F8-F15, base in X8-X15. + if rs1 != -1 && rd != -1 && rd >= 8 && rd <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 { + return rvcCL(0x1, uint32(rd-8), rvcReg3(rs1), uint32(imm)), true + } case "FSD": // FSD rs2, imm(SP) → C.FSDSP (CSS-type, funct3=0x5). @@ -873,6 +928,10 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) { if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcSSP(0x5, uint32(rs2), uint32(imm)), true } + // Register-relative C.FSD: source in F8-F15, base in X8-X15. + if rs1 != -1 && rs2 != -1 && rs2 >= 8 && rs2 <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 { + return rvcCS(0x5, uint32(rs2-8), rvcReg3(rs1), uint32(imm)), true + } case "LUI": // LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in 6 bits. diff --git a/asm/riscv_encode.go b/asm/riscv_encode.go index 3f59cc3..92c66f2 100644 --- a/asm/riscv_encode.go +++ b/asm/riscv_encode.go @@ -466,45 +466,75 @@ func rvcSLLI(rd, shamt uint32) uint16 { return uint16(((shamt>>5)&1)<<12 | (rd << 7) | (shamt&0x1F)<<2 | 0x2) } -// rvcLSP encodes a CI-type stack-relative load: C.LDSP (funct3=3) or -// C.FLDSP (funct3=1). offset is the full byte offset; the immediate bits -// are interleaved per the RISC-V spec: [5:3|8:6]. -func rvcLSP(funct3, rd uint32, offset uint32) uint16 { - // Bit interleave offset bits [5,4,3,8,7,6] → packed value. +// encodeRVCPattern extracts the bits listed in pattern (MSB first) from imm +// into a packed value, matching cmd/internal/obj/riscv's encodeBitPattern. +func encodeRVCPattern(imm uint32, pattern []int) uint32 { packed := uint32(0) - for i, b := range []int{5, 4, 3, 8, 7, 6} { + for _, bit := range pattern { + packed = packed<<1 | (imm>>bit)&1 + } + return packed +} + +// rvcLSP encodes a stack-relative compressed load (op=10 quadrant): C.LWSP +// (funct3=2, 4-byte scale), C.LDSP (funct3=3) or C.FLDSP (funct3=1, 8-byte +// scale). offset is the full byte offset. +func rvcLSP(funct3, rd uint32, offset uint32) uint16 { + pattern := []int{5, 4, 3, 8, 7, 6} + if funct3 == 0x2 { + pattern = []int{5, 4, 3, 2, 7, 6} + } + packed := uint32(0) + for i, b := range pattern { packed |= ((offset >> b) & 1) << (5 - i) } return uint16((funct3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x2) } -// rvcSSP encodes a CSS-type stack-relative store: C.SDSP (funct3=7) or -// C.FSDSP (funct3=5). offset is the full byte offset; the immediate bits -// are interleaved per the RISC-V spec: [5:3|8:6]. +// rvcSSP encodes a stack-relative compressed store (op=10 quadrant): C.SWSP +// (funct3=6, 4-byte scale), C.SDSP (funct3=7) or C.FSDSP (funct3=5, 8-byte +// scale). offset is the full byte offset. func rvcSSP(funct3, rs2 uint32, offset uint32) uint16 { - // Bit interleave offset bits [5,4,3,8,7,6] → packed value. + pattern := []int{5, 4, 3, 8, 7, 6} + if funct3 == 0x6 { + pattern = []int{5, 4, 3, 2, 7, 6} + } packed := uint32(0) - for i, b := range []int{5, 4, 3, 8, 7, 6} { + for i, b := range pattern { packed |= ((offset >> b) & 1) << (5 - i) } return uint16((funct3 << 13) | (packed << 7) | (rs2 << 2) | 0x2) } -// rvcCSS encodes a CSS-type (stack store) compressed instruction. -func rvcCSS(funct3, rs2 uint32, imm uint32) uint16 { - return uint16((funct3 << 13) | (imm << 7) | (rs2 << 2) | 0x2) -} - -// rvcCL encodes a CL-type (load) compressed instruction. -// imm layout: [5:3] in bits [12:10], [2|6] in bits [6:5]. +// rvcCL encodes a register-relative compressed load (op=00 quadrant): C.LW +// (funct3=2), C.LD (funct3=3) or C.FLD (funct3=1). imm is the full byte +// offset; the immediate bits are extracted per the RISC-V CL format. func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 { - bits := uint16((funct3 << 13) | ((imm>>3)&0x7)<<10 | (rs1 << 7) | ((imm & 0x7) << 5) | (rd << 2) | 0x0) - return bits + pattern := []int{5, 4, 3, 7, 6} + if funct3 == 0x2 { + pattern = []int{5, 4, 3, 2, 6} + } + packed := encodeRVCPattern(imm, pattern) + return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rd << 2)) } -// rvcCS encodes a CS-type (store) compressed instruction. +// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW +// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte +// offset; the immediate bits are extracted per the RISC-V CS format. func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 { - return uint16((funct3 << 13) | ((imm>>3)&0x7)<<10 | (rs1 << 7) | ((imm & 0x7) << 5) | (rs2 << 2) | 0x0) + pattern := []int{5, 3, 7, 6} + if funct3 == 0x6 { + pattern = []int{5, 3, 2, 6} + } + packed := encodeRVCPattern(imm, pattern) + return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2)) +} + +// rvcCIW encodes a CIW-type compressed immediate wide instruction: C.ADDI4SPN +// (funct3=0). imm is the raw byte offset. +func rvcCIW(funct3, rd uint32, imm uint32) uint16 { + packed := encodeRVCPattern(imm, []int{5, 4, 9, 8, 7, 6, 2, 3}) + return uint16((funct3 << 13) | (packed << 5) | (rd << 2)) } // rvcCJ encodes a CJ-type (jump) compressed instruction. diff --git a/asm/riscv_encode_test.go b/asm/riscv_encode_test.go index f3f1577..3c4d316 100644 --- a/asm/riscv_encode_test.go +++ b/asm/riscv_encode_test.go @@ -81,9 +81,9 @@ TEXT ·mem(SB), NOSPLIT, $0 RET `) code := assembleRISCVHelper(t, fn) - // 4 loads/stores (4B each) + JALR RET (4B) = 20 - if len(code) != 20 { - t.Errorf("expected 20 bytes, got %d", len(code)) + // Four register-relative loads/stores compress (2B each) + JALR (4B) = 12. + if len(code) != 12 { + t.Errorf("expected 12 bytes, got %d", len(code)) } } @@ -400,7 +400,7 @@ func TestRISCV_encodings(t *testing.T) { {"LWU", "LWU (X10), X11\nRET\n", 8}, {"SB", "SB X10, (X11)\nRET\n", 8}, {"SH", "SH X10, (X11)\nRET\n", 8}, - {"SW", "SW X10, (X11)\nRET\n", 8}, + {"SW", "SW X10, (X11)\nRET\n", 6}, {"LUI", "LUI X10, $0x12345\nRET\n", 8}, {"AUIPC", "AUIPC X10, $0\nRET\n", 8}, {"FLW", "FLW (X10), F10\nRET\n", 8}, diff --git a/testdata/verify/loadstore_riscv64.s b/testdata/verify/loadstore_riscv64.s new file mode 100644 index 0000000..4633a3b --- /dev/null +++ b/testdata/verify/loadstore_riscv64.s @@ -0,0 +1,24 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +#include "textflag.h" + +TEXT ·ldst(SB), NOSPLIT, $0 + LD (X8), X9 + SD X9, (X8) + LW (X8), X9 + SW X9, (X8) + LD 8(X2), X10 + SD X10, 16(X2) + LW 4(X2), X11 + SW X11, 8(X2) + RET + +TEXT ·addi4spn(SB), NOSPLIT, $0 + ADDI $16, X2, X8 + RET + +TEXT ·wordarith(SB), NOSPLIT, $0 + ADDW X9, X8, X8 + SUBW X9, X8, X8 + RET diff --git a/verify/riscv_groundtruth_test.go b/verify/riscv_groundtruth_test.go index b3c2cd2..d252bc8 100644 --- a/verify/riscv_groundtruth_test.go +++ b/verify/riscv_groundtruth_test.go @@ -20,6 +20,7 @@ func TestGroundTruthRISCV(t *testing.T) { for _, path := range []string{ "../testdata/verify/basic_riscv64.s", "../testdata/verify/rvc_riscv64.s", + "../testdata/verify/loadstore_riscv64.s", } { t.Run(path, func(t *testing.T) { testGroundTruthRISCVFile(t, path)