fix(asm): complete RISC-V compressed loads/stores and word arithmetic

Assisted-by: DeepSeek V4 Pro
This commit is contained in:
2026-08-13 15:42:38 +02:00
parent 373c09f725
commit f0512a4e1c
6 changed files with 146 additions and 26 deletions
+59
View File
@@ -749,6 +749,10 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcLSP(0x3, uint32(rd), uint32(imm)), true
}
// Register-relative C.LD: both in prime regs, 8-byte scaled offset.
if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCL(0x3, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true
}
// MOV reg, mem → store, try C.SDSP.
if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && isMemOperand(ops[1]) {
rs2, rs1, imm := extractSDParams(instr, fi)
@@ -763,6 +767,28 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcSSP(0x7, uint32(rs2), uint32(imm)), true
}
// Register-relative C.SD: base and source in prime regs.
if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCS(0x7, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true
}
case "LW":
rd, rs1, imm := extractLDParams(instr, fi)
if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 256 && imm%4 == 0 {
return rvcLSP(0x2, uint32(rd), uint32(imm)), true
}
if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 128 && imm%4 == 0 {
return rvcCL(0x2, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true
}
case "SW":
rs2, rs1, imm := extractSDParams(instr, fi)
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 256 && imm%4 == 0 {
return rvcSSP(0x6, uint32(rs2), uint32(imm)), true
}
if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 128 && imm%4 == 0 {
return rvcCS(0x6, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true
}
case "ADDI":
rd, rs1, imm := extractITypeParams(instr, fi)
@@ -777,6 +803,10 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
// C.ADDI: funct3=0x0, rs1/rd, nzimm[5:0]
return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true
}
if isRVCIntReg(rd) && rs1 == 2 && imm != 0 && imm >= 0 && imm < 1024 && imm%4 == 0 {
// C.ADDI4SPN: ADDI $imm, SP, rd for a prime rd.
return rvcCIW(0x0, rvcReg3(rd), uint32(imm)), true
}
if rs1 == 0 && rd != 0 && imm >= -32 && imm <= 31 {
// C.LI: funct3=0x2, rd, imm[5:0]
return rvcCI(0x2, uint32(rd), uint32(imm)&0x3F), true
@@ -860,12 +890,37 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
}
}
case "ADDW", "SUBW":
// C.ADDW (0x27,1) / C.SUBW (0x27,0) — CA-type, prime regs.
if len(ops) == 3 {
funct2 := uint32(0x0)
if mnem == "ADDW" {
funct2 = 0x1
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd != -1 && rs1 != -1 && rs2 != -1 && isRVCIntReg(rd) {
if rd == rs1 && isRVCIntReg(rs2) {
return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs2)), true
}
// ADDW is commutative; SUBW is not.
if mnem == "ADDW" && isRVCIntReg(rs1) && rd == rs2 {
return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs1)), true
}
}
}
case "FLD":
// FLD rd, imm(SP) → C.FLDSP (CI-type, funct3=0x1).
rd, rs1, imm := extractLDParams(instr, fi)
if rs1 == 2 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcLSP(0x1, uint32(rd), uint32(imm)), true
}
// Register-relative C.FLD: rd in F8-F15, base in X8-X15.
if rs1 != -1 && rd != -1 && rd >= 8 && rd <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCL(0x1, uint32(rd-8), rvcReg3(rs1), uint32(imm)), true
}
case "FSD":
// FSD rs2, imm(SP) → C.FSDSP (CSS-type, funct3=0x5).
@@ -873,6 +928,10 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcSSP(0x5, uint32(rs2), uint32(imm)), true
}
// Register-relative C.FSD: source in F8-F15, base in X8-X15.
if rs1 != -1 && rs2 != -1 && rs2 >= 8 && rs2 <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCS(0x5, uint32(rs2-8), rvcReg3(rs1), uint32(imm)), true
}
case "LUI":
// LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in 6 bits.
+52 -22
View File
@@ -466,45 +466,75 @@ func rvcSLLI(rd, shamt uint32) uint16 {
return uint16(((shamt>>5)&1)<<12 | (rd << 7) | (shamt&0x1F)<<2 | 0x2)
}
// rvcLSP encodes a CI-type stack-relative load: C.LDSP (funct3=3) or
// C.FLDSP (funct3=1). offset is the full byte offset; the immediate bits
// are interleaved per the RISC-V spec: [5:3|8:6].
func rvcLSP(funct3, rd uint32, offset uint32) uint16 {
// Bit interleave offset bits [5,4,3,8,7,6] → packed value.
// encodeRVCPattern extracts the bits listed in pattern (MSB first) from imm
// into a packed value, matching cmd/internal/obj/riscv's encodeBitPattern.
func encodeRVCPattern(imm uint32, pattern []int) uint32 {
packed := uint32(0)
for i, b := range []int{5, 4, 3, 8, 7, 6} {
for _, bit := range pattern {
packed = packed<<1 | (imm>>bit)&1
}
return packed
}
// rvcLSP encodes a stack-relative compressed load (op=10 quadrant): C.LWSP
// (funct3=2, 4-byte scale), C.LDSP (funct3=3) or C.FLDSP (funct3=1, 8-byte
// scale). offset is the full byte offset.
func rvcLSP(funct3, rd uint32, offset uint32) uint16 {
pattern := []int{5, 4, 3, 8, 7, 6}
if funct3 == 0x2 {
pattern = []int{5, 4, 3, 2, 7, 6}
}
packed := uint32(0)
for i, b := range pattern {
packed |= ((offset >> b) & 1) << (5 - i)
}
return uint16((funct3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x2)
}
// rvcSSP encodes a CSS-type stack-relative store: C.SDSP (funct3=7) or
// C.FSDSP (funct3=5). offset is the full byte offset; the immediate bits
// are interleaved per the RISC-V spec: [5:3|8:6].
// rvcSSP encodes a stack-relative compressed store (op=10 quadrant): C.SWSP
// (funct3=6, 4-byte scale), C.SDSP (funct3=7) or C.FSDSP (funct3=5, 8-byte
// scale). offset is the full byte offset.
func rvcSSP(funct3, rs2 uint32, offset uint32) uint16 {
// Bit interleave offset bits [5,4,3,8,7,6] → packed value.
pattern := []int{5, 4, 3, 8, 7, 6}
if funct3 == 0x6 {
pattern = []int{5, 4, 3, 2, 7, 6}
}
packed := uint32(0)
for i, b := range []int{5, 4, 3, 8, 7, 6} {
for i, b := range pattern {
packed |= ((offset >> b) & 1) << (5 - i)
}
return uint16((funct3 << 13) | (packed << 7) | (rs2 << 2) | 0x2)
}
// rvcCSS encodes a CSS-type (stack store) compressed instruction.
func rvcCSS(funct3, rs2 uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | (imm << 7) | (rs2 << 2) | 0x2)
}
// rvcCL encodes a CL-type (load) compressed instruction.
// imm layout: [5:3] in bits [12:10], [2|6] in bits [6:5].
// rvcCL encodes a register-relative compressed load (op=00 quadrant): C.LW
// (funct3=2), C.LD (funct3=3) or C.FLD (funct3=1). imm is the full byte
// offset; the immediate bits are extracted per the RISC-V CL format.
func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
bits := uint16((funct3 << 13) | ((imm>>3)&0x7)<<10 | (rs1 << 7) | ((imm & 0x7) << 5) | (rd << 2) | 0x0)
return bits
pattern := []int{5, 4, 3, 7, 6}
if funct3 == 0x2 {
pattern = []int{5, 4, 3, 2, 6}
}
packed := encodeRVCPattern(imm, pattern)
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rd << 2))
}
// rvcCS encodes a CS-type (store) compressed instruction.
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
// offset; the immediate bits are extracted per the RISC-V CS format.
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | ((imm>>3)&0x7)<<10 | (rs1 << 7) | ((imm & 0x7) << 5) | (rs2 << 2) | 0x0)
pattern := []int{5, 3, 7, 6}
if funct3 == 0x6 {
pattern = []int{5, 3, 2, 6}
}
packed := encodeRVCPattern(imm, pattern)
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
}
// rvcCIW encodes a CIW-type compressed immediate wide instruction: C.ADDI4SPN
// (funct3=0). imm is the raw byte offset.
func rvcCIW(funct3, rd uint32, imm uint32) uint16 {
packed := encodeRVCPattern(imm, []int{5, 4, 9, 8, 7, 6, 2, 3})
return uint16((funct3 << 13) | (packed << 5) | (rd << 2))
}
// rvcCJ encodes a CJ-type (jump) compressed instruction.
+4 -4
View File
@@ -81,9 +81,9 @@ TEXT ·mem(SB), NOSPLIT, $0
RET
`)
code := assembleRISCVHelper(t, fn)
// 4 loads/stores (4B each) + JALR RET (4B) = 20
if len(code) != 20 {
t.Errorf("expected 20 bytes, got %d", len(code))
// Four register-relative loads/stores compress (2B each) + JALR (4B) = 12.
if len(code) != 12 {
t.Errorf("expected 12 bytes, got %d", len(code))
}
}
@@ -400,7 +400,7 @@ func TestRISCV_encodings(t *testing.T) {
{"LWU", "LWU (X10), X11\nRET\n", 8},
{"SB", "SB X10, (X11)\nRET\n", 8},
{"SH", "SH X10, (X11)\nRET\n", 8},
{"SW", "SW X10, (X11)\nRET\n", 8},
{"SW", "SW X10, (X11)\nRET\n", 6},
{"LUI", "LUI X10, $0x12345\nRET\n", 8},
{"AUIPC", "AUIPC X10, $0\nRET\n", 8},
{"FLW", "FLW (X10), F10\nRET\n", 8},