fix(asm): match RISC-V branch and jump encodings

This commit is contained in:
2026-08-13 17:57:10 +02:00
parent 31a2cee382
commit 681a449c01
6 changed files with 62 additions and 72 deletions
+7
View File
@@ -75,6 +75,13 @@ Unreleased changes on the `development` branch.
`ADDI`/`LUI`/`ADDIW` to `C.LI`/`C.LUI`/`C.ADDIW` when their immediate `ADDI`/`LUI`/`ADDIW` to `C.LI`/`C.LUI`/`C.ADDIW` when their immediate
fits six signed bits. A byte-exact ground-truth test covers zero, small, fits six signed bits. A byte-exact ground-truth test covers zero, small,
negative, and 32-bit immediates against `GOARCH=riscv64 go tool asm`. negative, and 32-bit immediates against `GOARCH=riscv64 go tool asm`.
- **RISC-V branch/jump compression.** `JMP`/`JAL` were being compressed to
`C.J` and `BEQ`/`BNE` (with `X0`) to `C.BEQZ`/`C.BNEZ`, but `go tool asm`
never emits these compressed forms. They now emit the 32-bit `JAL` and
branch encodings the toolchain writes; the dead `C.J`/`C.BEQZ`/`C.BNEZ`
encoders were removed, and the `C.LUI` direct-instruction compression now
uses the correct six-bit signed range. A byte-exact ground-truth test
covers the branch family and jumps against `GOARCH=riscv64 go tool asm`.
- **Debugger watchpoint slots.** `gasm debug`'s `watch` command always used - **Debugger watchpoint slots.** `gasm debug`'s `watch` command always used
hardware watchpoint slot 0, so a second `watch` call silently overwrote hardware watchpoint slot 0, so a second `watch` call silently overwrote
the first. Watchpoint slots are now tracked in the `Session` (DR0–DR3); the first. Watchpoint slots are now tracked in the `Session` (DR0–DR3);
+10 -34
View File
@@ -203,7 +203,8 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
word = riscvIType(riscvEnc{0x13, 0x0, 0x00}, 0, 0, 0) word = riscvIType(riscvEnc{0x13, 0x0, 0x00}, 0, 0, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JMP": case "JMP":
// JMP = JAL X0, target. Try C.J compression. // JMP = JAL X0, target. The Go assembler never compresses this to
// C.J, so always emit the 32-bit JAL.
var target string var target string
if len(ops) >= 1 { if len(ops) >= 1 {
target = labelFromOperand(ops[0]) target = labelFromOperand(ops[0])
@@ -213,11 +214,6 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
} }
offset := int32(targetOff - pc) offset := int32(targetOff - pc)
// C.J: funct3=0x5, offset in ±2 KB, bit 0 must be 0.
if offset >= -2048 && offset <= 2046 && offset%2 == 0 {
c16 := rvcCJ(0x5, offset)
return []byte{byte(c16), byte(c16 >> 8)}, nil
}
word = riscvJType(0, offset) word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JAL": case "JAL":
@@ -234,11 +230,6 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
} }
offset := int32(targetOff - pc) offset := int32(targetOff - pc)
// JAL X0, target → C.J when offset fits.
if rd == 0 && offset >= -2048 && offset <= 2046 && offset%2 == 0 {
c16 := rvcCJ(0x5, offset)
return []byte{byte(c16), byte(c16 >> 8)}, nil
}
word = riscvJType(rd, offset) word = riscvJType(rd, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
@@ -492,18 +483,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return nil, fmt.Errorf("invalid register in %s", mnem) return nil, fmt.Errorf("invalid register in %s", mnem)
} }
// Try C.BEQZ / C.BNEZ compression. // The Go assembler never compresses branches to C.BEQZ/C.BNEZ.
if (mnem == "BEQ" || mnem == "BNE") && rs2 == 0 && isRVCIntReg(rs1) {
if cOff := offset; cOff >= -256 && cOff <= 254 && cOff%2 == 0 {
funct3 := uint32(0x6) // C.BEQZ
if mnem == "BNE" {
funct3 = 0x7 // C.BNEZ
}
c16 := rvcCB(funct3, rvcReg3(rs1), offset)
return []byte{byte(c16), byte(c16 >> 8)}, nil
}
}
word = riscvBType(enc, rs1, rs2, offset) word = riscvBType(enc, rs1, rs2, offset)
// U-type: rd, imm. // U-type: rd, imm.
@@ -969,24 +949,19 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
} }
case "JAL": case "JAL":
// JAL X0, target → C.J when offset fits in ±2KB. // JAL/JMP are never compressed to C.J by the Go assembler.
if len(ops) >= 1 {
// For JAL with implicit rd=0 (JMP alias), check target.
// C.J: funct3=0x5
// Offset is computed at encode time — we can't check it here.
return 0, false return 0, false
}
case "JMP": case "JMP":
// C.J — handled in encodeRISCVInstr with actual offset. // JAL/JMP are never compressed to C.J by the Go assembler.
return 0, false return 0, false
case "BEQ": case "BEQ":
// C.BEQZ — handled in encodeRISCVInstr with actual offset. // Branches are never compressed to C.BEQZ/C.BNEZ.
return 0, false return 0, false
case "BNE": case "BNE":
// C.BNEZ — handled in encodeRISCVInstr with actual offset. // Branches are never compressed to C.BEQZ/C.BNEZ.
return 0, false return 0, false
case "ADD": case "ADD":
@@ -1082,11 +1057,12 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
} }
case "LUI": case "LUI":
// LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in 6 bits. // LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in six
// signed bits (matching the toolchain's compress pass).
if len(ops) == 2 { if len(ops) == 2 {
rd := regFromOperand(ops[0]) rd := regFromOperand(ops[0])
imm := immFromOperand(ops[1]) imm := immFromOperand(ops[1])
if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= 1 && imm <= 63 { if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= -32 && imm <= 31 {
return rvcCI(0x3, uint32(rd), uint32(imm)&0x3F), true return rvcCI(0x3, uint32(rd), uint32(imm)&0x3F), true
} }
} }
-29
View File
@@ -537,41 +537,12 @@ func rvcCIW(funct3, rd uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | (packed << 5) | (rd << 2)) return uint16((funct3 << 13) | (packed << 5) | (rd << 2))
} }
// rvcCJ encodes a CJ-type (jump) compressed instruction.
// offset is a 12-bit signed offset (bit 0 is always 0).
func rvcCJ(funct3 uint32, offset int32) uint16 {
uoff := uint32(offset) & 0xFFE
bits := ((uoff >> 11) & 1) << 10
bits |= ((uoff >> 4) & 1) << 9
bits |= ((uoff >> 9) & 0x3) << 7
bits |= ((uoff >> 10) & 1) << 6
bits |= ((uoff >> 6) & 1) << 5
bits |= ((uoff >> 7) & 1) << 4
bits |= ((uoff >> 1) & 0x7) << 1
bits |= ((uoff >> 5) & 1)
return uint16((funct3 << 13) | (bits << 2) | 0x1)
}
// rvcCA encodes a CA-type (arithmetic) compressed instruction. // rvcCA encodes a CA-type (arithmetic) compressed instruction.
// Format: funct6[15:10] | rd'/rs1'[9:7] | funct2[6:5] | rs2'[4:2] | op=01. // Format: funct6[15:10] | rd'/rs1'[9:7] | funct2[6:5] | rs2'[4:2] | op=01.
func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 { func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 {
return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1) return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1)
} }
// rvcCB encodes a CB-type (branch) compressed instruction.
// Format: funct3[15:13] | offset[8|4:3] | rs1'[9:7] | offset[7:6|2:1|5] | op=01.
// Bit pattern for offset: [8|4:3|7:6|2:1|5]
func rvcCB(funct3, rs1 uint32, offset int32) uint16 {
uoff := uint32(offset) & 0x1FE // bits [8:1]
offBits := uint32(0)
offBits |= ((uoff >> 8) & 1) << 10 // bit 10 = offset[8]
offBits |= ((uoff >> 3) & 0x3) << 8 // bits 9:8 = offset[4:3]
offBits |= ((uoff >> 6) & 0x3) << 6 // bits 7:6 = offset[7:6]
offBits |= ((uoff >> 1) & 0x3) << 3 // bits 4:3 = offset[2:1]
offBits |= ((uoff >> 5) & 1) << 2 // bit 2 = offset[5]
return uint16((funct3 << 13) | offBits | (rs1 << 7) | 0x1)
}
// rvcCBShift encodes a CB-type shift/immediate compressed instruction // rvcCBShift encodes a CB-type shift/immediate compressed instruction
// (C.SRLI, C.SRAI, C.ANDI). rd is the 3-bit prime-register index; imm is // (C.SRLI, C.SRAI, C.ANDI). rd is the 3-bit prime-register index; imm is
// the 6-bit shamt/immediate; funct2 selects the operation (0=SRLI, 1=SRAI, // the 6-bit shamt/immediate; funct2 selects the operation (0=SRLI, 1=SRAI,
+8 -8
View File
@@ -428,7 +428,7 @@ TEXT ·`+tt.name+`(SB), NOSPLIT, $0
} }
func TestRISCV_RVC_branch(t *testing.T) { func TestRISCV_RVC_branch(t *testing.T) {
// BEQ rs, X0, target → C.BEQZ when rs is in prime regs and offset fits. // Branches are never RVC-compressed (no C.BEQZ/C.BNEZ), matching go tool asm.
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cbeqz(SB), NOSPLIT, $0 TEXT ·cbeqz(SB), NOSPLIT, $0
ADDI $1, X10, X10 ADDI $1, X10, X10
@@ -438,14 +438,14 @@ done:
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.ADDI(2) + C.BEQZ(2) + C.ADDI(2) + JALR(4) = 10 (all compress) // C.ADDI(2) + BEQ(4) + C.ADDI(2) + JALR(4) = 12
if len(code) != 10 { if len(code) != 12 {
t.Errorf("expected 10 bytes with C.BEQZ, got %d", len(code)) t.Errorf("expected 12 bytes with uncompressed BEQ, got %d", len(code))
} }
} }
func TestRISCV_RVC_CJ(t *testing.T) { func TestRISCV_RVC_CJ(t *testing.T) {
// JMP target → C.J when offset fits. // JMP target → JAL X0 (never compressed to C.J), matching go tool asm.
fn := firstTextRISCV(t, `#include "textflag.h" fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cj(SB), NOSPLIT, $0 TEXT ·cj(SB), NOSPLIT, $0
JMP done JMP done
@@ -453,9 +453,9 @@ func TestRISCV_RVC_CJ(t *testing.T) {
RET RET
`) `)
code := assembleRISCVHelper(t, fn) code := assembleRISCVHelper(t, fn)
// C.J(2) + JALR(4) = 6 // JAL(4) + JALR(4) = 8
if len(code) != 6 { if len(code) != 8 {
t.Errorf("expected 6 bytes with C.J, got %d", len(code)) t.Errorf("expected 8 bytes with uncompressed JMP, got %d", len(code))
} }
} }
+35
View File
@@ -0,0 +1,35 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
TEXT ·branches(SB), NOSPLIT, $0
ADDI $1, X10, X10
BEQ X10, X11, beq_done
ADDI $2, X10, X10
beq_done:
BNE X10, X11, bne_done
ADDI $3, X10, X10
bne_done:
BLT X10, X11, blt_done
ADDI $4, X10, X10
blt_done:
BGE X10, X11, bge_done
ADDI $5, X10, X10
bge_done:
BLTU X10, X11, bltu_done
ADDI $6, X10, X10
bltu_done:
BGEU X10, X11, bgeu_done
ADDI $7, X10, X10
bgeu_done:
RET
TEXT ·jumps(SB), NOSPLIT, $0
JMP done
ADDI $1, X10, X10
done:
JAL X11, skip
ADDI $2, X10, X10
skip:
RET
+1
View File
@@ -23,6 +23,7 @@ func TestGroundTruthRISCV(t *testing.T) {
"../testdata/verify/loadstore_riscv64.s", "../testdata/verify/loadstore_riscv64.s",
"../testdata/verify/largeimm_riscv64.s", "../testdata/verify/largeimm_riscv64.s",
"../testdata/verify/movimm_riscv64.s", "../testdata/verify/movimm_riscv64.s",
"../testdata/verify/branch_riscv64.s",
} { } {
t.Run(path, func(t *testing.T) { t.Run(path, func(t *testing.T) {
testGroundTruthRISCVFile(t, path) testGroundTruthRISCVFile(t, path)