fix(asm): match RISC-V branch and jump encodings
This commit is contained in:
@@ -75,6 +75,13 @@ Unreleased changes on the `development` branch.
|
|||||||
`ADDI`/`LUI`/`ADDIW` to `C.LI`/`C.LUI`/`C.ADDIW` when their immediate
|
`ADDI`/`LUI`/`ADDIW` to `C.LI`/`C.LUI`/`C.ADDIW` when their immediate
|
||||||
fits six signed bits. A byte-exact ground-truth test covers zero, small,
|
fits six signed bits. A byte-exact ground-truth test covers zero, small,
|
||||||
negative, and 32-bit immediates against `GOARCH=riscv64 go tool asm`.
|
negative, and 32-bit immediates against `GOARCH=riscv64 go tool asm`.
|
||||||
|
- **RISC-V branch/jump compression.** `JMP`/`JAL` were being compressed to
|
||||||
|
`C.J` and `BEQ`/`BNE` (with `X0`) to `C.BEQZ`/`C.BNEZ`, but `go tool asm`
|
||||||
|
never emits these compressed forms. They now emit the 32-bit `JAL` and
|
||||||
|
branch encodings the toolchain writes; the dead `C.J`/`C.BEQZ`/`C.BNEZ`
|
||||||
|
encoders were removed, and the `C.LUI` direct-instruction compression now
|
||||||
|
uses the correct six-bit signed range. A byte-exact ground-truth test
|
||||||
|
covers the branch family and jumps against `GOARCH=riscv64 go tool asm`.
|
||||||
- **Debugger watchpoint slots.** `gasm debug`'s `watch` command always used
|
- **Debugger watchpoint slots.** `gasm debug`'s `watch` command always used
|
||||||
hardware watchpoint slot 0, so a second `watch` call silently overwrote
|
hardware watchpoint slot 0, so a second `watch` call silently overwrote
|
||||||
the first. Watchpoint slots are now tracked in the `Session` (DR0–DR3);
|
the first. Watchpoint slots are now tracked in the `Session` (DR0–DR3);
|
||||||
|
|||||||
+11
-35
@@ -203,7 +203,8 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
|||||||
word = riscvIType(riscvEnc{0x13, 0x0, 0x00}, 0, 0, 0)
|
word = riscvIType(riscvEnc{0x13, 0x0, 0x00}, 0, 0, 0)
|
||||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||||
case "JMP":
|
case "JMP":
|
||||||
// JMP = JAL X0, target. Try C.J compression.
|
// JMP = JAL X0, target. The Go assembler never compresses this to
|
||||||
|
// C.J, so always emit the 32-bit JAL.
|
||||||
var target string
|
var target string
|
||||||
if len(ops) >= 1 {
|
if len(ops) >= 1 {
|
||||||
target = labelFromOperand(ops[0])
|
target = labelFromOperand(ops[0])
|
||||||
@@ -213,11 +214,6 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
|||||||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||||||
}
|
}
|
||||||
offset := int32(targetOff - pc)
|
offset := int32(targetOff - pc)
|
||||||
// C.J: funct3=0x5, offset in ±2 KB, bit 0 must be 0.
|
|
||||||
if offset >= -2048 && offset <= 2046 && offset%2 == 0 {
|
|
||||||
c16 := rvcCJ(0x5, offset)
|
|
||||||
return []byte{byte(c16), byte(c16 >> 8)}, nil
|
|
||||||
}
|
|
||||||
word = riscvJType(0, offset)
|
word = riscvJType(0, offset)
|
||||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||||
case "JAL":
|
case "JAL":
|
||||||
@@ -234,11 +230,6 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
|||||||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||||||
}
|
}
|
||||||
offset := int32(targetOff - pc)
|
offset := int32(targetOff - pc)
|
||||||
// JAL X0, target → C.J when offset fits.
|
|
||||||
if rd == 0 && offset >= -2048 && offset <= 2046 && offset%2 == 0 {
|
|
||||||
c16 := rvcCJ(0x5, offset)
|
|
||||||
return []byte{byte(c16), byte(c16 >> 8)}, nil
|
|
||||||
}
|
|
||||||
word = riscvJType(rd, offset)
|
word = riscvJType(rd, offset)
|
||||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||||
|
|
||||||
@@ -492,18 +483,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
|||||||
return nil, fmt.Errorf("invalid register in %s", mnem)
|
return nil, fmt.Errorf("invalid register in %s", mnem)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Try C.BEQZ / C.BNEZ compression.
|
// The Go assembler never compresses branches to C.BEQZ/C.BNEZ.
|
||||||
if (mnem == "BEQ" || mnem == "BNE") && rs2 == 0 && isRVCIntReg(rs1) {
|
|
||||||
if cOff := offset; cOff >= -256 && cOff <= 254 && cOff%2 == 0 {
|
|
||||||
funct3 := uint32(0x6) // C.BEQZ
|
|
||||||
if mnem == "BNE" {
|
|
||||||
funct3 = 0x7 // C.BNEZ
|
|
||||||
}
|
|
||||||
c16 := rvcCB(funct3, rvcReg3(rs1), offset)
|
|
||||||
return []byte{byte(c16), byte(c16 >> 8)}, nil
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
word = riscvBType(enc, rs1, rs2, offset)
|
word = riscvBType(enc, rs1, rs2, offset)
|
||||||
|
|
||||||
// U-type: rd, imm.
|
// U-type: rd, imm.
|
||||||
@@ -969,24 +949,19 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
case "JAL":
|
case "JAL":
|
||||||
// JAL X0, target → C.J when offset fits in ±2KB.
|
// JAL/JMP are never compressed to C.J by the Go assembler.
|
||||||
if len(ops) >= 1 {
|
return 0, false
|
||||||
// For JAL with implicit rd=0 (JMP alias), check target.
|
|
||||||
// C.J: funct3=0x5
|
|
||||||
// Offset is computed at encode time — we can't check it here.
|
|
||||||
return 0, false
|
|
||||||
}
|
|
||||||
|
|
||||||
case "JMP":
|
case "JMP":
|
||||||
// C.J — handled in encodeRISCVInstr with actual offset.
|
// JAL/JMP are never compressed to C.J by the Go assembler.
|
||||||
return 0, false
|
return 0, false
|
||||||
|
|
||||||
case "BEQ":
|
case "BEQ":
|
||||||
// C.BEQZ — handled in encodeRISCVInstr with actual offset.
|
// Branches are never compressed to C.BEQZ/C.BNEZ.
|
||||||
return 0, false
|
return 0, false
|
||||||
|
|
||||||
case "BNE":
|
case "BNE":
|
||||||
// C.BNEZ — handled in encodeRISCVInstr with actual offset.
|
// Branches are never compressed to C.BEQZ/C.BNEZ.
|
||||||
return 0, false
|
return 0, false
|
||||||
|
|
||||||
case "ADD":
|
case "ADD":
|
||||||
@@ -1082,11 +1057,12 @@ func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
case "LUI":
|
case "LUI":
|
||||||
// LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in 6 bits.
|
// LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in six
|
||||||
|
// signed bits (matching the toolchain's compress pass).
|
||||||
if len(ops) == 2 {
|
if len(ops) == 2 {
|
||||||
rd := regFromOperand(ops[0])
|
rd := regFromOperand(ops[0])
|
||||||
imm := immFromOperand(ops[1])
|
imm := immFromOperand(ops[1])
|
||||||
if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= 1 && imm <= 63 {
|
if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= -32 && imm <= 31 {
|
||||||
return rvcCI(0x3, uint32(rd), uint32(imm)&0x3F), true
|
return rvcCI(0x3, uint32(rd), uint32(imm)&0x3F), true
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -537,41 +537,12 @@ func rvcCIW(funct3, rd uint32, imm uint32) uint16 {
|
|||||||
return uint16((funct3 << 13) | (packed << 5) | (rd << 2))
|
return uint16((funct3 << 13) | (packed << 5) | (rd << 2))
|
||||||
}
|
}
|
||||||
|
|
||||||
// rvcCJ encodes a CJ-type (jump) compressed instruction.
|
|
||||||
// offset is a 12-bit signed offset (bit 0 is always 0).
|
|
||||||
func rvcCJ(funct3 uint32, offset int32) uint16 {
|
|
||||||
uoff := uint32(offset) & 0xFFE
|
|
||||||
bits := ((uoff >> 11) & 1) << 10
|
|
||||||
bits |= ((uoff >> 4) & 1) << 9
|
|
||||||
bits |= ((uoff >> 9) & 0x3) << 7
|
|
||||||
bits |= ((uoff >> 10) & 1) << 6
|
|
||||||
bits |= ((uoff >> 6) & 1) << 5
|
|
||||||
bits |= ((uoff >> 7) & 1) << 4
|
|
||||||
bits |= ((uoff >> 1) & 0x7) << 1
|
|
||||||
bits |= ((uoff >> 5) & 1)
|
|
||||||
return uint16((funct3 << 13) | (bits << 2) | 0x1)
|
|
||||||
}
|
|
||||||
|
|
||||||
// rvcCA encodes a CA-type (arithmetic) compressed instruction.
|
// rvcCA encodes a CA-type (arithmetic) compressed instruction.
|
||||||
// Format: funct6[15:10] | rd'/rs1'[9:7] | funct2[6:5] | rs2'[4:2] | op=01.
|
// Format: funct6[15:10] | rd'/rs1'[9:7] | funct2[6:5] | rs2'[4:2] | op=01.
|
||||||
func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 {
|
func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 {
|
||||||
return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1)
|
return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1)
|
||||||
}
|
}
|
||||||
|
|
||||||
// rvcCB encodes a CB-type (branch) compressed instruction.
|
|
||||||
// Format: funct3[15:13] | offset[8|4:3] | rs1'[9:7] | offset[7:6|2:1|5] | op=01.
|
|
||||||
// Bit pattern for offset: [8|4:3|7:6|2:1|5]
|
|
||||||
func rvcCB(funct3, rs1 uint32, offset int32) uint16 {
|
|
||||||
uoff := uint32(offset) & 0x1FE // bits [8:1]
|
|
||||||
offBits := uint32(0)
|
|
||||||
offBits |= ((uoff >> 8) & 1) << 10 // bit 10 = offset[8]
|
|
||||||
offBits |= ((uoff >> 3) & 0x3) << 8 // bits 9:8 = offset[4:3]
|
|
||||||
offBits |= ((uoff >> 6) & 0x3) << 6 // bits 7:6 = offset[7:6]
|
|
||||||
offBits |= ((uoff >> 1) & 0x3) << 3 // bits 4:3 = offset[2:1]
|
|
||||||
offBits |= ((uoff >> 5) & 1) << 2 // bit 2 = offset[5]
|
|
||||||
return uint16((funct3 << 13) | offBits | (rs1 << 7) | 0x1)
|
|
||||||
}
|
|
||||||
|
|
||||||
// rvcCBShift encodes a CB-type shift/immediate compressed instruction
|
// rvcCBShift encodes a CB-type shift/immediate compressed instruction
|
||||||
// (C.SRLI, C.SRAI, C.ANDI). rd is the 3-bit prime-register index; imm is
|
// (C.SRLI, C.SRAI, C.ANDI). rd is the 3-bit prime-register index; imm is
|
||||||
// the 6-bit shamt/immediate; funct2 selects the operation (0=SRLI, 1=SRAI,
|
// the 6-bit shamt/immediate; funct2 selects the operation (0=SRLI, 1=SRAI,
|
||||||
|
|||||||
@@ -428,7 +428,7 @@ TEXT ·`+tt.name+`(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestRISCV_RVC_branch(t *testing.T) {
|
func TestRISCV_RVC_branch(t *testing.T) {
|
||||||
// BEQ rs, X0, target → C.BEQZ when rs is in prime regs and offset fits.
|
// Branches are never RVC-compressed (no C.BEQZ/C.BNEZ), matching go tool asm.
|
||||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
TEXT ·cbeqz(SB), NOSPLIT, $0
|
TEXT ·cbeqz(SB), NOSPLIT, $0
|
||||||
ADDI $1, X10, X10
|
ADDI $1, X10, X10
|
||||||
@@ -438,14 +438,14 @@ done:
|
|||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
code := assembleRISCVHelper(t, fn)
|
code := assembleRISCVHelper(t, fn)
|
||||||
// C.ADDI(2) + C.BEQZ(2) + C.ADDI(2) + JALR(4) = 10 (all compress)
|
// C.ADDI(2) + BEQ(4) + C.ADDI(2) + JALR(4) = 12
|
||||||
if len(code) != 10 {
|
if len(code) != 12 {
|
||||||
t.Errorf("expected 10 bytes with C.BEQZ, got %d", len(code))
|
t.Errorf("expected 12 bytes with uncompressed BEQ, got %d", len(code))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRISCV_RVC_CJ(t *testing.T) {
|
func TestRISCV_RVC_CJ(t *testing.T) {
|
||||||
// JMP target → C.J when offset fits.
|
// JMP target → JAL X0 (never compressed to C.J), matching go tool asm.
|
||||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
TEXT ·cj(SB), NOSPLIT, $0
|
TEXT ·cj(SB), NOSPLIT, $0
|
||||||
JMP done
|
JMP done
|
||||||
@@ -453,9 +453,9 @@ func TestRISCV_RVC_CJ(t *testing.T) {
|
|||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
code := assembleRISCVHelper(t, fn)
|
code := assembleRISCVHelper(t, fn)
|
||||||
// C.J(2) + JALR(4) = 6
|
// JAL(4) + JALR(4) = 8
|
||||||
if len(code) != 6 {
|
if len(code) != 8 {
|
||||||
t.Errorf("expected 6 bytes with C.J, got %d", len(code))
|
t.Errorf("expected 8 bytes with uncompressed JMP, got %d", len(code))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Vendored
+35
@@ -0,0 +1,35 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·branches(SB), NOSPLIT, $0
|
||||||
|
ADDI $1, X10, X10
|
||||||
|
BEQ X10, X11, beq_done
|
||||||
|
ADDI $2, X10, X10
|
||||||
|
beq_done:
|
||||||
|
BNE X10, X11, bne_done
|
||||||
|
ADDI $3, X10, X10
|
||||||
|
bne_done:
|
||||||
|
BLT X10, X11, blt_done
|
||||||
|
ADDI $4, X10, X10
|
||||||
|
blt_done:
|
||||||
|
BGE X10, X11, bge_done
|
||||||
|
ADDI $5, X10, X10
|
||||||
|
bge_done:
|
||||||
|
BLTU X10, X11, bltu_done
|
||||||
|
ADDI $6, X10, X10
|
||||||
|
bltu_done:
|
||||||
|
BGEU X10, X11, bgeu_done
|
||||||
|
ADDI $7, X10, X10
|
||||||
|
bgeu_done:
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·jumps(SB), NOSPLIT, $0
|
||||||
|
JMP done
|
||||||
|
ADDI $1, X10, X10
|
||||||
|
done:
|
||||||
|
JAL X11, skip
|
||||||
|
ADDI $2, X10, X10
|
||||||
|
skip:
|
||||||
|
RET
|
||||||
@@ -23,6 +23,7 @@ func TestGroundTruthRISCV(t *testing.T) {
|
|||||||
"../testdata/verify/loadstore_riscv64.s",
|
"../testdata/verify/loadstore_riscv64.s",
|
||||||
"../testdata/verify/largeimm_riscv64.s",
|
"../testdata/verify/largeimm_riscv64.s",
|
||||||
"../testdata/verify/movimm_riscv64.s",
|
"../testdata/verify/movimm_riscv64.s",
|
||||||
|
"../testdata/verify/branch_riscv64.s",
|
||||||
} {
|
} {
|
||||||
t.Run(path, func(t *testing.T) {
|
t.Run(path, func(t *testing.T) {
|
||||||
testGroundTruthRISCVFile(t, path)
|
testGroundTruthRISCVFile(t, path)
|
||||||
|
|||||||
Reference in New Issue
Block a user