feat(riscv64,loong64): encode AMO atomics, vector slices and bit ops

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 06:44:51 +02:00
parent ca3fdce0e0
commit de5d9f358e
12 changed files with 2059 additions and 37 deletions
+125 -27
View File
@@ -141,10 +141,36 @@ func riscvRegNum(name string) int {
case "F31", "FT11":
return 31
default:
// Vector registers V0-V31 (the "V" extension). They share the
// register numbering with the integer file: a bare number 0-31.
if len(name) >= 2 && name[0] == 'V' {
if n, ok := parseRegDigits(name[1:], 31); ok {
return n
}
}
return -1
}
}
// parseRegDigits parses a decimal register suffix and reports whether it is
// within [0, max].
func parseRegDigits(digits string, max int) (int, bool) {
if digits == "" {
return 0, false
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return 0, false
}
n = n*10 + int(digits[i]-'0')
if n > max {
return 0, false
}
}
return n, true
}
// RISC-V instruction encoding parameters.
type riscvEnc struct {
opcode uint32 // bits [6:0]
@@ -232,25 +258,28 @@ var riscvInstrTable = map[string]riscvEnc{
"JALR": {0x67, 0x0, 0x00},
// RV64A, atomics (AMO opcode 0x2F).
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
"AMOORW": {0x2F, 0x2, 0x06 << 2},
"AMOORD": {0x2F, 0x3, 0x06 << 2},
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
// funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
// 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
// the lower two, exactly as the toolchain writes them: every AMO sets both
// aq and rl (funct7 |= 3).
"AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
"AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
"AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
"AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
"AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
"AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
"AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
"AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
"AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
"AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
"AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
"AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
"AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
"AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
"AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
// RV64F/D, floating-point arithmetic.
"FADDS": {0x53, 0x0, 0x00},
@@ -273,12 +302,16 @@ var riscvInstrTable = map[string]riscvEnc{
"FMAXS": {0x53, 0x1, 0x14},
"FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15},
// FP sign injection (double): rs2 carries the sign source.
"FSGNJD": {0x53, 0x0, 0x11},
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
"LRW": {0x2F, 0x2, 0x02 << 2},
"LRD": {0x2F, 0x3, 0x02 << 2},
"SCW": {0x2F, 0x2, 0x03 << 2},
"SCD": {0x2F, 0x3, 0x03 << 2},
// The toolchain gives LR acquire ordering (aq = 1) and SC release
// ordering (rl = 1).
"LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
// FP compare, result in integer register (funct7 0x50/0x51).
"FEQS": {0x53, 0x2, 0x50},
@@ -296,11 +329,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
}
// riscvAMOType encodes an atomic (AMO) instruction.
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
// funct5 in its upper five bits and the aq/rl ordering bits in the lower two
// (the table stores the full field, so the word needs no reassembly).
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
}
@@ -441,6 +474,71 @@ func riscvJType(rd int, offset int32) uint32 {
0x6F // JAL opcode
}
// ---- RVV ("V" extension) encoding helpers ----
// The OP-V major opcode and its funct3 subclasses.
const (
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
riscvVf3VV = 0x0 // vector-vector
riscvVf3MV = 0x2 // vector mask
riscvVf3VI = 0x3 // vector-immediate
riscvVf3VX = 0x4 // vector-scalar
riscvVf3Cfg = 0x7 // vsetvli
)
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
// group multiplier in [2:0], the selected element width in [5:3] and the
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
func riscvVType(vsew, vlmul, vta, vma int) int {
return vlmul | vsew<<3 | vta<<6 | vma<<7
}
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
// the toolchain writes above its 10-bit vtype).
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
imm := vtype & 0x3FF
if vsetivli {
imm |= 0xC00
}
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
uint32(rd)<<7 | riscvOpV
}
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
// segment count in bits [31:29], the addressing mode in bits [28:26], the
// unmasked bit at 25 and the width in funct3. width follows the load
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
uint32(rd)<<7 | op
}
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
// the standard positions. vs1 may name an integer register for the *VX forms
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
}
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
// fields live where the second source register would be: rs1Field and vs2 are
// written verbatim (the oracle writes fixed non-zero constants there for some
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
}
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
func riscvSegNF(n int) int32 { return int32(n - 1) }
// ---- RVC (compressed) encoding helpers ----
// isRVCIntReg reports whether a register number can be encoded in the 3-bit