feat(riscv64,loong64): encode AMO atomics, vector slices and bit ops
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+125
-27
@@ -141,10 +141,36 @@ func riscvRegNum(name string) int {
|
||||
case "F31", "FT11":
|
||||
return 31
|
||||
default:
|
||||
// Vector registers V0-V31 (the "V" extension). They share the
|
||||
// register numbering with the integer file: a bare number 0-31.
|
||||
if len(name) >= 2 && name[0] == 'V' {
|
||||
if n, ok := parseRegDigits(name[1:], 31); ok {
|
||||
return n
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
}
|
||||
|
||||
// parseRegDigits parses a decimal register suffix and reports whether it is
|
||||
// within [0, max].
|
||||
func parseRegDigits(digits string, max int) (int, bool) {
|
||||
if digits == "" {
|
||||
return 0, false
|
||||
}
|
||||
n := 0
|
||||
for i := 0; i < len(digits); i++ {
|
||||
if digits[i] < '0' || digits[i] > '9' {
|
||||
return 0, false
|
||||
}
|
||||
n = n*10 + int(digits[i]-'0')
|
||||
if n > max {
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
return n, true
|
||||
}
|
||||
|
||||
// RISC-V instruction encoding parameters.
|
||||
type riscvEnc struct {
|
||||
opcode uint32 // bits [6:0]
|
||||
@@ -232,25 +258,28 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"JALR": {0x67, 0x0, 0x00},
|
||||
|
||||
// RV64A, atomics (AMO opcode 0x2F).
|
||||
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
|
||||
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
|
||||
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
|
||||
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
|
||||
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
|
||||
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
|
||||
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
|
||||
"AMOORW": {0x2F, 0x2, 0x06 << 2},
|
||||
"AMOORD": {0x2F, 0x3, 0x06 << 2},
|
||||
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
|
||||
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
|
||||
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
|
||||
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
|
||||
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
|
||||
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
|
||||
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
|
||||
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
|
||||
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
|
||||
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
|
||||
// funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
|
||||
// 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
|
||||
// the lower two, exactly as the toolchain writes them: every AMO sets both
|
||||
// aq and rl (funct7 |= 3).
|
||||
"AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
|
||||
"AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
|
||||
"AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
|
||||
"AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
|
||||
"AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
|
||||
"AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
|
||||
"AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
|
||||
"AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
|
||||
"AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
|
||||
"AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
|
||||
"AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
|
||||
"AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
|
||||
"AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
|
||||
"AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
|
||||
"AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
|
||||
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
|
||||
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
|
||||
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
|
||||
|
||||
// RV64F/D, floating-point arithmetic.
|
||||
"FADDS": {0x53, 0x0, 0x00},
|
||||
@@ -273,12 +302,16 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"FMAXS": {0x53, 0x1, 0x14},
|
||||
"FMIND": {0x53, 0x0, 0x15},
|
||||
"FMAXD": {0x53, 0x1, 0x15},
|
||||
// FP sign injection (double): rs2 carries the sign source.
|
||||
"FSGNJD": {0x53, 0x0, 0x11},
|
||||
|
||||
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||
"LRW": {0x2F, 0x2, 0x02 << 2},
|
||||
"LRD": {0x2F, 0x3, 0x02 << 2},
|
||||
"SCW": {0x2F, 0x2, 0x03 << 2},
|
||||
"SCD": {0x2F, 0x3, 0x03 << 2},
|
||||
// The toolchain gives LR acquire ordering (aq = 1) and SC release
|
||||
// ordering (rl = 1).
|
||||
"LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
|
||||
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
|
||||
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
|
||||
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
|
||||
|
||||
// FP compare, result in integer register (funct7 0x50/0x51).
|
||||
"FEQS": {0x53, 0x2, 0x50},
|
||||
@@ -296,11 +329,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||
}
|
||||
|
||||
// riscvAMOType encodes an atomic (AMO) instruction.
|
||||
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
|
||||
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
|
||||
// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
|
||||
// funct5 in its upper five bits and the aq/rl ordering bits in the lower two
|
||||
// (the table stores the full field, so the word needs no reassembly).
|
||||
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
|
||||
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
@@ -441,6 +474,71 @@ func riscvJType(rd int, offset int32) uint32 {
|
||||
0x6F // JAL opcode
|
||||
}
|
||||
|
||||
// ---- RVV ("V" extension) encoding helpers ----
|
||||
|
||||
// The OP-V major opcode and its funct3 subclasses.
|
||||
const (
|
||||
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
|
||||
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
|
||||
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
|
||||
riscvVf3VV = 0x0 // vector-vector
|
||||
riscvVf3MV = 0x2 // vector mask
|
||||
riscvVf3VI = 0x3 // vector-immediate
|
||||
riscvVf3VX = 0x4 // vector-scalar
|
||||
riscvVf3Cfg = 0x7 // vsetvli
|
||||
)
|
||||
|
||||
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
|
||||
// group multiplier in [2:0], the selected element width in [5:3] and the
|
||||
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
|
||||
func riscvVType(vsew, vlmul, vta, vma int) int {
|
||||
return vlmul | vsew<<3 | vta<<6 | vma<<7
|
||||
}
|
||||
|
||||
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
|
||||
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
|
||||
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
|
||||
// the toolchain writes above its 10-bit vtype).
|
||||
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
|
||||
imm := vtype & 0x3FF
|
||||
if vsetivli {
|
||||
imm |= 0xC00
|
||||
}
|
||||
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
|
||||
uint32(rd)<<7 | riscvOpV
|
||||
}
|
||||
|
||||
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
|
||||
// segment count in bits [31:29], the addressing mode in bits [28:26], the
|
||||
// unmasked bit at 25 and the width in funct3. width follows the load
|
||||
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
|
||||
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
|
||||
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
|
||||
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
|
||||
uint32(rd)<<7 | op
|
||||
}
|
||||
|
||||
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
|
||||
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
|
||||
// the standard positions. vs1 may name an integer register for the *VX forms
|
||||
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
|
||||
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
|
||||
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
|
||||
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
|
||||
}
|
||||
|
||||
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
|
||||
// fields live where the second source register would be: rs1Field and vs2 are
|
||||
// written verbatim (the oracle writes fixed non-zero constants there for some
|
||||
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
|
||||
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
|
||||
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
|
||||
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
|
||||
}
|
||||
|
||||
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
|
||||
func riscvSegNF(n int) int32 { return int32(n - 1) }
|
||||
|
||||
// ---- RVC (compressed) encoding helpers ----
|
||||
|
||||
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
||||
|
||||
Reference in New Issue
Block a user