feat(riscv64,loong64): encode AMO atomics, vector slices and bit ops

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 06:44:51 +02:00
parent ca3fdce0e0
commit de5d9f358e
12 changed files with 2059 additions and 37 deletions
+422
View File
@@ -321,6 +321,16 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
} }
// The LSX/LASX vector slice and the VMOVQ/XVMOVQ move family, before
// the integer/FP table (their mnemonics overlap the table's 2R format
// but resolve vector-bank registers).
if code, handled, err := encodeLOONG64Vector(instr, mnem, fi); handled {
if err != nil {
return nil, err
}
return code, nil
}
enc, ok := l64InstrTable[mnem] enc, ok := l64InstrTable[mnem]
if !ok { if !ok {
return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem) return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem)
@@ -1490,3 +1500,415 @@ func l64Label(op *ast.Operand) string {
} }
return op.Raw return op.Raw
} }
// ---- LSX/LASX (V*/XV*) vector dispatch ----
// l64VecOperand describes a vector register operand: the 5-bit register
// number, its bank and an optional width or element suffix (V0.B16,
// V1.V[0], X3.WU[2]). The parser hands suffixed operands over verbatim
// (the element index survives only in the raw text), so the suffix is
// scanned from op.Raw.
type l64VecOperand struct {
num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i])
unsig bool // the suffix carries the U marker (.BU[0])
hasSuf bool // any suffix present
}
// l64ParseVecOperand parses a vector register operand with an optional
// width or element suffix. ok reports whether the operand names a vector
// register at all (V or X bank, with or without a suffix).
func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
if op.Kind == ast.OpImmediate {
return v, false
}
name := strings.ReplaceAll(op.Raw, " ", "")
if name == "" || (name[0] != 'V' && name[0] != 'X') {
return v, false
}
i := 1
num := 0
for i < len(name) && name[i] >= '0' && name[i] <= '9' {
num = num*10 + int(name[i]-'0')
if num > 31 {
return v, false
}
i++
}
if i == 1 {
return v, false // no register digits
}
v.num, v.lasx = num, name[0] == 'X'
if i == len(name) {
return v, true
}
if name[i] != '.' || i+2 > len(name) {
return v, false
}
i++
w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
return v, false
}
v.width, v.hasSuf = w, true
i++
if i < len(name) && name[i] == 'U' {
v.unsig = true
i++
}
if i < len(name) && name[i] == '[' {
// Element form .T[i]: the closing bracket ends the operand.
if name[len(name)-1] != ']' || i+2 > len(name)-1 {
return v, false
}
idx := 0
for _, c := range name[i+1 : len(name)-1] {
if c < '0' || c > '9' {
return v, false
}
idx = idx*10 + int(c-'0')
if idx > 31 {
return v, false
}
}
v.elem, v.hasEl = idx, true
return v, true
}
// Width form .T<lanes>: the trailing digits give the lane count.
lanes := 0
if i >= len(name) {
return v, false
}
for ; i < len(name); i++ {
if name[i] < '0' || name[i] > '9' {
return v, false
}
lanes = lanes*10 + int(name[i]-'0')
if lanes > 64 {
return v, false
}
}
v.lanes = lanes
return v, true
}
// l64VecSuffixWidth validates a width suffix against the bank (LSX:
// B16/H8/W4/V2, LASX: B32/H16/W8/V4) and returns the encoded 2-bit width
// selector of vreplgr2vr and vldrepl.
func l64VecSuffixWidth(lasx bool, v l64VecOperand) (int, bool) {
want := map[byte]int{'B': 16, 'H': 8, 'W': 4, 'V': 2}
if lasx {
want = map[byte]int{'B': 32, 'H': 16, 'W': 8, 'V': 4}
}
lanes, ok := want[v.width]
if !ok || lanes != v.lanes {
return 0, false
}
switch v.width {
case 'B':
return 0, true
case 'H':
return 1, true
case 'W':
return 2, true
default:
return 3, true
}
}
// l64VecElementBase validates an element suffix against the bank and
// returns the encoded index field: the index rides in the rk field above a
// per-width base (vpickve2gr/vinsgr2vr give ui4 to .b, ui3 to .h, ui2 to .w
// and ui1 to .d). The LASX bank has no .b/.h element forms: the toolchain
// rejects `XVMOVQ R4, X2.B[0]` and `XVMOVQ X3.B[31], R5`.
func l64VecElementBase(lasx bool, v l64VecOperand) (int, bool) {
limit, base := 0, 0
switch v.width {
case 'B':
if lasx {
return 0, false
}
limit, base = 15, 0
case 'H':
if lasx {
return 0, false
}
limit, base = 7, 16
case 'W':
limit, base = 3, 24
if lasx {
limit, base = 7, 16
}
case 'V':
limit, base = 1, 28
if lasx {
limit, base = 3, 24
}
default:
return 0, false
}
if v.elem > limit {
return 0, false
}
return base + v.elem, true
}
// encodeLOONG64Vector encodes the LSX/LASX mnemonics the table marks as
// vector plus the VMOVQ/XVMOVQ move family. handled reports whether the
// mnemonic belongs to the vector slice; the operand shapes and opcode
// constants reproduce GOARCH=loong64 `go tool asm` exactly.
func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]byte, bool, error) {
if mnem == "VMOVQ" || mnem == "XVMOVQ" {
code, err := encodeLOONG64Vmovq(mnem == "XVMOVQ", instr.Operands, fi)
return code, true, err
}
lasx, ok := l64VecBank[mnem]
if !ok {
return nil, false, nil
}
ops := instr.Operands
bank := "V"
if lasx {
bank = "X"
}
vec := func(op *ast.Operand) (int, error) {
v, isVec := l64ParseVecOperand(op)
if !isVec || v.lasx != lasx || v.hasSuf {
return -1, fmt.Errorf("%s: expected a bare %s0-%s31 vector register, got %q", mnem, bank, bank, op.Raw)
}
return v.num, nil
}
// Two-operand forms (vpcnt.v): INSTR vj, vd.
if l64Vec2R[mnem] {
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vj, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[1])
if err != nil {
return nil, true, err
}
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, vd)), true, nil
}
// Immediate forms: INSTR $imm, vd or INSTR $imm, vj, vd.
if e, imm := l64VecImmInfo[mnem]; imm && len(ops) >= 2 && isImmOperand(ops[0]) {
if len(ops) > 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
imm := int(immFromOperand(ops[0]))
if imm < e.min || imm > e.max {
return nil, true, fmt.Errorf("%s: immediate out of range [%d, %d]", mnem, e.min, e.max)
}
vd, err := vec(ops[len(ops)-1])
if err != nil {
return nil, true, err
}
vj := vd
if len(ops) == 3 {
if vj, err = vec(ops[1]); err != nil {
return nil, true, err
}
}
return l64wordLE(l64irr(e.op, (imm+e.bias)&e.mask, vj, vd)), true, nil
}
// Vector-to-condition forms: INSTR vj, FCCn.
if l64InstrTable[mnem].format == l64Fvcf {
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vj, err := vec(ops[0])
if err != nil {
return nil, true, err
}
if loong64RegClass(operandRegName(ops[1])) != l64ClsFCC {
return nil, true, fmt.Errorf("%s: expected an FCC condition flag, got %q", mnem, ops[1].Raw)
}
fcc := loong64RegNum(operandRegName(ops[1]))
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, fcc)), true, nil
}
// Three-register forms: INSTR vk, vj, vd or INSTR vk, vd (vj = vd).
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
vk, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[len(ops)-1])
if err != nil {
return nil, true, err
}
vj := vd
if len(ops) == 3 {
if vj, err = vec(ops[1]); err != nil {
return nil, true, err
}
}
return l64wordLE(l64rrr(l64InstrTable[mnem].op, vk, vj, vd)), true, nil
}
// encodeLOONG64Vmovq encodes the VMOVQ/XVMOVQ move family. One mnemonic
// covers the whole LSX/LASX transfer surface, dispatched by operand shape
// exactly as the toolchain's table does:
//
// VMOVQ vd, off(rj) vst VMOVQ off(rj), vd vld
// VMOVQ vd, (rj)(rk) vstx VMOVQ (rj)(rk), vd vldx
// VMOVQ off(rj), vd.T vldrepl (load and replicate one element)
// VMOVQ vj, vd vori.b $0 (a register move)
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx]
bank := "V"
if lasx {
bank = "X"
}
if len(ops) != 2 {
return nil, fmt.Errorf("VMOVQ expects 2 operands, got %d", len(ops))
}
src, srcVec := l64ParseVecOperand(ops[0])
dst, dstVec := l64ParseVecOperand(ops[1])
srcMem := isMemOperand(ops[0])
dstMem := isMemOperand(ops[1])
srcIdx := srcMem && ops[0].Addr.Index != ""
dstIdx := dstMem && ops[1].Addr.Index != ""
intReg := func(op *ast.Operand) (int, error) {
if isMemOperand(op) {
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
}
name := operandRegName(op)
if loong64RegClass(name) != l64ClsGR {
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
}
return loong64RegNum(name), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank.
if srcVec && dstVec {
if src.hasSuf || dst.hasSuf {
return nil, fmt.Errorf("VMOVQ: a register move takes bare %s registers", bank)
}
if src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
return l64wordLE(l64rr(enc.move, src.num, dst.num)), nil
}
// Store: VMOVQ vd, off(rj) or VMOVQ vd, (rj)(rk).
if srcVec && dstMem {
if src.hasSuf || src.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected a bare %s0-%s31 register as the stored value", bank, bank)
}
if dstIdx {
rj, rk := loong64RegNum(ops[1].Addr.Base), loong64RegNum(ops[1].Addr.Index)
if rj < 0 || rk < 0 {
return nil, fmt.Errorf("VMOVQ: invalid register operand")
}
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
}
rj, off := l64MemWithFrame(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
}
return l64wordLE(l64irr(enc.st, int(off), rj, src.num)), nil
}
// Load: VMOVQ off(rj), vd, the indexed VMOVQ (rj)(rk), vd, and the
// load-and-replicate form VMOVQ off(rj), vd.T.
if srcMem && dstVec {
if dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
if srcIdx {
if dst.hasSuf {
return nil, fmt.Errorf("VMOVQ: an indexed load takes a bare %s register", bank)
}
rj, rk := loong64RegNum(ops[0].Addr.Base), loong64RegNum(ops[0].Addr.Index)
if rj < 0 || rk < 0 {
return nil, fmt.Errorf("VMOVQ: invalid register operand")
}
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
}
rj, off := l64MemWithFrame(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
}
op := enc.ld
if dst.hasSuf {
w, ok := l64VecSuffixWidth(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
}
switch w {
case 0:
op = enc.replB
case 1:
op = enc.replH
case 2:
op = enc.replW
default:
op = enc.replD
}
}
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
}
// Element extract: VMOVQ vj.T[i], rd (vpickve2gr, signed or unsigned).
if srcVec && src.hasEl && !dstVec && !dstMem {
if src.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
idx, ok := l64VecElementBase(lasx, src)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[0].Raw)
}
rd, err := intReg(ops[1])
if err != nil {
return nil, err
}
op := enc.pickS
if src.unsig {
op = enc.pickU
}
return l64wordLE(l64irr(op, idx, src.num, rd)), nil
}
// Insert and duplicate: VMOVQ rj, vd.T[i] (vinsgr2vr) and
// VMOVQ rj, vd.T (vreplgr2vr).
if !srcVec && !srcMem && dstVec && dst.hasSuf {
if dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
rs, err := intReg(ops[0])
if err != nil {
return nil, err
}
if dst.hasEl {
idx, ok := l64VecElementBase(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[1].Raw)
}
return l64wordLE(l64irr(enc.ins, idx, rs, dst.num)), nil
}
w, ok := l64VecSuffixWidth(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid width suffix %q", ops[1].Raw)
}
return l64wordLE(l64irr(enc.dup, w, rs, dst.num)), nil
}
return nil, fmt.Errorf("VMOVQ: unsupported operand combination %q, %q", ops[0].Raw, ops[1].Raw)
}
+171 -6
View File
@@ -30,7 +30,10 @@ package asm
// of the immediate and register fields), mirroring the toolchain's OP_* // of the immediate and register fields), mirroring the toolchain's OP_*
// helpers, so each l64* function only ORs its fields in. // helpers, so each l64* function only ORs its fields in.
import "maps" import (
"maps"
"strings"
)
// loong64RegNum returns the 5-bit register number for a LoongArch register // loong64RegNum returns the 5-bit register number for a LoongArch register
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition // name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
@@ -103,7 +106,12 @@ func loong64RegNum(name string) int {
case "R31", "S8": case "R31", "S8":
return 31 return 31
} }
// F0-F31, FCC0-FCC7, FCSR0-FCSR31. // F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31,
// X0-X31) are deliberately NOT accepted here: they are a separate
// register class, and the toolchain rejects V/X names wherever an
// integer or FP register is expected (GOARCH=loong64 go tool asm reports
// "unrecognized instruction" for `BEQZ X0`). Vector operands are
// resolved only through loong64VecRegNum.
if len(name) >= 4 && name[:4] == "FCSR" { if len(name) >= 4 && name[:4] == "FCSR" {
return loong64RegSpecial(name[4:], 31) return loong64RegSpecial(name[4:], 31)
} }
@@ -148,6 +156,19 @@ func loong64RegSpecial(digits string, max int) int {
return -1 return -1
} }
// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or
// X0-X31) to its 5-bit number, or -1. The vector banks are a register class
// of their own: the toolchain accepts them only in the vector operands of the
// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1,
// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X
// spellings never reach the integer/FP resolver.
func loong64VecRegNum(name string) int {
if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') {
return -1
}
return loong64RegSpecial(name[1:], 31)
}
// ---- format helpers ---- // ---- format helpers ----
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd. // l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
@@ -247,7 +268,7 @@ const (
l64Firr14 // 2RI14 (ldptr/stptr) l64Firr14 // 2RI14 (ldptr/stptr)
l64Firr16 // 2RI16 (addu16i.d) l64Firr16 // 2RI16 (addu16i.d)
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i) l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub) l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel)
l64Firir // bstrins/bstrpick l64Firir // bstrins/bstrpick
l64Firrr // alsl l64Firrr // alsl
l64Fi15 // syscall/break/dbar l64Fi15 // syscall/break/dbar
@@ -255,6 +276,8 @@ const (
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0]) l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
l64Fshift // 2RI12 with a 5/6-bit shift immediate l64Fshift // 2RI12 with a 5/6-bit shift immediate
l64Fpreld // preld (2RI12 + 5-bit hint) l64Fpreld // preld (2RI12 + 5-bit hint)
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
) )
// l64Enc is one instruction's encoding: its bit layout (format) and the // l64Enc is one instruction's encoding: its bit layout (format) and the
@@ -277,10 +300,68 @@ type l64DualEnc struct {
var l64DualTable = map[string]l64DualEnc{} var l64DualTable = map[string]l64DualEnc{}
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them) // l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered // to their encoding.
// yet; the base integer, memory and floating-point ISA is complete.
var l64InstrTable = map[string]l64Enc{} var l64InstrTable = map[string]l64Enc{}
// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX
// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per
// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and
// `XVADDV X1, X2, X3`, and rejects the crossed spellings.
type l64Vec3Enc struct {
op uint32
lasx bool
}
// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic:
// the opcode, the bank, the accepted immediate range, the bias the toolchain
// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b
// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one).
type l64VecImmEnc struct {
op uint32
lasx bool
min, max int
bias int
mask int
}
// l64VecBank marks the LSX/LASX mnemonics and records which register bank
// each accepts; presence in the map routes the mnemonic through the vector
// dispatcher rather than the integer/FP formats.
var l64VecBank = map[string]bool{}
// l64VecImmInfo mirrors l64VecImmTable for the dispatcher.
var l64VecImmInfo = map[string]l64VecImmEnc{}
// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as
// vpcnt.v).
var l64Vec2R = map[string]bool{}
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
type l64VmovqEnc struct {
ld, st, ldx, stx uint32 // plain and indexed load/store
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
pickS, pickU uint32 // vpickve2gr.{,u} element extract
ins uint32 // vinsgr2vr element insert
dup uint32 // vreplgr2vr duplicate (width in [11:10])
move uint32 // vori.b/xvori.b $0 register move
}
var l64VmovqTable = map[bool]l64VmovqEnc{
false: { // VMOVQ, the LSX (V) bank
ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15,
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
},
true: { // XVMOVQ, the LASX (X) bank
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
},
}
func init() { func init() {
// 3R, integer. // 3R, integer.
rrr := map[string]uint32{ rrr := map[string]uint32{
@@ -360,6 +441,10 @@ func init() {
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10, "FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10, "FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10, "FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
// LSX: convert a 64-bit integer lane to a double float. The operand
// bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`),
// so the entry stays on the 2R integer/FP format.
"FFINTDV": 0x474a << 10,
} }
for m, op := range rr { for m, op := range rr {
l64InstrTable[m] = l64Enc{format: l64Frr, op: op} l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
@@ -416,12 +501,14 @@ func init() {
// LUI is the Plan 9 spelling of lu12i.w. // LUI is the Plan 9 spelling of lu12i.w.
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25} l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
// 4R, fused multiply-add. // 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC
// condition flag, the layout matches the 4R shape).
rrrr := map[string]uint32{ rrrr := map[string]uint32{
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20, "FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20, "FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20, "FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20, "FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
"FSEL": 0x340 << 18,
} }
for m, op := range rrrr { for m, op := range rrrr {
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op} l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
@@ -455,6 +542,10 @@ func init() {
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22} l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result). // Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
// The toolchain's form is three operands, `AMADDW rk, (rj), rd`
// (cmd/asm/internal/asm/testdata/loong64enc1.s and
// internal/runtime/atomic/atomic_loong64.s); the two-register spelling
// is rejected by the oracle.
am := map[string]uint32{ am := map[string]uint32{
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15, "AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15, "AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
@@ -472,10 +563,84 @@ func init() {
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15, "AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15, "AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15, "AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
// The _dbar (acquire/release) add, and, or variants: opcodes read off
// `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends.
"AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15,
"AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15,
"AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15,
} }
for m, op := range am { for m, op := range am {
l64InstrTable[m] = l64Enc{format: l64Fam, op: op} l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
} }
// ---- LSX/LASX (V*/XV*) ----
// Every opcode below was read off `go tool objdump` of a GOARCH=loong64
// `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks
// most of them), not assumed from the LoongArch manual.
// Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with
// vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain
// accepts: LSX spellings take V0-V31, LASX spellings X0-X31.
vec3 := map[string]l64Vec3Enc{
"VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false},
"VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false},
"VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false},
"VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false},
"XVADDV": {0xE817 << 15, true},
"XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true},
"XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true},
}
for m, e := range vec3 {
l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op}
l64VecBank[m] = e.lasx
}
// Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate
// range, bias and field mask are the ones the toolchain encodes: vandi.b
// stores the raw 8-bit constant, vsrai.b stores imm+8 (byte-lane bias),
// vseqi.b and vseqi.d store 5-bit and 7-bit two's-complement values.
// The mnemonics that also have a register form (VSEQB, VSEQV, VSRAB,
// VROTRW) keep their three-register entry in l64InstrTable; the
// dispatcher picks the immediate opcode from l64VecImmInfo by operand
// kind, so the immediate entries must not overwrite the table.
vecImm := map[string]l64VecImmEnc{
"VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF},
"XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF},
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
"VSEQV": {0xE503 << 15, false, -64, 63, 0, 0x7F},
"XVSEQV": {0xE903 << 15, true, -64, 63, 0, 0x7F},
"VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F},
"VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F},
}
for m, e := range vecImm {
l64VecImmInfo[m] = e
l64VecBank[m] = e.lasx
}
// Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*,
// vsetallnez.*): the sub-op rides in the rk field.
vecCf := map[string]uint32{
"VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10,
"VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10,
"VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10,
"VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10,
}
for m, op := range vecCf {
l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op}
l64VecBank[m] = strings.HasPrefix(m, "XV")
}
// Lane popcount: INSTR vj, vd (the 2R layout with the opcode extending
// over the unused vk field).
vec2r := map[string]l64Vec3Enc{
"VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true},
}
for m, e := range vec2r {
l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op}
l64VecBank[m] = e.lasx
l64Vec2R[m] = true
}
} }
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the // l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
+261
View File
@@ -263,6 +263,20 @@ func TestLOONG64_regNames(t *testing.T) {
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want) t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
} }
} }
// The X/V spellings name the LSX/LASX vector banks, a register class of
// their own: the oracle (GOARCH=loong64 go tool asm) rejects `BEQZ X0`
// with "unrecognized instruction" while assembling `VADDV V0, V1, V2`
// and `XVADDV X0, X1, X2`, so loong64RegNum stays strict and the vector
// operands resolve through loong64VecRegNum only.
vecCases := map[string]int{
"V0": 0, "V31": 31, "X0": 0, "X31": 31,
"R4": -1, "F0": -1, "FCC0": -1, "V32": -1, "X32": -1, "V": -1, "X": -1,
}
for name, want := range vecCases {
if got := loong64VecRegNum(name); got != want {
t.Errorf("loong64VecRegNum(%q) = %d, want %d", name, got, want)
}
}
} }
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) { func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
@@ -328,3 +342,250 @@ TEXT ·f(SB), NOSPLIT, $0-0
0x4C000020, // jirl r0, r1, 0 (RET) 0x4C000020, // jirl r0, r1, 0 (RET)
) )
} }
// TestLOONG64_vector pins the LSX/LASX slice against words read off
// GOARCH=loong64 go tool asm (cross-checked against the toolchain's own
// loong64enc1.s): the three-register forms, the immediate forms with their
// biases, the vector-to-condition forms, lane popcount, the FP conversion,
// FSEL and the VMOVQ move family.
func TestLOONG64_vector(t *testing.T) {
t.Run("three-register and immediate forms", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VADDV V1, V2, V3
VADDW V1, V2, V3
VADDV V2, V1
VANDV V1, V2
VXORV V1, V2, V3
VSEQB V1, V2, V3
VSEQV V1, V2, V3
VSRAB V1, V2, V3
VROTRW V1, V2, V3
VANDB $0, V2, V3
VANDB $255, V2
VSEQB $3, V2, V3
VSEQV $15, V2, V3
VSEQV $-15, V2, V3
VSRAB $7, V1, V2
VROTRW $16, V1, V2
VPCNTV V1, V2
XVADDV X1, X2, X3
XVXORV X1, X2, X3
XVSEQB X1, X2, X3
XVPCNTV X1, X2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x700B8443, // vadd.v v3, v2, v1
0x700B0443, // vadd.w
0x700B8821, // vadd.v v1, v1, v2 (two-operand form)
0x71260442, // vand.v v2, v2, v1
0x71270443, // vxor.v
0x70000443, // vseq.b
0x70018443, // vseq.d
0x70EC0443, // vsra.b
0x70EF0443, // vrotr.w
0x73D00043, // vandi.b v3, v2, 0
0x73D3FC42, // vandi.b v2, v2, 255 (two-operand form)
0x72800C43, // vseqi.b v3, v2, 3
0x7281BC43, // vseqi.d v3, v2, 15
0x7281C443, // vseqi.d v3, v2, -15 (7-bit two's complement)
0x73343C22, // vsrai.b v2, v1, 7 (encoded as 7+8)
0x72A0C022, // vrotri.w v2, v1, 16
0x729C2C22, // vpcnt.d v2, v1
0x740B8443, // xvadd.d x3, x2, x1
0x75270443, // xvxor.d
0x74000443, // xvseq.b
0x769C2C22, // xvpcnt.d x2, x1
0x4C000020,
)
})
t.Run("vector-to-condition", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSETNEV V1, FCC0
VSETANYEQB V1, FCC0
VSETANYEQV V2, FCC0
VSETALLNEV V0, FCC0
XVSETNEV X1, FCC0
XVSETALLNEV X1, FCC0
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x729C9C20, // vsetnez.d fcc0, v1
0x729CA020, // vsetanyeqz.b
0x729CAC40, // vsetanyeqz.d
0x729CBC00, // vsetallnez.d
0x769C9C20, // xvsetnez.d
0x769CBC20, // xvsetallnez.d
0x4C000020,
)
})
t.Run("FP convert and FSEL", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
FFINTDV F0, F1
FSEL FCC0, F3, F4, F3
FSEL FCC1, F1, F2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x011D2801, // ffint.d.v f1, f0
0x0D000C83, // fsel f3, f4, f3, fcc0
0x0D008442, // fsel f2, f2, f1, fcc1
0x4C000020,
)
})
t.Run("VMOVQ move family", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VMOVQ V1, V9
VMOVQ (R4), V2
VMOVQ 16(R4), V2
VMOVQ V0, (R4)
VMOVQ V0, 32(R4)
VMOVQ (R4)(R7), V3
VMOVQ V3, (R4)(R7)
VMOVQ R6, V0.B16
VMOVQ R6, V12.W4
VMOVQ (R4), V4.W4
XVMOVQ X3, X7
XVMOVQ (R4), X2
XVMOVQ X0, (R4)
XVMOVQ (R4)(R7), X4
XVMOVQ X0, (R4)(R7)
XVMOVQ R6, X0.B32
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x732D0029, // vori.b v9, v1, 0 (register move)
0x2C000082, // vld v2, r4, 0
0x2C004082, // vld v2, r4, 16
0x2C400080, // vst v0, r4, 0
0x2C408080, // vst v0, r4, 32
0x38401C83, // vldx v3, r4, r7
0x38441C83, // vstx v3, r4, r7
0x729F00C0, // vreplgr2vr.b v0, r6
0x729F08CC, // vreplgr2vr.w v12, r6
0x30200084, // vldrepl.w v4, r4, 0
0x772D0067, // xvori.b x7, x3, 0
0x2C800082, // xvld x2, r4, 0
0x2CC00080, // xvst x0, r4, 0
0x38481C84, // xvldx x4, r4, r7
0x384C1C80, // xvstx x0, r4, r7
0x769F00C0, // xvreplgr2vr.b x0, r6
0x4C000020,
)
})
t.Run("element extract and insert", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VMOVQ V0.V[0], R10
VMOVQ V6.V[1], R8
VMOVQ R9, V1.V[0]
XVMOVQ X0.V[0], R10
XVMOVQ X5.W[7], R7
XVMOVQ R4, X7.V[3]
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x72EFF00A, // vpickve2gr.d r10, v0, 0
0x72EFF4C8, // vpickve2gr.d r8, v6, 1
0x72EBF121, // vinsgr2vr.d v1, r9, 0
0x76EFE00A, // xvpickve2gr.d r10, x0, 0
0x76EFDCA7, // xvpickve2gr.w r7, x5, 7
0x76EBEC87, // xvinsgr2vr.d x7, r4, 3
0x4C000020,
)
})
}
// TestLOONG64_vectorErrors pins the register-class and range diagnostics of
// the vector slice; each shape is rejected by the oracle as well
// (GOARCH=loong64 go tool asm).
func TestLOONG64_vectorErrors(t *testing.T) {
cases := []string{
// Integer registers in vector positions.
`TEXT ·e(SB), NOSPLIT, $0
VADDV R4, R5, R6
RET
`,
// Crossed banks: LSX spellings take V, LASX spellings X.
`TEXT ·e(SB), NOSPLIT, $0
VADDV X1, X2, X3
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVADDV V1, V2, V3
RET
`,
// The LASX bank has no .b/.h element forms.
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ R4, X2.B[0]
RET
`,
// Immediate ranges.
`TEXT ·e(SB), NOSPLIT, $0
VANDB $256, V2
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VSEQB $16, V2, V3
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VROTRW $32, V1, V2
RET
`,
// VSET* wants an FCC flag, not a vector register.
`TEXT ·e(SB), NOSPLIT, $0
VSETNEV V1, V2
RET
`,
}
for i, src := range cases {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_dbarAtomics pins the _dbar (acquire/release) AMO variants.
// The oracle words come from GOARCH=loong64 go tool objdump of kernels
// assembled with go tool asm, and match the toolchain's loong64enc1.s.
func TestLOONG64_dbarAtomics(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·atoms(SB), NOSPLIT, $0
AMADDDBW R14, (R13), R12
AMADDDBV R14, (R13), R12
AMANDDBW R5, (R4), R6
AMANDDBV R5, (R4), R6
AMORDBW R5, (R4), R0
AMORDBV R5, (R4), R6
AMSWAPDBW R5, (R4), R6
AMCASDBV R6, (R4), R5
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x386A39AC, // amadd_db.w r12, r13, r14
0x386AB9AC, // amadd_db.d
0x386B1486, // amand_db.w r6, r4, r5
0x386B9486, // amand_db.d
0x386C1480, // amor_db.w r0, r4, r5
0x386C9486, // amor_db.d
0x38691486, // amswap_db.w
0x385B9885, // amcas_db.w
0x4C000020,
)
}
+4 -1
View File
@@ -232,7 +232,10 @@ DATA ·table+0(SB)/8, $42
} }
// TestLOONG64_errors checks the encoder's error paths: undefined labels, // TestLOONG64_errors checks the encoder's error paths: undefined labels,
// invalid register operands and operand-count mismatches. // invalid register operands and operand-count mismatches. The X0 and
// AMADDW cases follow the oracle: GOARCH=loong64 go tool asm rejects
// `BEQZ X0` (the X bank is not an integer register) and the two-register
// `AMADDW R4, R5` (the AM* family is strictly `val, (addr), result`).
func TestLOONG64_errors(t *testing.T) { func TestLOONG64_errors(t *testing.T) {
cases := []string{ cases := []string{
`TEXT ·e(SB), NOSPLIT, $0 `TEXT ·e(SB), NOSPLIT, $0
+567 -3
View File
@@ -224,6 +224,87 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
} }
return riscvItypeImmediateSize(mnem, imm) return riscvItypeImmediateSize(mnem, imm)
} }
// The toolchain's synthesised instructions: some emit one word, others
// expand to a fixed sequence.
return riscvExtendedSize(mnem, ops)
}
// riscvExtendedSize returns the encoded size of the instructions the
// toolchain synthesises from other instructions (the ternary expansions and
// the vector slice); every caller keeps the layout in step with
// encodeRISCVExtended, which emits exactly these bytes.
func riscvExtendedSize(mnem string, ops []*ast.Operand) int {
switch mnem {
case "NOP":
// The toolchain drops a bare NOP entirely.
return 0
case "ANDN", "ORN":
return 8
case "MAX", "MAXU", "MIN", "MINU":
if riscvIdenticalMinMax(mnem, ops) {
rd := regFromOperand(ops[1])
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rd != 0 {
return 2 // C.MV, or C.LI when the sources are X0
}
return 4
}
return 20
case "ROR", "RORW":
if len(ops) >= 1 && isImmOperand(ops[0]) {
// SRL + [compressed] SLL of the reverse shift + OR.
return 4 + riscvRevShiftSize(mnem, ops) + 4
}
return 16 // SUB + shift + shift + OR
case "RORIW":
return 12
}
return 4
}
// riscvIdenticalMinMax reports whether a MIN/MAX sees two identical source
// registers (the toolchain folds that to ADDI $0).
func riscvIdenticalMinMax(mnem string, ops []*ast.Operand) bool {
if mnem != "MAX" && mnem != "MAXU" && mnem != "MIN" && mnem != "MINU" {
return false
}
if len(ops) != 2 && len(ops) != 3 {
return false
}
rs1 := regFromOperand(ops[1])
rs2 := regFromOperand(ops[0])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 == rd {
// The toolchain swaps the sources so the destination-identical one
// is processed first; identical sources stay identical.
rs1, rs2 = rs2, rs1
}
return rs1 >= 0 && rs1 == rs2
}
// riscvRevShiftSize returns the size of the reverse-shift instruction inside
// a ROR/RORW immediate expansion: the SLLI of the complementary amount, which
// compresses to C.SLLI only in the 64-bit form when rd == rs1, both non-zero,
// and the amount lands in 1-63. The W forms have no compressed shift.
func riscvRevShiftSize(mnem string, ops []*ast.Operand) int {
if mnem != "ROR" {
return 4 // SLLIW has no compressed form
}
imm := int(immFromOperand(ops[0]))
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
sll := (-imm) & 63
if rd == rs1 && rd != 0 && sll >= 1 && sll <= 63 {
return 2 // C.SLLI
}
return 4 return 4
} }
@@ -482,6 +563,16 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
} }
// The toolchain's synthesised instructions and the RVV slice: expanded
// encodings the main table does not carry. FSGNJD is a plain table
// entry and stays with the FP arithmetic path.
if code, handled, err := encodeRISCVExtended(mnem, instr, pc, offsets); handled {
if err != nil {
return nil, err
}
return code, nil
}
enc, ok := riscvInstrTable[mnem] enc, ok := riscvInstrTable[mnem]
if !ok { if !ok {
return nil, fmt.Errorf("unsupported RISC-V instruction %q", mnem) return nil, fmt.Errorf("unsupported RISC-V instruction %q", mnem)
@@ -573,9 +664,11 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
} }
word = riscvSType(enc, rs1, rs2, imm) word = riscvSType(enc, rs1, rs2, imm)
// LR (load-reserved): INSTR (addr), dst, 2 operands. // LR (load-reserved): INSTR (addr), dst. The toolchain reads the
// operands positionally, so the base register comes from the first
// operand and the destination from the second whatever their parens.
case len(ops) == 2 && isLRInstr(mnem): case len(ops) == 2 && isLRInstr(mnem):
rs1, _ := memFromOperandWithFrame(ops[0], fi) rs1 := regFromOperand(ops[0])
rd := regFromOperand(ops[1]) rd := regFromOperand(ops[1])
if rd < 0 || rs1 < 0 { if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem) return nil, fmt.Errorf("invalid operand in %s", mnem)
@@ -1475,6 +1568,477 @@ func extractITypeParams(instr *ast.Instr) (rd, rs1 int, imm int32) {
return return
} }
// ---- toolchain-synthesised instructions and the RVV slice ----
// encodeRISCVExtended encodes the instructions the Go toolchain synthesises
// from other instructions (ANDN/ORN, MIN/MAX, ROR and friends, the branch
// pseudos and FABSD), the CSR read RDTIME, and the RVV vector slice the
// compiler's kernels use. handled reports whether the mnemonic belongs to
// this group; err carries the diagnostic when it does but cannot be encoded.
// Each expansion reproduces the toolchain's instruction-for-instruction
// sequence, including its use of X31 (TMP) and its RVC compression.
func encodeRISCVExtended(mnem string, instr *ast.Instr, pc int, offsets map[string]int) ([]byte, bool, error) {
ops := instr.Operands
switch mnem {
case "NOP":
if len(ops) != 0 {
return nil, true, fmt.Errorf("NOP takes no operands")
}
// The toolchain drops a bare NOP: no bytes at all.
return nil, true, nil
case "RDTIME":
// RDTIME rd reads the time CSR through CSRRS with a zero source.
if len(ops) != 1 {
return nil, true, fmt.Errorf("RDTIME expects 1 operand, got %d", len(ops))
}
rd := regFromOperand(ops[0])
if rd < 0 {
return nil, true, fmt.Errorf("RDTIME: invalid register")
}
return wordLE(riscvIType(riscvEnc{0x73, 0x2, 0x00}, rd, 0, 0xC01)), true, nil
case "NEG", "NOT", "SEQZ":
if len(ops) != 1 && len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops))
}
rs := regFromOperand(ops[0])
rd := rs
if len(ops) == 2 {
rd = regFromOperand(ops[1])
}
if rs < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
var word uint32
switch mnem {
case "NEG":
word = riscvRType(riscvInstrTable["SUB"], rd, 0, rs)
case "NOT":
word = riscvIType(riscvInstrTable["XORI"], rd, rs, -1)
case "SEQZ":
word = riscvIType(riscvInstrTable["SLTIU"], rd, rs, 1)
}
return wordLE(word), true, nil
case "ANDN", "ORN":
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
rs2 := regFromOperand(ops[0]) // the operand to invert
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
notReg := rd
if rs1 == notReg {
notReg = 31 // TMP, when the destination would be clobbered
}
out := wordLE(riscvIType(riscvInstrTable["XORI"], notReg, rs2, -1))
op := riscvInstrTable["AND"]
if mnem == "ORN" {
op = riscvInstrTable["OR"]
}
return append(out, wordLE(riscvRType(op, rd, rs1, notReg))...), true, nil
case "MAX", "MAXU", "MIN", "MINU":
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
if rs1 == rd {
// Process the destination-identical source first, as the
// toolchain does, so the sequence stays in place.
rs1, rs2 = rs2, rs1
}
if rs1 == rs2 {
// Identical inputs fold to ADDI $0 (compressed to C.MV and
// friends by the toolchain's compressor).
return riscvFoldedMove(rd, rs1), true, nil
}
slt1, slt2 := rs2, rs1
cmp := riscvInstrTable["SLT"]
if mnem == "MAX" || mnem == "MAXU" {
slt1, slt2 = slt2, slt1
}
if mnem == "MAXU" || mnem == "MINU" {
cmp = riscvInstrTable["SLTU"]
}
var out []byte
out = append(out, wordLE(riscvRType(cmp, 31, slt1, slt2))...) // the compare into TMP
out = append(out, wordLE(riscvRType(riscvInstrTable["SUB"], 31, 0, 31))...) // NEG TMP
out = append(out, wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rs2))...)
out = append(out, wordLE(riscvRType(riscvInstrTable["AND"], rd, 31, rd))...)
out = append(out, wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rd))...)
return out, true, nil
case "ROR", "RORW", "RORIW":
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
if isImmOperand(ops[0]) {
// Immediate rotate: SRLI the amount, SLLI the complement, OR.
imm := int(immFromOperand(ops[0]))
shiftW := 63
srlEnc := riscvInstrTable["SRLI"]
sllEnc := riscvInstrTable["SLLI"]
if mnem != "ROR" {
shiftW = 31
srlEnc = riscvInstrTable["SRLIW"]
sllEnc = riscvInstrTable["SLLIW"]
}
if imm < 0 || imm > shiftW {
return nil, true, fmt.Errorf("%s: shift amount out of range [0, %d]", mnem, shiftW)
}
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
var out []byte
out = append(out, wordLE(riscvRType(srlEnc, 31, rs1, imm))...)
sll := (-imm) & shiftW
if mnem == "ROR" && rd == rs1 && rd != 0 && sll >= 1 && sll <= 63 {
out = append(out, word16(rvcSLLI(uint32(rd), uint32(sll)))...) // C.SLLI
} else {
out = append(out, wordLE(riscvRType(sllEnc, rd, rs1, sll))...)
}
return append(out, wordLE(riscvRType(riscvInstrTable["OR"], rd, 31, rd))...), true, nil
}
// Register rotate: OR of the two opposite shifts through TMP.
if mnem == "RORIW" {
return nil, true, fmt.Errorf("RORIW takes an immediate shift amount")
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
sllEnc := riscvInstrTable["SLL"]
srlEnc := riscvInstrTable["SRL"]
if mnem == "RORW" {
sllEnc = riscvInstrTable["SLLW"]
srlEnc = riscvInstrTable["SRLW"]
}
var out []byte
out = append(out, wordLE(riscvRType(riscvInstrTable["SUB"], 31, 0, rs2))...) // NEG
out = append(out, wordLE(riscvRType(sllEnc, 31, rs1, 31))...)
out = append(out, wordLE(riscvRType(srlEnc, rd, rs1, rs2))...)
out = append(out, wordLE(riscvRType(riscvInstrTable["OR"], rd, 31, rd))...)
return out, true, nil
case "BGT", "BGTU", "BLE", "BLEU":
// The reversed conditional branches: BGT a, b, label is BLT b, a.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
a := regFromOperand(ops[0])
b := regFromOperand(ops[1])
if a < 0 || b < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
target := labelFromOperand(ops[2])
targetOff, ok := offsets[target]
if !ok {
return nil, true, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
if err := riscvCheckBranchOffset(target, offset); err != nil {
return nil, true, err
}
var enc riscvEnc
switch mnem {
case "BGT":
enc = riscvEnc{0x63, 0x4, 0x00} // blt b, a
case "BGTU":
enc = riscvEnc{0x63, 0x6, 0x00} // bltu b, a
case "BLE":
enc = riscvEnc{0x63, 0x5, 0x00} // bge b, a
case "BLEU":
enc = riscvEnc{0x63, 0x7, 0x00} // bgeu b, a
}
return wordLE(riscvBType(enc, b, a, offset)), true, nil
case "FABSD":
// FABSD rs, rd is FSGNJX.D (sign XOR, funct3 2) with the source in
if len(ops) != 2 {
return nil, true, fmt.Errorf("FABSD expects 2 operands, got %d", len(ops))
}
rs := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rs < 0 || rd < 0 {
return nil, true, fmt.Errorf("FABSD: invalid register")
}
return wordLE(riscvRType(riscvEnc{0x53, 0x2, 0x11}, rd, rs, rs)), true, nil
default:
return encodeRISCVVector(mnem, ops)
}
}
// riscvFoldedMove emits the ADDI $0, rs, rd the toolchain folds identical
// MIN/MAX inputs into, with the same compression its compressor applies to
// the folded form.
func riscvFoldedMove(rd, rs int) []byte {
switch {
case rd != 0 && rs != 0:
return word16(rvcCR(0x8, uint32(rd), uint32(rs))) // C.MV
case rd == 0 && rs == 0:
return word16(0x0001) // C.NOP
case rs == 0:
return word16(rvcCI(0x2, uint32(rd), 0)) // C.LI rd, $0
default:
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs, 0))
}
}
// encodeRISCVVector encodes the RVV slice GOROOT's kernels use. Registers
// are accepted in either spelling: the vector V registers and the integer
// registers share their 5-bit numbers, and the superset keeps hand-written
// probes simple. handled is always true: every name reaching here is one of
// the vector mnemonics.
func encodeRISCVVector(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
reg := regFromOperand
switch mnem {
case "VSETVLI", "VSETIVLI":
// INSTR avl, vsew, vlmul, vta, vma, rd.
if len(ops) != 6 {
return nil, true, fmt.Errorf("%s expects 6 operands, got %d", mnem, len(ops))
}
avl := 0
if isImmOperand(ops[0]) {
avl = int(immFromOperand(ops[0]))
if avl < 0 || avl > 31 {
return nil, true, fmt.Errorf("%s: avl immediate out of range [0, 31]", mnem)
}
} else {
avl = reg(ops[0])
if avl < 0 {
return nil, true, fmt.Errorf("%s: invalid avl register", mnem)
}
}
if mnem == "VSETIVLI" && !isImmOperand(ops[0]) {
return nil, true, fmt.Errorf("VSETIVLI expects an immediate avl")
}
vsew, err := riscvVTypeToken(operandRegName(ops[1]), "E", map[string]int{"8": 0, "16": 1, "32": 2, "64": 3})
if err != nil {
return nil, true, fmt.Errorf("%s: %w", mnem, err)
}
vlmul, err := riscvVTypeToken(operandRegName(ops[2]), "M", map[string]int{"1": 0, "2": 1, "4": 2, "8": 3, "F8": 5, "F4": 6, "F2": 7})
if err != nil {
return nil, true, fmt.Errorf("%s: %w", mnem, err)
}
vta := 0
switch operandRegName(ops[3]) {
case "TA":
vta = 1
case "TU":
default:
return nil, true, fmt.Errorf("%s: invalid tail policy %q (want TA or TU)", mnem, operandRegName(ops[3]))
}
vma := 0
switch operandRegName(ops[4]) {
case "MA":
vma = 1
case "MU":
default:
return nil, true, fmt.Errorf("%s: invalid mask policy %q (want MA or MU)", mnem, operandRegName(ops[4]))
}
rd := reg(ops[5])
if rd < 0 {
return nil, true, fmt.Errorf("%s: invalid destination register", mnem)
}
// An immediate avl always encodes as vsetivli, even under the
// VSETVLI spelling: the toolchain canonicalises the pair, and
// `VSETVLI $15` and `VSETIVLI $15` come out byte-identical
// (0xcd07f657) from GOARCH=riscv64 go tool asm.
ivli := mnem == "VSETIVLI" || isImmOperand(ops[0])
return wordLE(riscvVSetEnc(ivli, avl, riscvVType(vsew, vlmul, vta, vma), rd)), true, nil
case "VLE8V":
// Unit-stride load: INSTR (base), vd.
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rs1, ok := riscvVecMem(ops[0])
if !ok {
return nil, true, fmt.Errorf("%s: invalid memory operand", mnem)
}
vd := reg(ops[1])
if vd < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
return wordLE(riscvVLSType(0x07, 0, 0, 0, 0, rs1, vd)), true, nil
case "VSE8V", "VSE32V":
// Unit-stride store: INSTR vs3, (base).
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vs3 := reg(ops[0])
rs1, ok := riscvVecMem(ops[1])
if !ok {
return nil, true, fmt.Errorf("%s: invalid memory operand", mnem)
}
if vs3 < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
width := 0
if mnem == "VSE32V" {
width = 6
}
return wordLE(riscvVLSType(0x27, 0, 0, width, 0, rs1, vs3)), true, nil
case "VLSSEG4E32V", "VLSSEG8E32V":
// Constant-stride segmented load: INSTR (base), stride, vd.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rs1, ok := riscvVecMem(ops[0])
if !ok {
return nil, true, fmt.Errorf("%s: invalid memory operand", mnem)
}
rs2 := reg(ops[1])
vd := reg(ops[2])
if rs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid register operand", mnem)
}
nf := 3 // 4 fields
if mnem == "VLSSEG8E32V" {
nf = 7 // 8 fields
}
return wordLE(riscvVLSType(0x07, nf, 2, 6, int32(rs2), rs1, vd)), true, nil
case "VADDVV", "VXORVV", "VMSNEVV":
// Vector-vector: INSTR vs1, vs2, vd.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
vs1, vs2, vd := reg(ops[0]), reg(ops[1]), reg(ops[2])
if vs1 < 0 || vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
funct6 := map[string]int{"VADDVV": 0x00, "VXORVV": 0x0B, "VMSNEVV": 0x19}[mnem]
return wordLE(riscvVVInstr(funct6, riscvVf3VV, int32(vs1), vs2, vd)), true, nil
case "VADDVX", "VMSEQVX":
// Vector-scalar: INSTR rs1, vs2, vd (the scalar in the rs1 field).
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rs1, vs2, vd := reg(ops[0]), reg(ops[1]), reg(ops[2])
if rs1 < 0 || vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid register operand", mnem)
}
funct6 := 0x00
if mnem == "VMSEQVX" {
funct6 = 0x18
}
return wordLE(riscvVVInstr(funct6, riscvVf3VX, int32(rs1), vs2, vd)), true, nil
case "VSLLVI", "VSRLVI":
// Vector-immediate shift: INSTR $uimm, vs2, vd.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
imm := int(immFromOperand(ops[0]))
if imm < 0 || imm > 31 {
return nil, true, fmt.Errorf("%s: immediate out of range [0, 31]", mnem)
}
vs2, vd := reg(ops[1]), reg(ops[2])
if vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
funct6 := 0x25 // vsll.vi
if mnem == "VSRLVI" {
funct6 = 0x28 // vsrl.vi
}
return wordLE(riscvVVInstr(funct6, riscvVf3VI, int32(imm), vs2, vd)), true, nil
case "VFIRSTM":
// vmfirst.m rd, vs2: the unmasked form carries 0x11 in the rs1 field
// and sets the mask bit (funct7 = 0x20 | 1).
if len(ops) != 2 {
return nil, true, fmt.Errorf("VFIRSTM expects 2 operands, got %d", len(ops))
}
vs2, rd := reg(ops[0]), reg(ops[1])
if vs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("VFIRSTM: invalid register operand")
}
return wordLE(riscvVUnaryInstr(0x10, riscvVf3MV, 0x11, vs2, rd)), true, nil
case "VIDV":
// vid.v vd (vs2 must be v0; the unmasked form sets the mask bit).
if len(ops) != 1 {
return nil, true, fmt.Errorf("VIDV expects 1 operand, got %d", len(ops))
}
vd := reg(ops[0])
if vd < 0 {
return nil, true, fmt.Errorf("VIDV: invalid vector register")
}
return wordLE(riscvVUnaryInstr(0x14, riscvVf3MV, 0x11, 0, vd)), true, nil
case "VMV4RV":
// vmv4r.v vd, vs2: whole-register group move.
if len(ops) != 2 {
return nil, true, fmt.Errorf("VMV4RV expects 2 operands, got %d", len(ops))
}
vs2, vd := reg(ops[0]), reg(ops[1])
if vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("VMV4RV: invalid vector register")
}
return wordLE(riscvVUnaryInstr(0x27, 0x3, 0x3, vs2, vd)), true, nil
}
return nil, false, nil
}
// riscvVTypeToken parses a vsetvli configuration token (E8, M8, MF2 and
// friends): the letter prefix selects the field and the suffix its value
// through the given table.
func riscvVTypeToken(name, prefix string, codes map[string]int) (int, error) {
if len(name) <= len(prefix) || name[:len(prefix)] != prefix {
return 0, fmt.Errorf("invalid vtype token %q (want %s<width>)", name, prefix)
}
code, ok := codes[name[len(prefix):]]
if !ok {
return 0, fmt.Errorf("invalid vtype token %q", name)
}
return code, nil
}
// riscvVecMem reads a vector memory operand: a bare base register, the only
// addressing form the vector loads and stores carry. Frame-pseudo bases are
// rejected: the toolchain resolves no frame reference on the vector forms.
func riscvVecMem(op *ast.Operand) (rs1 int, ok bool) {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
return -1, false
}
if op.Addr.Base == "" || op.Addr.Offset != 0 {
return -1, false
}
rs1 = riscvRegNum(op.Addr.Base)
return rs1, rs1 >= 0
}
// Instruction type classifiers. // Instruction type classifiers.
func isRTypeInstr(m string) bool { func isRTypeInstr(m string) bool {
switch m { switch m {
@@ -1547,7 +2111,7 @@ func isFPArithInstr(m string) bool {
switch m { switch m {
case "FADDS", "FSUBS", "FMULS", "FDIVS", case "FADDS", "FSUBS", "FMULS", "FDIVS",
"FADDD", "FSUBD", "FMULD", "FDIVD", "FADDD", "FSUBD", "FMULD", "FDIVD",
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD": "FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD":
return true return true
} }
return false return false
+125 -27
View File
@@ -141,10 +141,36 @@ func riscvRegNum(name string) int {
case "F31", "FT11": case "F31", "FT11":
return 31 return 31
default: default:
// Vector registers V0-V31 (the "V" extension). They share the
// register numbering with the integer file: a bare number 0-31.
if len(name) >= 2 && name[0] == 'V' {
if n, ok := parseRegDigits(name[1:], 31); ok {
return n
}
}
return -1 return -1
} }
} }
// parseRegDigits parses a decimal register suffix and reports whether it is
// within [0, max].
func parseRegDigits(digits string, max int) (int, bool) {
if digits == "" {
return 0, false
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return 0, false
}
n = n*10 + int(digits[i]-'0')
if n > max {
return 0, false
}
}
return n, true
}
// RISC-V instruction encoding parameters. // RISC-V instruction encoding parameters.
type riscvEnc struct { type riscvEnc struct {
opcode uint32 // bits [6:0] opcode uint32 // bits [6:0]
@@ -232,25 +258,28 @@ var riscvInstrTable = map[string]riscvEnc{
"JALR": {0x67, 0x0, 0x00}, "JALR": {0x67, 0x0, 0x00},
// RV64A, atomics (AMO opcode 0x2F). // RV64A, atomics (AMO opcode 0x2F).
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27]. // funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2}, // 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2}, // the lower two, exactly as the toolchain writes them: every AMO sets both
"AMOADDW": {0x2F, 0x2, 0x00 << 2}, // aq and rl (funct7 |= 3).
"AMOADDD": {0x2F, 0x3, 0x00 << 2}, "AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
"AMOANDW": {0x2F, 0x2, 0x0C << 2}, "AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
"AMOANDD": {0x2F, 0x3, 0x0C << 2}, "AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
"AMOORW": {0x2F, 0x2, 0x06 << 2}, "AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
"AMOORD": {0x2F, 0x3, 0x06 << 2}, "AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
"AMOXORW": {0x2F, 0x2, 0x04 << 2}, "AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
"AMOXORD": {0x2F, 0x3, 0x04 << 2}, "AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
"AMOMAXW": {0x2F, 0x2, 0x14 << 2}, "AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
"AMOMAXD": {0x2F, 0x3, 0x14 << 2}, "AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
"AMOMINW": {0x2F, 0x2, 0x10 << 2}, "AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
"AMOMIND": {0x2F, 0x3, 0x10 << 2}, "AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2}, "AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2}, "AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
"AMOMINUW": {0x2F, 0x2, 0x18 << 2}, "AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
"AMOMINUD": {0x2F, 0x3, 0x18 << 2}, "AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
// RV64F/D, floating-point arithmetic. // RV64F/D, floating-point arithmetic.
"FADDS": {0x53, 0x0, 0x00}, "FADDS": {0x53, 0x0, 0x00},
@@ -273,12 +302,16 @@ var riscvInstrTable = map[string]riscvEnc{
"FMAXS": {0x53, 0x1, 0x14}, "FMAXS": {0x53, 0x1, 0x14},
"FMIND": {0x53, 0x0, 0x15}, "FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15}, "FMAXD": {0x53, 0x1, 0x15},
// FP sign injection (double): rs2 carries the sign source.
"FSGNJD": {0x53, 0x0, 0x11},
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03). // RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
"LRW": {0x2F, 0x2, 0x02 << 2}, // The toolchain gives LR acquire ordering (aq = 1) and SC release
"LRD": {0x2F, 0x3, 0x02 << 2}, // ordering (rl = 1).
"SCW": {0x2F, 0x2, 0x03 << 2}, "LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
"SCD": {0x2F, 0x3, 0x03 << 2}, "LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
// FP compare, result in integer register (funct7 0x50/0x51). // FP compare, result in integer register (funct7 0x50/0x51).
"FEQS": {0x53, 0x2, 0x50}, "FEQS": {0x53, 0x2, 0x50},
@@ -296,11 +329,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
} }
// riscvAMOType encodes an atomic (AMO) instruction. // riscvAMOType encodes an atomic (AMO) instruction.
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode. // Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2). // funct5 in its upper five bits and the aq/rl ordering bits in the lower two
// (the table stores the full field, so the word needs no reassembly).
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 { func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode (enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
} }
@@ -441,6 +474,71 @@ func riscvJType(rd int, offset int32) uint32 {
0x6F // JAL opcode 0x6F // JAL opcode
} }
// ---- RVV ("V" extension) encoding helpers ----
// The OP-V major opcode and its funct3 subclasses.
const (
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
riscvVf3VV = 0x0 // vector-vector
riscvVf3MV = 0x2 // vector mask
riscvVf3VI = 0x3 // vector-immediate
riscvVf3VX = 0x4 // vector-scalar
riscvVf3Cfg = 0x7 // vsetvli
)
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
// group multiplier in [2:0], the selected element width in [5:3] and the
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
func riscvVType(vsew, vlmul, vta, vma int) int {
return vlmul | vsew<<3 | vta<<6 | vma<<7
}
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
// the toolchain writes above its 10-bit vtype).
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
imm := vtype & 0x3FF
if vsetivli {
imm |= 0xC00
}
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
uint32(rd)<<7 | riscvOpV
}
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
// segment count in bits [31:29], the addressing mode in bits [28:26], the
// unmasked bit at 25 and the width in funct3. width follows the load
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
uint32(rd)<<7 | op
}
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
// the standard positions. vs1 may name an integer register for the *VX forms
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
}
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
// fields live where the second source register would be: rs1Field and vs2 are
// written verbatim (the oracle writes fixed non-zero constants there for some
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
}
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
func riscvSegNF(n int) int32 { return int32(n - 1) }
// ---- RVC (compressed) encoding helpers ---- // ---- RVC (compressed) encoding helpers ----
// isRVCIntReg reports whether a register number can be encoded in the 3-bit // isRVCIntReg reports whether a register number can be encoded in the 3-bit
+223
View File
@@ -5,6 +5,8 @@ package asm
import ( import (
"bytes" "bytes"
"encoding/binary"
"encoding/hex"
"strings" "strings"
"testing" "testing"
@@ -971,3 +973,224 @@ TEXT ·edge(SB), NOSPLIT, $0
t.Errorf("int32-span immediates must assemble: %v", err) t.Errorf("int32-span immediates must assemble: %v", err)
} }
} }
// riscvWants decodes code as little-endian words and pins each one; the
// expected values below were read off GOARCH=riscv64 go tool objdump of
// kernels assembled with go tool asm (the toolchain's riscv64.s testdata
// cross-checks the same words).
func riscvWants(t *testing.T, code []byte, want ...uint32) {
t.Helper()
got := make([]uint32, 0, len(code)/4)
for i := 0; i+4 <= len(code); i += 4 {
got = append(got, binary.LittleEndian.Uint32(code[i:]))
}
if len(got) < len(want) {
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
}
// The RET (JALR) ends the sequence; only the pinned prefix is compared.
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// riscvWantsHex pins the exact hex encoding of a function's instruction
// bytes, including any 2-byte compressed instructions in the stream; the
// expected strings were read off GOARCH=riscv64 go tool objdump of kernels
// assembled with go tool asm (the toolchain's riscv64.s testdata
// cross-checks the same words).
func riscvWantsHex(t *testing.T, code []byte, wantHex string) {
t.Helper()
got := hex.EncodeToString(code)
if got != wantHex {
t.Errorf("code = %s, want %s", got, wantHex)
}
}
// TestRISCV_extendedPseudos pins the toolchain-synthesised instructions:
// ANDN/ORN (XORI + AND/OR through the destination or TMP), the five-word
// MIN/MAX expansion, the four-word rotate, ROR's compressed reverse shift
// (C.SLLI when rd == rs1, both non-zero, 1 <= sll <= 63), the identical-
// input MIN/MAX fold to C.MV, FABSD (FSGNJX.D), SEQZ and RDTIME (csrrs with
// the time CSR).
func TestRISCV_extendedPseudos(t *testing.T) {
t.Run("logic and minmax", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·l(SB), NOSPLIT, $0
ANDN X19, X20, X21
ANDN X19, X20
ORN X20, X19
MAX X26, X28, X29
MIN X29, X30, X5
MAX X5, X5
MAX X5, X5, X6
SEQZ X5, X6
NEG X5, X6
NOT X5
RDTIME X5
RET
`)
code := assembleRISCVHelper(t, fn)
// Words 0-10 up to the folded C.MV pair (halfwords 96 82 and 16 83),
// then SEQZ, NEG, NOT and RDTIME.
riscvWantsHex(t, code,
"93caf9ffb37a5a01"+"93cff9ff337afa01"+"934ffaffb3e9f901"+
"b32fae01b30ff041b34eae01b3fedf01b34ede01"+
"b3afee01b30ff041b342df01b3f25f00b3425f00"+
"9682"+"1683"+
"13b31200"+"33035040"+"93c2f2ff"+"f32210c0"+"67800000")
})
t.Run("rotate", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·r(SB), NOSPLIT, $0
ROR X10, X11, X12
ROR X10, X11
ROR $63, X11
RORIW $31, X13, X14
RORIW $1, X14, X15
RORIW $3, X14
RORW X15, X16, X17
RORW $31, X13
RET
`)
code := assembleRISCVHelper(t, fn)
// The third ROR carries the compressed C.SLLI (05 86) in mid-stream.
riscvWantsHex(t, code,
"b30fa040b39ff50133d6a50033e6cf00"+
"b30fa040b39ff501b3d5a500b3e5bf00"+
"93dff5038605b3e5bf00"+
"9bdff6011b97160033e7ef00"+
"9b5f17009b17f701b3e7ff00"+
"9b5f37001b17d70133e7ef00"+
"b30ff040bb1ff801bb58f800b3e81f01"+
"9bdff6019b961600b3e6df00"+"67800000")
})
t.Run("fp and branches", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
FABSD F1, F2
FSGNJD F1, F0, F2
FMADDD F1, F2, F3, F4
FMSUBD F1, F2, F3, F4
FNMSUBD F1, F2, F3, F4
BGT X5, X6, tgt
BLE X5, X6, tgt
BGTU X5, X6, tgt
BLEU X5, X6, tgt
tgt:
RDTIME X5
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWantsHex(t, code,
"53a11022"+"53011022"+"4382201a4782201a4b82201a"+
"63485300635653006364530063725300"+ // blt/bge/bltu/bgeu x6, x5
"f32210c0"+"67800000")
})
}
// TestRISCV_amoWords pins the full AMO family: every AMO carries aq and rl
// (funct7 |= 3), LR is acquire (funct7 |= 2) and SC release (funct7 |= 1),
// exactly as GOARCH=riscv64 go tool asm encodes them.
func TestRISCV_amoWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·amo(SB), NOSPLIT, $0
AMOSWAPW X5, (X6), X7
AMOSWAPD X5, (X6), X7
AMOADDW X5, (X6), X7
AMOADDD X5, (X6), X7
AMOANDW X5, (X6), X7
AMOANDD X5, (X6), X7
AMOORW X5, (X6), X7
AMOORD X5, (X6), X7
AMOXORW X5, (X6), X7
AMOXORD X5, (X6), X7
AMOMAXW X5, (X6), X7
AMOMAXD X5, (X6), X7
AMOMAXUW X5, (X6), X7
AMOMAXUD X5, (X6), X7
AMOMINUW X5, (X6), X7
AMOMINUD X5, (X6), X7
LRW (X5), X6
LRD (X5), X6
SCW X5, (X6), X7
SCD X5, (X6), X7
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x0E5323AF, // amoswap.w
0x0E5333AF, // amoswap.d
0x065323AF, // amoaddd.w
0x065333AF, // amoadd.d
0x665323AF, // amoand.w
0x665333AF, // amoand.d
0x465323AF, // amoor.w
0x465333AF, // amoor.d
0x265323AF, // amoxor.w
0x265333AF, // amoxor.d
0xA65323AF, // amomax.w
0xA65333AF, // amomax.d
0xE65323AF, // amomaxu.w
0xE65333AF, // amomaxu.d
0xC65323AF, // amominu.w
0xC65333AF, // amominu.d
0x1402A32F, // lr.w (aq)
0x1402B32F, // lr.d
0x1A5323AF, // sc.w (rl)
0x1A5333AF, // sc.d
)
}
// TestRISCV_vectorWords pins the RVV slice and the VSET* encodings. The
// toolchain canonicalises an immediate avl to vsetivli even under the
// VSETVLI spelling (`VSETVLI $15` and `VSETIVLI $15` come out byte-
// identical), which is what the 0xC00 bit of the first word carries.
func TestRISCV_vectorWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSETVLI X5, E8, M8, TA, MA, X6
VSETIVLI $4, E32, M1, TA, MA, X0
VSETVLI $15, E32, M1, TA, MA, X12
VADDVV V1, V2, V3
VADDVX X12, V12, V12
VXORVV V8, V16, V24
VMSEQVX X12, V8, V0
VMSNEVV V8, V16, V0
VSLLVI $8, V28, V30
VSRLVI $25, V29, V29
VFIRSTM V0, X6
VIDV V12
VMV4RV V8, V24
VLE8V (X10), V8
VSE8V V24, (X10)
VSE32V V9, (X11)
VLSSEG4E32V (X14), X0, V0
VLSSEG8E32V (X10), X0, V4
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x0C32F357, // vsetvli x6, x5, vtype 0xc3 (E8, M8, TA, MA)
0xCD027057, // vsetivli x0, 4
0xCD07F657, // vsetivli x12, 15: VSETVLI $15 canonicalises to the same word
0x022081D7, // vadd.vv v3, v2, v1
0x02C64657, // vadd.vx v12, v12, x12
0x2F040C57, // vxor.vv v24, v16, v8
0x62864057, // vmseq.vx v0, v8, x12
0x67040057, // vmsne.vv v0, v16, v8
0x97C43F57, // vsll.vi v30, v28, 8
0xA3DCBED7, // vsrl.vi v29, v29, 25
0x4208A357, // vmfirst.m x6, v0
0x5208A657, // vid.v v12
0x9E81BC57, // vmv4r.v v24, v8
0x02050407, // vle8.v v8, (x10)
0x02050C27, // vse8.v v24, (x10)
0x0205E4A7, // vse32.v v9, (x11)
0x6A076007, // vlsseg4e32.v v0, (x14), x0
0xEA056207, // vlsseg8e32.v v4, (x10), x0
)
}
+49
View File
@@ -0,0 +1,49 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the loong64 atomics: the AM* family in its plain
// and _dbar (acquire/release) forms, spelled as the runtime's
// atomic_loong64.s spells them. Every AM* takes three operands:
// value, (address), result.
#include "textflag.h"
TEXT ·plain(SB), NOSPLIT, $0-0
AMSWAPB R14, (R13), R12
AMSWAPH R14, (R13), R12
AMSWAPW R5, (R4), R6
AMSWAPV R5, (R4), R0
AMCASB R14, (R13), R12
AMCASH R6, (R4), R5
AMCASW R6, (R4), R5
AMCASV R6, (R4), R5
AMADDW R5, (R4), R0
AMADDV R14, (R13), R12
AMANDW R5, (R4), R6
AMANDV R5, (R4), R6
AMORW R5, (R4), R0
AMORV R5, (R4), R6
AMXORW R5, (R4), R6
AMXORV R5, (R4), R6
AMMAXW R5, (R4), R6
AMMAXV R5, (R4), R6
AMMINW R5, (R4), R6
AMMINV R5, (R4), R6
AMMAXWU R5, (R4), R6
AMMAXVU R5, (R4), R6
AMMINWU R5, (R4), R6
AMMINVU R5, (R4), R6
RET
TEXT ·dbar(SB), NOSPLIT, $0-0
AMADDDBW R5, (R4), R6
AMADDDBV R5, (R4), R6
AMANDDBW R5, (R6), R0
AMANDDBV R5, (R4), R6
AMORDBW R5, (R6), R0
AMORDBV R5, (R4), R6
AMSWAPDBW R5, (R4), R6
AMSWAPDBV R5, (R4), R0
AMCASDBW R6, (R4), R5
AMCASDBV R6, (R4), R5
RET
+35
View File
@@ -0,0 +1,35 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the riscv64 atomics: the RV64A AMO family and the
// load-reserved / store-conditional pair, in the toolchain's spelling
// (value, (address), result). Both orderings sit in the encodings: the
// table gives every AMO aq and rl, LR acquire and SC release.
#include "textflag.h"
TEXT ·amo(SB), NOSPLIT, $0-0
AMOSWAPW X5, (X6), X7
AMOSWAPD X5, (X6), X7
AMOADDW X5, (X6), X7
AMOADDD X5, (X6), X7
AMOANDW X5, (X6), X7
AMOANDD X5, (X6), X7
AMOORW X5, (X6), X7
AMOORD X5, (X6), X7
AMOXORW X5, (X6), X7
AMOXORD X5, (X6), X7
AMOMAXW X5, (X6), X7
AMOMAXD X5, (X6), X7
AMOMAXUW X5, (X6), X7
AMOMAXUD X5, (X6), X7
AMOMINUW X5, (X6), X7
AMOMINUD X5, (X6), X7
RET
TEXT ·lrsc(SB), NOSPLIT, $0-0
LRW (X5), X6
LRD (X5), X6
SCW X5, (X6), X7
SCD X5, (X6), X7
RET
+64
View File
@@ -0,0 +1,64 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the riscv64 toolchain-synthesised instructions:
// the Zbb-style pseudos the assembler expands instruction-for-instruction
// (ANDN/ORN, MIN/MAX, ROR and friends, the reversed branches, FABSD), the
// CSR read RDTIME and the FP sign-injection and fused-multiply-add forms.
#include "textflag.h"
TEXT ·logic(SB), NOSPLIT, $0-0
ANDN X19, X20, X21
ANDN X19, X20
ANDN X21, X19, X21
ORN X20, X19
ORN X20, X19, X21
MAX X26, X28, X29
MAX X26, X28
MAXU X28, X29, X30
MAXU X28, X29
MIN X29, X30, X5
MIN X29, X30
MINU X30, X5, X6
MINU X30, X5
MAX X5, X5
MAX X5, X5, X6
SEQZ X5, X6
NEG X5, X6
NEG X5
NOT X5
NOT X5, X6
NOP
RET
TEXT ·rotate(SB), NOSPLIT, $0-0
ROR X10, X11, X12
ROR X10, X11
ROR $63, X11
RORIW $31, X13, X14
RORIW $1, X14, X15
RORIW $3, X14
RORW X15, X16, X17
RORW $31, X13
RET
TEXT ·fp(SB), NOSPLIT, $0-0
FABSD F1, F2
FSGNJD F1, F0, F2
FMADDD F1, F2, F3, F4
FMSUBD F1, F2, F3, F4
FNMSUBD F1, F2, F3, F4
FMADDS F1, F2, F3, F4
FNMADDS F1, F2, F3, F4
RET
TEXT ·branches(SB), NOSPLIT, $0-0
BGT X5, X6, tgt
BLE X5, X6, tgt
BGTU X5, X6, tgt
BLEU X5, X6, tgt
tgt:
RDTIME X5
RET
+85
View File
@@ -0,0 +1,85 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the loong64 LSX/LASX slice: every function pairs
// with the same instructions in the go tool asm ground truth.
#include "textflag.h"
TEXT ·threeReg(SB), NOSPLIT, $0-0
VADDV V1, V2, V3
VADDW V1, V2, V3
VADDV V2, V1
VANDV V1, V2, V3
VANDV V1, V2
VXORV V1, V2, V3
VXORV V1, V2
VSEQB V1, V2, V3
VSEQV V1, V2, V3
VSRAB V1, V2, V3
VROTRW V1, V2, V3
VPCNTV V1, V2
XVADDV X1, X2, X3
XVADDV X2, X1
XVANDV X1, X2, X3
XVXORV X1, X2, X3
XVSEQB X1, X2, X3
XVSEQV X1, X2, X3
XVPCNTV X1, X2
RET
TEXT ·immediates(SB), NOSPLIT, $0-0
VANDB $0, V2, V3
VANDB $255, V2
VSEQB $3, V2, V3
VSEQV $15, V2, V3
VSRAB $0, V1, V2
VSRAB $7, V1, V2
VSRAB $6, V1
VROTRW $0, V1, V2
VROTRW $16, V1, V2
VROTRW $16, V1
XVANDB $1, X2, X2
RET
TEXT ·conditions(SB), NOSPLIT, $0-0
VSETNEV V1, FCC0
VSETANYEQB V1, FCC0
VSETANYEQV V2, FCC0
VSETALLNEV V0, FCC0
XVSETNEV X1, FCC0
XVSETANYEQB X1, FCC0
XVSETANYEQV X1, FCC0
XVSETALLNEV X1, FCC0
RET
TEXT ·fpConvert(SB), NOSPLIT, $0-0
FFINTDV F0, F1
FSEL FCC0, F3, F4, F3
FSEL FCC1, F1, F2
RET
TEXT ·memMoves(SB), NOSPLIT, $0-0
VMOVQ V1, V9
VMOVQ (R4), V2
VMOVQ 16(R4), V2
VMOVQ V0, (R4)
VMOVQ V0, 32(R4)
VMOVQ V0,-16(R6)
VMOVQ (R4)(R7), V3
VMOVQ V3, (R4)(R7)
XVMOVQ X3, X7
XVMOVQ (R4), X2
XVMOVQ X0, (R4)
XVMOVQ (R4)(R7), X4
XVMOVQ X0, (R4)(R7)
RET
TEXT ·elements(SB), NOSPLIT, $0-0
VMOVQ R6, V0.B16
VMOVQ R6, V12.W4
XVMOVQ R6, X0.B32
VMOVQ (R4), V4.W4
VMOVQ (R10), V0.W4
XVMOVQ (R4), X0.B32
RET
+53
View File
@@ -0,0 +1,53 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the riscv64 RVV slice: the instructions GOROOT's
// vector kernels use (crypto/internal/fips140/subtle/xor_riscv64.s,
// internal/bytealg and internal/chacha8rand), spelled as they spell them.
#include "textflag.h"
TEXT ·config(SB), NOSPLIT, $0-0
VSETVLI X5, E8, M8, TA, MA, X6
VSETVLI X11, E8, M8, TA, MA, X5
VSETVLI X12, E8, M8, TA, MA, X5
VSETVLI X13, E8, M8, TU, MU, X15
VSETIVLI $4, E32, M1, TA, MA, X0
VSETIVLI $15, E32, M1, TA, MA, X12
VSETVLI $15, E32, M1, TA, MA, X12
VSETVLI X10, E16, M1, TU, MU, X12
VSETVLI X10, E32, M2, TA, MA, X12
VSETVLI X10, E64, M8, TU, MU, X12
VSETIVLI $31, E32, M1, TA, MA, X12
RET
TEXT ·loadsStores(SB), NOSPLIT, $0-0
VLE8V (X10), V8
VLE8V (X11), V16
VLE8V (X12), V16
VIDV V12
VMV4RV V8, V24
VSE8V V24, (X10)
VSE32V V0, (X11)
VSE32V V8, (X11)
VSE32V V15, (X11)
RET
TEXT ·segmented(SB), NOSPLIT, $0-0
VLSSEG4E32V (X14), X0, V0
VLSSEG8E32V (X10), X0, V4
RET
TEXT ·crypto(SB), NOSPLIT, $0-0
VADDVV V20, V4, V4
VADDVV V27, V11, V11
VADDVX X12, V12, V12
VXORVV V8, V16, V24
VXORVV V13, V13, V13
VMSEQVX X12, V8, V0
VMSNEVV V8, V16, V0
VFIRSTM V0, X6
VFIRSTM V0, X7
VSLLVI $8, V28, V30
VSRLVI $25, V29, V29
RET