diff --git a/asm/loong64_assemble.go b/asm/loong64_assemble.go index 0617d19..b20d164 100644 --- a/asm/loong64_assemble.go +++ b/asm/loong64_assemble.go @@ -321,6 +321,16 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } + // The LSX/LASX vector slice and the VMOVQ/XVMOVQ move family, before + // the integer/FP table (their mnemonics overlap the table's 2R format + // but resolve vector-bank registers). + if code, handled, err := encodeLOONG64Vector(instr, mnem, fi); handled { + if err != nil { + return nil, err + } + return code, nil + } + enc, ok := l64InstrTable[mnem] if !ok { return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem) @@ -1490,3 +1500,415 @@ func l64Label(op *ast.Operand) string { } return op.Raw } + +// ---- LSX/LASX (V*/XV*) vector dispatch ---- + +// l64VecOperand describes a vector register operand: the 5-bit register +// number, its bank and an optional width or element suffix (V0.B16, +// V1.V[0], X3.WU[2]). The parser hands suffixed operands over verbatim +// (the element index survives only in the raw text), so the suffix is +// scanned from op.Raw. +type l64VecOperand struct { + num int // 5-bit register number + lasx bool // X bank (LASX) rather than V (LSX) + width byte // suffix width letter (B/H/W/V), 0 on a bare register + lanes int // lane count of a width suffix (B16 → 16) + elem int // element index of a .T[i] suffix + hasEl bool // the suffix names an element (.T[i]) + unsig bool // the suffix carries the U marker (.BU[0]) + hasSuf bool // any suffix present +} + +// l64ParseVecOperand parses a vector register operand with an optional +// width or element suffix. ok reports whether the operand names a vector +// register at all (V or X bank, with or without a suffix). +func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) { + if op.Kind == ast.OpImmediate { + return v, false + } + name := strings.ReplaceAll(op.Raw, " ", "") + if name == "" || (name[0] != 'V' && name[0] != 'X') { + return v, false + } + i := 1 + num := 0 + for i < len(name) && name[i] >= '0' && name[i] <= '9' { + num = num*10 + int(name[i]-'0') + if num > 31 { + return v, false + } + i++ + } + if i == 1 { + return v, false // no register digits + } + v.num, v.lasx = num, name[0] == 'X' + if i == len(name) { + return v, true + } + if name[i] != '.' || i+2 > len(name) { + return v, false + } + i++ + w := name[i] + if w != 'B' && w != 'H' && w != 'W' && w != 'V' { + return v, false + } + v.width, v.hasSuf = w, true + i++ + if i < len(name) && name[i] == 'U' { + v.unsig = true + i++ + } + if i < len(name) && name[i] == '[' { + // Element form .T[i]: the closing bracket ends the operand. + if name[len(name)-1] != ']' || i+2 > len(name)-1 { + return v, false + } + idx := 0 + for _, c := range name[i+1 : len(name)-1] { + if c < '0' || c > '9' { + return v, false + } + idx = idx*10 + int(c-'0') + if idx > 31 { + return v, false + } + } + v.elem, v.hasEl = idx, true + return v, true + } + // Width form .T: the trailing digits give the lane count. + lanes := 0 + if i >= len(name) { + return v, false + } + for ; i < len(name); i++ { + if name[i] < '0' || name[i] > '9' { + return v, false + } + lanes = lanes*10 + int(name[i]-'0') + if lanes > 64 { + return v, false + } + } + v.lanes = lanes + return v, true +} + +// l64VecSuffixWidth validates a width suffix against the bank (LSX: +// B16/H8/W4/V2, LASX: B32/H16/W8/V4) and returns the encoded 2-bit width +// selector of vreplgr2vr and vldrepl. +func l64VecSuffixWidth(lasx bool, v l64VecOperand) (int, bool) { + want := map[byte]int{'B': 16, 'H': 8, 'W': 4, 'V': 2} + if lasx { + want = map[byte]int{'B': 32, 'H': 16, 'W': 8, 'V': 4} + } + lanes, ok := want[v.width] + if !ok || lanes != v.lanes { + return 0, false + } + switch v.width { + case 'B': + return 0, true + case 'H': + return 1, true + case 'W': + return 2, true + default: + return 3, true + } +} + +// l64VecElementBase validates an element suffix against the bank and +// returns the encoded index field: the index rides in the rk field above a +// per-width base (vpickve2gr/vinsgr2vr give ui4 to .b, ui3 to .h, ui2 to .w +// and ui1 to .d). The LASX bank has no .b/.h element forms: the toolchain +// rejects `XVMOVQ R4, X2.B[0]` and `XVMOVQ X3.B[31], R5`. +func l64VecElementBase(lasx bool, v l64VecOperand) (int, bool) { + limit, base := 0, 0 + switch v.width { + case 'B': + if lasx { + return 0, false + } + limit, base = 15, 0 + case 'H': + if lasx { + return 0, false + } + limit, base = 7, 16 + case 'W': + limit, base = 3, 24 + if lasx { + limit, base = 7, 16 + } + case 'V': + limit, base = 1, 28 + if lasx { + limit, base = 3, 24 + } + default: + return 0, false + } + if v.elem > limit { + return 0, false + } + return base + v.elem, true +} + +// encodeLOONG64Vector encodes the LSX/LASX mnemonics the table marks as +// vector plus the VMOVQ/XVMOVQ move family. handled reports whether the +// mnemonic belongs to the vector slice; the operand shapes and opcode +// constants reproduce GOARCH=loong64 `go tool asm` exactly. +func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]byte, bool, error) { + if mnem == "VMOVQ" || mnem == "XVMOVQ" { + code, err := encodeLOONG64Vmovq(mnem == "XVMOVQ", instr.Operands, fi) + return code, true, err + } + lasx, ok := l64VecBank[mnem] + if !ok { + return nil, false, nil + } + ops := instr.Operands + bank := "V" + if lasx { + bank = "X" + } + vec := func(op *ast.Operand) (int, error) { + v, isVec := l64ParseVecOperand(op) + if !isVec || v.lasx != lasx || v.hasSuf { + return -1, fmt.Errorf("%s: expected a bare %s0-%s31 vector register, got %q", mnem, bank, bank, op.Raw) + } + return v.num, nil + } + + // Two-operand forms (vpcnt.v): INSTR vj, vd. + if l64Vec2R[mnem] { + if len(ops) != 2 { + return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + vj, err := vec(ops[0]) + if err != nil { + return nil, true, err + } + vd, err := vec(ops[1]) + if err != nil { + return nil, true, err + } + return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, vd)), true, nil + } + + // Immediate forms: INSTR $imm, vd or INSTR $imm, vj, vd. + if e, imm := l64VecImmInfo[mnem]; imm && len(ops) >= 2 && isImmOperand(ops[0]) { + if len(ops) > 3 { + return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + imm := int(immFromOperand(ops[0])) + if imm < e.min || imm > e.max { + return nil, true, fmt.Errorf("%s: immediate out of range [%d, %d]", mnem, e.min, e.max) + } + vd, err := vec(ops[len(ops)-1]) + if err != nil { + return nil, true, err + } + vj := vd + if len(ops) == 3 { + if vj, err = vec(ops[1]); err != nil { + return nil, true, err + } + } + return l64wordLE(l64irr(e.op, (imm+e.bias)&e.mask, vj, vd)), true, nil + } + + // Vector-to-condition forms: INSTR vj, FCCn. + if l64InstrTable[mnem].format == l64Fvcf { + if len(ops) != 2 { + return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + vj, err := vec(ops[0]) + if err != nil { + return nil, true, err + } + if loong64RegClass(operandRegName(ops[1])) != l64ClsFCC { + return nil, true, fmt.Errorf("%s: expected an FCC condition flag, got %q", mnem, ops[1].Raw) + } + fcc := loong64RegNum(operandRegName(ops[1])) + return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, fcc)), true, nil + } + + // Three-register forms: INSTR vk, vj, vd or INSTR vk, vd (vj = vd). + if len(ops) != 2 && len(ops) != 3 { + return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + vk, err := vec(ops[0]) + if err != nil { + return nil, true, err + } + vd, err := vec(ops[len(ops)-1]) + if err != nil { + return nil, true, err + } + vj := vd + if len(ops) == 3 { + if vj, err = vec(ops[1]); err != nil { + return nil, true, err + } + } + return l64wordLE(l64rrr(l64InstrTable[mnem].op, vk, vj, vd)), true, nil +} + +// encodeLOONG64Vmovq encodes the VMOVQ/XVMOVQ move family. One mnemonic +// covers the whole LSX/LASX transfer surface, dispatched by operand shape +// exactly as the toolchain's table does: +// +// VMOVQ vd, off(rj) vst VMOVQ off(rj), vd vld +// VMOVQ vd, (rj)(rk) vstx VMOVQ (rj)(rk), vd vldx +// VMOVQ off(rj), vd.T vldrepl (load and replicate one element) +// VMOVQ vj, vd vori.b $0 (a register move) +// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register) +// VMOVQ vj.T[i], rd vpickve2gr (extract one element) +// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element) +func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) { + enc := l64VmovqTable[lasx] + bank := "V" + if lasx { + bank = "X" + } + if len(ops) != 2 { + return nil, fmt.Errorf("VMOVQ expects 2 operands, got %d", len(ops)) + } + src, srcVec := l64ParseVecOperand(ops[0]) + dst, dstVec := l64ParseVecOperand(ops[1]) + srcMem := isMemOperand(ops[0]) + dstMem := isMemOperand(ops[1]) + srcIdx := srcMem && ops[0].Addr.Index != "" + dstIdx := dstMem && ops[1].Addr.Index != "" + intReg := func(op *ast.Operand) (int, error) { + if isMemOperand(op) { + return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw) + } + name := operandRegName(op) + if loong64RegClass(name) != l64ClsGR { + return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw) + } + return loong64RegNum(name), nil + } + + // Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant), + // both operands bare registers of the same bank. + if srcVec && dstVec { + if src.hasSuf || dst.hasSuf { + return nil, fmt.Errorf("VMOVQ: a register move takes bare %s registers", bank) + } + if src.lasx != lasx || dst.lasx != lasx { + return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank) + } + return l64wordLE(l64rr(enc.move, src.num, dst.num)), nil + } + + // Store: VMOVQ vd, off(rj) or VMOVQ vd, (rj)(rk). + if srcVec && dstMem { + if src.hasSuf || src.lasx != lasx { + return nil, fmt.Errorf("VMOVQ: expected a bare %s0-%s31 register as the stored value", bank, bank) + } + if dstIdx { + rj, rk := loong64RegNum(ops[1].Addr.Base), loong64RegNum(ops[1].Addr.Index) + if rj < 0 || rk < 0 { + return nil, fmt.Errorf("VMOVQ: invalid register operand") + } + return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil + } + rj, off := l64MemWithFrame(ops[1], fi) + if rj < 0 || off < -2048 || off > 2047 { + return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]") + } + return l64wordLE(l64irr(enc.st, int(off), rj, src.num)), nil + } + + // Load: VMOVQ off(rj), vd, the indexed VMOVQ (rj)(rk), vd, and the + // load-and-replicate form VMOVQ off(rj), vd.T. + if srcMem && dstVec { + if dst.lasx != lasx { + return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank) + } + if srcIdx { + if dst.hasSuf { + return nil, fmt.Errorf("VMOVQ: an indexed load takes a bare %s register", bank) + } + rj, rk := loong64RegNum(ops[0].Addr.Base), loong64RegNum(ops[0].Addr.Index) + if rj < 0 || rk < 0 { + return nil, fmt.Errorf("VMOVQ: invalid register operand") + } + return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil + } + rj, off := l64MemWithFrame(ops[0], fi) + if rj < 0 || off < -2048 || off > 2047 { + return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]") + } + op := enc.ld + if dst.hasSuf { + w, ok := l64VecSuffixWidth(lasx, dst) + if !ok { + return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw) + } + switch w { + case 0: + op = enc.replB + case 1: + op = enc.replH + case 2: + op = enc.replW + default: + op = enc.replD + } + } + return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil + } + + // Element extract: VMOVQ vj.T[i], rd (vpickve2gr, signed or unsigned). + if srcVec && src.hasEl && !dstVec && !dstMem { + if src.lasx != lasx { + return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank) + } + idx, ok := l64VecElementBase(lasx, src) + if !ok { + return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[0].Raw) + } + rd, err := intReg(ops[1]) + if err != nil { + return nil, err + } + op := enc.pickS + if src.unsig { + op = enc.pickU + } + return l64wordLE(l64irr(op, idx, src.num, rd)), nil + } + + // Insert and duplicate: VMOVQ rj, vd.T[i] (vinsgr2vr) and + // VMOVQ rj, vd.T (vreplgr2vr). + if !srcVec && !srcMem && dstVec && dst.hasSuf { + if dst.lasx != lasx { + return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank) + } + rs, err := intReg(ops[0]) + if err != nil { + return nil, err + } + if dst.hasEl { + idx, ok := l64VecElementBase(lasx, dst) + if !ok { + return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[1].Raw) + } + return l64wordLE(l64irr(enc.ins, idx, rs, dst.num)), nil + } + w, ok := l64VecSuffixWidth(lasx, dst) + if !ok { + return nil, fmt.Errorf("VMOVQ: invalid width suffix %q", ops[1].Raw) + } + return l64wordLE(l64irr(enc.dup, w, rs, dst.num)), nil + } + + return nil, fmt.Errorf("VMOVQ: unsupported operand combination %q, %q", ops[0].Raw, ops[1].Raw) +} diff --git a/asm/loong64_encode.go b/asm/loong64_encode.go index 28f10b8..dcb9866 100644 --- a/asm/loong64_encode.go +++ b/asm/loong64_encode.go @@ -30,7 +30,10 @@ package asm // of the immediate and register fields), mirroring the toolchain's OP_* // helpers, so each l64* function only ORs its fields in. -import "maps" +import ( + "maps" + "strings" +) // loong64RegNum returns the 5-bit register number for a LoongArch register // name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition @@ -103,7 +106,12 @@ func loong64RegNum(name string) int { case "R31", "S8": return 31 } - // F0-F31, FCC0-FCC7, FCSR0-FCSR31. + // F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31, + // X0-X31) are deliberately NOT accepted here: they are a separate + // register class, and the toolchain rejects V/X names wherever an + // integer or FP register is expected (GOARCH=loong64 go tool asm reports + // "unrecognized instruction" for `BEQZ X0`). Vector operands are + // resolved only through loong64VecRegNum. if len(name) >= 4 && name[:4] == "FCSR" { return loong64RegSpecial(name[4:], 31) } @@ -148,6 +156,19 @@ func loong64RegSpecial(digits string, max int) int { return -1 } +// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or +// X0-X31) to its 5-bit number, or -1. The vector banks are a register class +// of their own: the toolchain accepts them only in the vector operands of the +// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1, +// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X +// spellings never reach the integer/FP resolver. +func loong64VecRegNum(name string) int { + if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') { + return -1 + } + return loong64RegSpecial(name[1:], 31) +} + // ---- format helpers ---- // l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd. @@ -247,7 +268,7 @@ const ( l64Firr14 // 2RI14 (ldptr/stptr) l64Firr16 // 2RI16 (addu16i.d) l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i) - l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub) + l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel) l64Firir // bstrins/bstrpick l64Firrr // alsl l64Fi15 // syscall/break/dbar @@ -255,6 +276,8 @@ const ( l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0]) l64Fshift // 2RI12 with a 5/6-bit shift immediate l64Fpreld // preld (2RI12 + 5-bit hint) + l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd + l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc ) // l64Enc is one instruction's encoding: its bit layout (format) and the @@ -277,10 +300,68 @@ type l64DualEnc struct { var l64DualTable = map[string]l64DualEnc{} // l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them) -// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered -// yet; the base integer, memory and floating-point ISA is complete. +// to their encoding. var l64InstrTable = map[string]l64Enc{} +// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX +// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per +// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and +// `XVADDV X1, X2, X3`, and rejects the crossed spellings. +type l64Vec3Enc struct { + op uint32 + lasx bool +} + +// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic: +// the opcode, the bank, the accepted immediate range, the bias the toolchain +// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b +// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one). +type l64VecImmEnc struct { + op uint32 + lasx bool + min, max int + bias int + mask int +} + +// l64VecBank marks the LSX/LASX mnemonics and records which register bank +// each accepts; presence in the map routes the mnemonic through the vector +// dispatcher rather than the integer/FP formats. +var l64VecBank = map[string]bool{} + +// l64VecImmInfo mirrors l64VecImmTable for the dispatcher. +var l64VecImmInfo = map[string]l64VecImmEnc{} + +// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as +// vpcnt.v). +var l64Vec2R = map[string]bool{} + +// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit +// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels. +type l64VmovqEnc struct { + ld, st, ldx, stx uint32 // plain and indexed load/store + replB, replH, replW, replD uint32 // vldrepl: load and replicate element + pickS, pickU uint32 // vpickve2gr.{,u} element extract + ins uint32 // vinsgr2vr element insert + dup uint32 // vreplgr2vr duplicate (width in [11:10]) + move uint32 // vori.b/xvori.b $0 register move +} + +var l64VmovqTable = map[bool]l64VmovqEnc{ + false: { // VMOVQ, the LSX (V) bank + ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15, + replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15, + pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15, + ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15, + }, + true: { // XVMOVQ, the LASX (X) bank + ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15, + replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15, + pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15, + ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15, + }, +} + func init() { // 3R, integer. rrr := map[string]uint32{ @@ -360,6 +441,10 @@ func init() { "FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10, "FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10, "FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10, + // LSX: convert a 64-bit integer lane to a double float. The operand + // bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`), + // so the entry stays on the 2R integer/FP format. + "FFINTDV": 0x474a << 10, } for m, op := range rr { l64InstrTable[m] = l64Enc{format: l64Frr, op: op} @@ -416,12 +501,14 @@ func init() { // LUI is the Plan 9 spelling of lu12i.w. l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25} - // 4R, fused multiply-add. + // 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC + // condition flag, the layout matches the 4R shape). rrrr := map[string]uint32{ "FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20, "FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20, "FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20, "FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20, + "FSEL": 0x340 << 18, } for m, op := range rrrr { l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op} @@ -455,6 +542,10 @@ func init() { l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22} // Atomics, 3R with the AM field order (rk=value, rj=address, rd=result). + // The toolchain's form is three operands, `AMADDW rk, (rj), rd` + // (cmd/asm/internal/asm/testdata/loong64enc1.s and + // internal/runtime/atomic/atomic_loong64.s); the two-register spelling + // is rejected by the oracle. am := map[string]uint32{ "AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15, "AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15, @@ -472,10 +563,84 @@ func init() { "AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15, "AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15, "AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15, + // The _dbar (acquire/release) add, and, or variants: opcodes read off + // `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends. + "AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15, + "AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15, + "AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15, } for m, op := range am { l64InstrTable[m] = l64Enc{format: l64Fam, op: op} } + + // ---- LSX/LASX (V*/XV*) ---- + // Every opcode below was read off `go tool objdump` of a GOARCH=loong64 + // `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks + // most of them), not assumed from the LoongArch manual. + + // Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with + // vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain + // accepts: LSX spellings take V0-V31, LASX spellings X0-X31. + vec3 := map[string]l64Vec3Enc{ + "VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false}, + "VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false}, + "VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false}, + "VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false}, + "XVADDV": {0xE817 << 15, true}, + "XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true}, + "XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true}, + } + for m, e := range vec3 { + l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op} + l64VecBank[m] = e.lasx + } + + // Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate + // range, bias and field mask are the ones the toolchain encodes: vandi.b + // stores the raw 8-bit constant, vsrai.b stores imm+8 (byte-lane bias), + // vseqi.b and vseqi.d store 5-bit and 7-bit two's-complement values. + // The mnemonics that also have a register form (VSEQB, VSEQV, VSRAB, + // VROTRW) keep their three-register entry in l64InstrTable; the + // dispatcher picks the immediate opcode from l64VecImmInfo by operand + // kind, so the immediate entries must not overwrite the table. + vecImm := map[string]l64VecImmEnc{ + "VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF}, + "XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF}, + "VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F}, + "XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F}, + "VSEQV": {0xE503 << 15, false, -64, 63, 0, 0x7F}, + "XVSEQV": {0xE903 << 15, true, -64, 63, 0, 0x7F}, + "VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F}, + "VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F}, + } + for m, e := range vecImm { + l64VecImmInfo[m] = e + l64VecBank[m] = e.lasx + } + + // Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*, + // vsetallnez.*): the sub-op rides in the rk field. + vecCf := map[string]uint32{ + "VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10, + "VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10, + "VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10, + "VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10, + } + for m, op := range vecCf { + l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op} + l64VecBank[m] = strings.HasPrefix(m, "XV") + } + + // Lane popcount: INSTR vj, vd (the 2R layout with the opcode extending + // over the unused vk field). + vec2r := map[string]l64Vec3Enc{ + "VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true}, + } + for m, e := range vec2r { + l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op} + l64VecBank[m] = e.lasx + l64Vec2R[m] = true + } } // l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the diff --git a/asm/loong64_encode_test.go b/asm/loong64_encode_test.go index 2cf8bfb..85d5817 100644 --- a/asm/loong64_encode_test.go +++ b/asm/loong64_encode_test.go @@ -263,6 +263,20 @@ func TestLOONG64_regNames(t *testing.T) { t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want) } } + // The X/V spellings name the LSX/LASX vector banks, a register class of + // their own: the oracle (GOARCH=loong64 go tool asm) rejects `BEQZ X0` + // with "unrecognized instruction" while assembling `VADDV V0, V1, V2` + // and `XVADDV X0, X1, X2`, so loong64RegNum stays strict and the vector + // operands resolve through loong64VecRegNum only. + vecCases := map[string]int{ + "V0": 0, "V31": 31, "X0": 0, "X31": 31, + "R4": -1, "F0": -1, "FCC0": -1, "V32": -1, "X32": -1, "V": -1, "X": -1, + } + for name, want := range vecCases { + if got := loong64VecRegNum(name); got != want { + t.Errorf("loong64VecRegNum(%q) = %d, want %d", name, got, want) + } + } } func TestLOONG64_bytesEqualGroundTruth(t *testing.T) { @@ -328,3 +342,250 @@ TEXT ·f(SB), NOSPLIT, $0-0 0x4C000020, // jirl r0, r1, 0 (RET) ) } + +// TestLOONG64_vector pins the LSX/LASX slice against words read off +// GOARCH=loong64 go tool asm (cross-checked against the toolchain's own +// loong64enc1.s): the three-register forms, the immediate forms with their +// biases, the vector-to-condition forms, lane popcount, the FP conversion, +// FSEL and the VMOVQ move family. +func TestLOONG64_vector(t *testing.T) { + t.Run("three-register and immediate forms", func(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·v(SB), NOSPLIT, $0 + VADDV V1, V2, V3 + VADDW V1, V2, V3 + VADDV V2, V1 + VANDV V1, V2 + VXORV V1, V2, V3 + VSEQB V1, V2, V3 + VSEQV V1, V2, V3 + VSRAB V1, V2, V3 + VROTRW V1, V2, V3 + VANDB $0, V2, V3 + VANDB $255, V2 + VSEQB $3, V2, V3 + VSEQV $15, V2, V3 + VSEQV $-15, V2, V3 + VSRAB $7, V1, V2 + VROTRW $16, V1, V2 + VPCNTV V1, V2 + XVADDV X1, X2, X3 + XVXORV X1, X2, X3 + XVSEQB X1, X2, X3 + XVPCNTV X1, X2 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x700B8443, // vadd.v v3, v2, v1 + 0x700B0443, // vadd.w + 0x700B8821, // vadd.v v1, v1, v2 (two-operand form) + 0x71260442, // vand.v v2, v2, v1 + 0x71270443, // vxor.v + 0x70000443, // vseq.b + 0x70018443, // vseq.d + 0x70EC0443, // vsra.b + 0x70EF0443, // vrotr.w + 0x73D00043, // vandi.b v3, v2, 0 + 0x73D3FC42, // vandi.b v2, v2, 255 (two-operand form) + 0x72800C43, // vseqi.b v3, v2, 3 + 0x7281BC43, // vseqi.d v3, v2, 15 + 0x7281C443, // vseqi.d v3, v2, -15 (7-bit two's complement) + 0x73343C22, // vsrai.b v2, v1, 7 (encoded as 7+8) + 0x72A0C022, // vrotri.w v2, v1, 16 + 0x729C2C22, // vpcnt.d v2, v1 + 0x740B8443, // xvadd.d x3, x2, x1 + 0x75270443, // xvxor.d + 0x74000443, // xvseq.b + 0x769C2C22, // xvpcnt.d x2, x1 + 0x4C000020, + ) + }) + + t.Run("vector-to-condition", func(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·v(SB), NOSPLIT, $0 + VSETNEV V1, FCC0 + VSETANYEQB V1, FCC0 + VSETANYEQV V2, FCC0 + VSETALLNEV V0, FCC0 + XVSETNEV X1, FCC0 + XVSETALLNEV X1, FCC0 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x729C9C20, // vsetnez.d fcc0, v1 + 0x729CA020, // vsetanyeqz.b + 0x729CAC40, // vsetanyeqz.d + 0x729CBC00, // vsetallnez.d + 0x769C9C20, // xvsetnez.d + 0x769CBC20, // xvsetallnez.d + 0x4C000020, + ) + }) + + t.Run("FP convert and FSEL", func(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·v(SB), NOSPLIT, $0 + FFINTDV F0, F1 + FSEL FCC0, F3, F4, F3 + FSEL FCC1, F1, F2 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x011D2801, // ffint.d.v f1, f0 + 0x0D000C83, // fsel f3, f4, f3, fcc0 + 0x0D008442, // fsel f2, f2, f1, fcc1 + 0x4C000020, + ) + }) + + t.Run("VMOVQ move family", func(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·v(SB), NOSPLIT, $0 + VMOVQ V1, V9 + VMOVQ (R4), V2 + VMOVQ 16(R4), V2 + VMOVQ V0, (R4) + VMOVQ V0, 32(R4) + VMOVQ (R4)(R7), V3 + VMOVQ V3, (R4)(R7) + VMOVQ R6, V0.B16 + VMOVQ R6, V12.W4 + VMOVQ (R4), V4.W4 + XVMOVQ X3, X7 + XVMOVQ (R4), X2 + XVMOVQ X0, (R4) + XVMOVQ (R4)(R7), X4 + XVMOVQ X0, (R4)(R7) + XVMOVQ R6, X0.B32 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x732D0029, // vori.b v9, v1, 0 (register move) + 0x2C000082, // vld v2, r4, 0 + 0x2C004082, // vld v2, r4, 16 + 0x2C400080, // vst v0, r4, 0 + 0x2C408080, // vst v0, r4, 32 + 0x38401C83, // vldx v3, r4, r7 + 0x38441C83, // vstx v3, r4, r7 + 0x729F00C0, // vreplgr2vr.b v0, r6 + 0x729F08CC, // vreplgr2vr.w v12, r6 + 0x30200084, // vldrepl.w v4, r4, 0 + 0x772D0067, // xvori.b x7, x3, 0 + 0x2C800082, // xvld x2, r4, 0 + 0x2CC00080, // xvst x0, r4, 0 + 0x38481C84, // xvldx x4, r4, r7 + 0x384C1C80, // xvstx x0, r4, r7 + 0x769F00C0, // xvreplgr2vr.b x0, r6 + 0x4C000020, + ) + }) + + t.Run("element extract and insert", func(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·v(SB), NOSPLIT, $0 + VMOVQ V0.V[0], R10 + VMOVQ V6.V[1], R8 + VMOVQ R9, V1.V[0] + XVMOVQ X0.V[0], R10 + XVMOVQ X5.W[7], R7 + XVMOVQ R4, X7.V[3] + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x72EFF00A, // vpickve2gr.d r10, v0, 0 + 0x72EFF4C8, // vpickve2gr.d r8, v6, 1 + 0x72EBF121, // vinsgr2vr.d v1, r9, 0 + 0x76EFE00A, // xvpickve2gr.d r10, x0, 0 + 0x76EFDCA7, // xvpickve2gr.w r7, x5, 7 + 0x76EBEC87, // xvinsgr2vr.d x7, r4, 3 + 0x4C000020, + ) + }) +} + +// TestLOONG64_vectorErrors pins the register-class and range diagnostics of +// the vector slice; each shape is rejected by the oracle as well +// (GOARCH=loong64 go tool asm). +func TestLOONG64_vectorErrors(t *testing.T) { + cases := []string{ + // Integer registers in vector positions. + `TEXT ·e(SB), NOSPLIT, $0 + VADDV R4, R5, R6 + RET +`, + // Crossed banks: LSX spellings take V, LASX spellings X. + `TEXT ·e(SB), NOSPLIT, $0 + VADDV X1, X2, X3 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + XVADDV V1, V2, V3 + RET +`, + // The LASX bank has no .b/.h element forms. + `TEXT ·e(SB), NOSPLIT, $0 + XVMOVQ R4, X2.B[0] + RET +`, + // Immediate ranges. + `TEXT ·e(SB), NOSPLIT, $0 + VANDB $256, V2 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + VSEQB $16, V2, V3 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + VROTRW $32, V1, V2 + RET +`, + // VSET* wants an FCC flag, not a vector register. + `TEXT ·e(SB), NOSPLIT, $0 + VSETNEV V1, V2 + RET +`, + } + for i, src := range cases { + fn := firstTextLOONG64(t, src) + if _, _, _, _, _, err := assembleLOONG64(fn); err == nil { + t.Errorf("case %d: expected an error, got none", i) + } + } +} + +// TestLOONG64_dbarAtomics pins the _dbar (acquire/release) AMO variants. +// The oracle words come from GOARCH=loong64 go tool objdump of kernels +// assembled with go tool asm, and match the toolchain's loong64enc1.s. +func TestLOONG64_dbarAtomics(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·atoms(SB), NOSPLIT, $0 + AMADDDBW R14, (R13), R12 + AMADDDBV R14, (R13), R12 + AMANDDBW R5, (R4), R6 + AMANDDBV R5, (R4), R6 + AMORDBW R5, (R4), R0 + AMORDBV R5, (R4), R6 + AMSWAPDBW R5, (R4), R6 + AMCASDBV R6, (R4), R5 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x386A39AC, // amadd_db.w r12, r13, r14 + 0x386AB9AC, // amadd_db.d + 0x386B1486, // amand_db.w r6, r4, r5 + 0x386B9486, // amand_db.d + 0x386C1480, // amor_db.w r0, r4, r5 + 0x386C9486, // amor_db.d + 0x38691486, // amswap_db.w + 0x385B9885, // amcas_db.w + 0x4C000020, + ) +} diff --git a/asm/loong64_more_test.go b/asm/loong64_more_test.go index 88b9a83..693f51a 100644 --- a/asm/loong64_more_test.go +++ b/asm/loong64_more_test.go @@ -232,7 +232,10 @@ DATA ·table+0(SB)/8, $42 } // TestLOONG64_errors checks the encoder's error paths: undefined labels, -// invalid register operands and operand-count mismatches. +// invalid register operands and operand-count mismatches. The X0 and +// AMADDW cases follow the oracle: GOARCH=loong64 go tool asm rejects +// `BEQZ X0` (the X bank is not an integer register) and the two-register +// `AMADDW R4, R5` (the AM* family is strictly `val, (addr), result`). func TestLOONG64_errors(t *testing.T) { cases := []string{ `TEXT ·e(SB), NOSPLIT, $0 diff --git a/asm/riscv_assemble.go b/asm/riscv_assemble.go index 4a3fb17..4724eff 100644 --- a/asm/riscv_assemble.go +++ b/asm/riscv_assemble.go @@ -224,6 +224,87 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int { } return riscvItypeImmediateSize(mnem, imm) } + // The toolchain's synthesised instructions: some emit one word, others + // expand to a fixed sequence. + return riscvExtendedSize(mnem, ops) +} + +// riscvExtendedSize returns the encoded size of the instructions the +// toolchain synthesises from other instructions (the ternary expansions and +// the vector slice); every caller keeps the layout in step with +// encodeRISCVExtended, which emits exactly these bytes. +func riscvExtendedSize(mnem string, ops []*ast.Operand) int { + switch mnem { + case "NOP": + // The toolchain drops a bare NOP entirely. + return 0 + case "ANDN", "ORN": + return 8 + case "MAX", "MAXU", "MIN", "MINU": + if riscvIdenticalMinMax(mnem, ops) { + rd := regFromOperand(ops[1]) + if len(ops) == 3 { + rd = regFromOperand(ops[2]) + } + if rd != 0 { + return 2 // C.MV, or C.LI when the sources are X0 + } + return 4 + } + return 20 + case "ROR", "RORW": + if len(ops) >= 1 && isImmOperand(ops[0]) { + // SRL + [compressed] SLL of the reverse shift + OR. + return 4 + riscvRevShiftSize(mnem, ops) + 4 + } + return 16 // SUB + shift + shift + OR + case "RORIW": + return 12 + } + return 4 +} + +// riscvIdenticalMinMax reports whether a MIN/MAX sees two identical source +// registers (the toolchain folds that to ADDI $0). +func riscvIdenticalMinMax(mnem string, ops []*ast.Operand) bool { + if mnem != "MAX" && mnem != "MAXU" && mnem != "MIN" && mnem != "MINU" { + return false + } + if len(ops) != 2 && len(ops) != 3 { + return false + } + rs1 := regFromOperand(ops[1]) + rs2 := regFromOperand(ops[0]) + rd := rs1 + if len(ops) == 3 { + rd = regFromOperand(ops[2]) + } + if rs1 == rd { + // The toolchain swaps the sources so the destination-identical one + // is processed first; identical sources stay identical. + rs1, rs2 = rs2, rs1 + } + return rs1 >= 0 && rs1 == rs2 +} + +// riscvRevShiftSize returns the size of the reverse-shift instruction inside +// a ROR/RORW immediate expansion: the SLLI of the complementary amount, which +// compresses to C.SLLI only in the 64-bit form when rd == rs1, both non-zero, +// and the amount lands in 1-63. The W forms have no compressed shift. +func riscvRevShiftSize(mnem string, ops []*ast.Operand) int { + if mnem != "ROR" { + return 4 // SLLIW has no compressed form + } + imm := int(immFromOperand(ops[0])) + rs1 := regFromOperand(ops[1]) + rd := rs1 + if len(ops) == 3 { + rd = regFromOperand(ops[2]) + } + sll := (-imm) & 63 + if rd == rs1 && rd != 0 && sll >= 1 && sll <= 63 { + return 2 // C.SLLI + } return 4 } @@ -482,6 +563,16 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } + // The toolchain's synthesised instructions and the RVV slice: expanded + // encodings the main table does not carry. FSGNJD is a plain table + // entry and stays with the FP arithmetic path. + if code, handled, err := encodeRISCVExtended(mnem, instr, pc, offsets); handled { + if err != nil { + return nil, err + } + return code, nil + } + enc, ok := riscvInstrTable[mnem] if !ok { return nil, fmt.Errorf("unsupported RISC-V instruction %q", mnem) @@ -573,9 +664,11 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv } word = riscvSType(enc, rs1, rs2, imm) - // LR (load-reserved): INSTR (addr), dst, 2 operands. + // LR (load-reserved): INSTR (addr), dst. The toolchain reads the + // operands positionally, so the base register comes from the first + // operand and the destination from the second whatever their parens. case len(ops) == 2 && isLRInstr(mnem): - rs1, _ := memFromOperandWithFrame(ops[0], fi) + rs1 := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) @@ -1475,6 +1568,477 @@ func extractITypeParams(instr *ast.Instr) (rd, rs1 int, imm int32) { return } +// ---- toolchain-synthesised instructions and the RVV slice ---- + +// encodeRISCVExtended encodes the instructions the Go toolchain synthesises +// from other instructions (ANDN/ORN, MIN/MAX, ROR and friends, the branch +// pseudos and FABSD), the CSR read RDTIME, and the RVV vector slice the +// compiler's kernels use. handled reports whether the mnemonic belongs to +// this group; err carries the diagnostic when it does but cannot be encoded. +// Each expansion reproduces the toolchain's instruction-for-instruction +// sequence, including its use of X31 (TMP) and its RVC compression. +func encodeRISCVExtended(mnem string, instr *ast.Instr, pc int, offsets map[string]int) ([]byte, bool, error) { + ops := instr.Operands + switch mnem { + case "NOP": + if len(ops) != 0 { + return nil, true, fmt.Errorf("NOP takes no operands") + } + // The toolchain drops a bare NOP: no bytes at all. + return nil, true, nil + + case "RDTIME": + // RDTIME rd reads the time CSR through CSRRS with a zero source. + if len(ops) != 1 { + return nil, true, fmt.Errorf("RDTIME expects 1 operand, got %d", len(ops)) + } + rd := regFromOperand(ops[0]) + if rd < 0 { + return nil, true, fmt.Errorf("RDTIME: invalid register") + } + return wordLE(riscvIType(riscvEnc{0x73, 0x2, 0x00}, rd, 0, 0xC01)), true, nil + + case "NEG", "NOT", "SEQZ": + if len(ops) != 1 && len(ops) != 2 { + return nil, true, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops)) + } + rs := regFromOperand(ops[0]) + rd := rs + if len(ops) == 2 { + rd = regFromOperand(ops[1]) + } + if rs < 0 || rd < 0 { + return nil, true, fmt.Errorf("%s: invalid register", mnem) + } + var word uint32 + switch mnem { + case "NEG": + word = riscvRType(riscvInstrTable["SUB"], rd, 0, rs) + case "NOT": + word = riscvIType(riscvInstrTable["XORI"], rd, rs, -1) + case "SEQZ": + word = riscvIType(riscvInstrTable["SLTIU"], rd, rs, 1) + } + return wordLE(word), true, nil + + case "ANDN", "ORN": + if len(ops) != 2 && len(ops) != 3 { + return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + rs2 := regFromOperand(ops[0]) // the operand to invert + rs1 := regFromOperand(ops[1]) + rd := rs1 + if len(ops) == 3 { + rd = regFromOperand(ops[2]) + } + if rs1 < 0 || rs2 < 0 || rd < 0 { + return nil, true, fmt.Errorf("%s: invalid register", mnem) + } + notReg := rd + if rs1 == notReg { + notReg = 31 // TMP, when the destination would be clobbered + } + out := wordLE(riscvIType(riscvInstrTable["XORI"], notReg, rs2, -1)) + op := riscvInstrTable["AND"] + if mnem == "ORN" { + op = riscvInstrTable["OR"] + } + return append(out, wordLE(riscvRType(op, rd, rs1, notReg))...), true, nil + + case "MAX", "MAXU", "MIN", "MINU": + if len(ops) != 2 && len(ops) != 3 { + return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + rs2 := regFromOperand(ops[0]) + rs1 := regFromOperand(ops[1]) + rd := rs1 + if len(ops) == 3 { + rd = regFromOperand(ops[2]) + } + if rs1 < 0 || rs2 < 0 || rd < 0 { + return nil, true, fmt.Errorf("%s: invalid register", mnem) + } + if rs1 == rd { + // Process the destination-identical source first, as the + // toolchain does, so the sequence stays in place. + rs1, rs2 = rs2, rs1 + } + if rs1 == rs2 { + // Identical inputs fold to ADDI $0 (compressed to C.MV and + // friends by the toolchain's compressor). + return riscvFoldedMove(rd, rs1), true, nil + } + slt1, slt2 := rs2, rs1 + cmp := riscvInstrTable["SLT"] + if mnem == "MAX" || mnem == "MAXU" { + slt1, slt2 = slt2, slt1 + } + if mnem == "MAXU" || mnem == "MINU" { + cmp = riscvInstrTable["SLTU"] + } + var out []byte + out = append(out, wordLE(riscvRType(cmp, 31, slt1, slt2))...) // the compare into TMP + out = append(out, wordLE(riscvRType(riscvInstrTable["SUB"], 31, 0, 31))...) // NEG TMP + out = append(out, wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rs2))...) + out = append(out, wordLE(riscvRType(riscvInstrTable["AND"], rd, 31, rd))...) + out = append(out, wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rd))...) + return out, true, nil + + case "ROR", "RORW", "RORIW": + if len(ops) != 2 && len(ops) != 3 { + return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + if isImmOperand(ops[0]) { + // Immediate rotate: SRLI the amount, SLLI the complement, OR. + imm := int(immFromOperand(ops[0])) + shiftW := 63 + srlEnc := riscvInstrTable["SRLI"] + sllEnc := riscvInstrTable["SLLI"] + if mnem != "ROR" { + shiftW = 31 + srlEnc = riscvInstrTable["SRLIW"] + sllEnc = riscvInstrTable["SLLIW"] + } + if imm < 0 || imm > shiftW { + return nil, true, fmt.Errorf("%s: shift amount out of range [0, %d]", mnem, shiftW) + } + rs1 := regFromOperand(ops[1]) + rd := rs1 + if len(ops) == 3 { + rd = regFromOperand(ops[2]) + } + if rs1 < 0 || rd < 0 { + return nil, true, fmt.Errorf("%s: invalid register", mnem) + } + var out []byte + out = append(out, wordLE(riscvRType(srlEnc, 31, rs1, imm))...) + sll := (-imm) & shiftW + if mnem == "ROR" && rd == rs1 && rd != 0 && sll >= 1 && sll <= 63 { + out = append(out, word16(rvcSLLI(uint32(rd), uint32(sll)))...) // C.SLLI + } else { + out = append(out, wordLE(riscvRType(sllEnc, rd, rs1, sll))...) + } + return append(out, wordLE(riscvRType(riscvInstrTable["OR"], rd, 31, rd))...), true, nil + } + // Register rotate: OR of the two opposite shifts through TMP. + if mnem == "RORIW" { + return nil, true, fmt.Errorf("RORIW takes an immediate shift amount") + } + rs2 := regFromOperand(ops[0]) + rs1 := regFromOperand(ops[1]) + rd := rs1 + if len(ops) == 3 { + rd = regFromOperand(ops[2]) + } + if rs1 < 0 || rs2 < 0 || rd < 0 { + return nil, true, fmt.Errorf("%s: invalid register", mnem) + } + sllEnc := riscvInstrTable["SLL"] + srlEnc := riscvInstrTable["SRL"] + if mnem == "RORW" { + sllEnc = riscvInstrTable["SLLW"] + srlEnc = riscvInstrTable["SRLW"] + } + var out []byte + out = append(out, wordLE(riscvRType(riscvInstrTable["SUB"], 31, 0, rs2))...) // NEG + out = append(out, wordLE(riscvRType(sllEnc, 31, rs1, 31))...) + out = append(out, wordLE(riscvRType(srlEnc, rd, rs1, rs2))...) + out = append(out, wordLE(riscvRType(riscvInstrTable["OR"], rd, 31, rd))...) + return out, true, nil + + case "BGT", "BGTU", "BLE", "BLEU": + // The reversed conditional branches: BGT a, b, label is BLT b, a. + if len(ops) != 3 { + return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + a := regFromOperand(ops[0]) + b := regFromOperand(ops[1]) + if a < 0 || b < 0 { + return nil, true, fmt.Errorf("%s: invalid register", mnem) + } + target := labelFromOperand(ops[2]) + targetOff, ok := offsets[target] + if !ok { + return nil, true, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) + } + offset := int32(targetOff - pc) + if err := riscvCheckBranchOffset(target, offset); err != nil { + return nil, true, err + } + var enc riscvEnc + switch mnem { + case "BGT": + enc = riscvEnc{0x63, 0x4, 0x00} // blt b, a + case "BGTU": + enc = riscvEnc{0x63, 0x6, 0x00} // bltu b, a + case "BLE": + enc = riscvEnc{0x63, 0x5, 0x00} // bge b, a + case "BLEU": + enc = riscvEnc{0x63, 0x7, 0x00} // bgeu b, a + } + return wordLE(riscvBType(enc, b, a, offset)), true, nil + + case "FABSD": + // FABSD rs, rd is FSGNJX.D (sign XOR, funct3 2) with the source in + if len(ops) != 2 { + return nil, true, fmt.Errorf("FABSD expects 2 operands, got %d", len(ops)) + } + rs := regFromOperand(ops[0]) + rd := regFromOperand(ops[1]) + if rs < 0 || rd < 0 { + return nil, true, fmt.Errorf("FABSD: invalid register") + } + return wordLE(riscvRType(riscvEnc{0x53, 0x2, 0x11}, rd, rs, rs)), true, nil + + default: + return encodeRISCVVector(mnem, ops) + } +} + +// riscvFoldedMove emits the ADDI $0, rs, rd the toolchain folds identical +// MIN/MAX inputs into, with the same compression its compressor applies to +// the folded form. +func riscvFoldedMove(rd, rs int) []byte { + switch { + case rd != 0 && rs != 0: + return word16(rvcCR(0x8, uint32(rd), uint32(rs))) // C.MV + case rd == 0 && rs == 0: + return word16(0x0001) // C.NOP + case rs == 0: + return word16(rvcCI(0x2, uint32(rd), 0)) // C.LI rd, $0 + default: + return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs, 0)) + } +} + +// encodeRISCVVector encodes the RVV slice GOROOT's kernels use. Registers +// are accepted in either spelling: the vector V registers and the integer +// registers share their 5-bit numbers, and the superset keeps hand-written +// probes simple. handled is always true: every name reaching here is one of +// the vector mnemonics. +func encodeRISCVVector(mnem string, ops []*ast.Operand) ([]byte, bool, error) { + reg := regFromOperand + switch mnem { + case "VSETVLI", "VSETIVLI": + // INSTR avl, vsew, vlmul, vta, vma, rd. + if len(ops) != 6 { + return nil, true, fmt.Errorf("%s expects 6 operands, got %d", mnem, len(ops)) + } + avl := 0 + if isImmOperand(ops[0]) { + avl = int(immFromOperand(ops[0])) + if avl < 0 || avl > 31 { + return nil, true, fmt.Errorf("%s: avl immediate out of range [0, 31]", mnem) + } + } else { + avl = reg(ops[0]) + if avl < 0 { + return nil, true, fmt.Errorf("%s: invalid avl register", mnem) + } + } + if mnem == "VSETIVLI" && !isImmOperand(ops[0]) { + return nil, true, fmt.Errorf("VSETIVLI expects an immediate avl") + } + vsew, err := riscvVTypeToken(operandRegName(ops[1]), "E", map[string]int{"8": 0, "16": 1, "32": 2, "64": 3}) + if err != nil { + return nil, true, fmt.Errorf("%s: %w", mnem, err) + } + vlmul, err := riscvVTypeToken(operandRegName(ops[2]), "M", map[string]int{"1": 0, "2": 1, "4": 2, "8": 3, "F8": 5, "F4": 6, "F2": 7}) + if err != nil { + return nil, true, fmt.Errorf("%s: %w", mnem, err) + } + vta := 0 + switch operandRegName(ops[3]) { + case "TA": + vta = 1 + case "TU": + default: + return nil, true, fmt.Errorf("%s: invalid tail policy %q (want TA or TU)", mnem, operandRegName(ops[3])) + } + vma := 0 + switch operandRegName(ops[4]) { + case "MA": + vma = 1 + case "MU": + default: + return nil, true, fmt.Errorf("%s: invalid mask policy %q (want MA or MU)", mnem, operandRegName(ops[4])) + } + rd := reg(ops[5]) + if rd < 0 { + return nil, true, fmt.Errorf("%s: invalid destination register", mnem) + } + // An immediate avl always encodes as vsetivli, even under the + // VSETVLI spelling: the toolchain canonicalises the pair, and + // `VSETVLI $15` and `VSETIVLI $15` come out byte-identical + // (0xcd07f657) from GOARCH=riscv64 go tool asm. + ivli := mnem == "VSETIVLI" || isImmOperand(ops[0]) + return wordLE(riscvVSetEnc(ivli, avl, riscvVType(vsew, vlmul, vta, vma), rd)), true, nil + + case "VLE8V": + // Unit-stride load: INSTR (base), vd. + if len(ops) != 2 { + return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rs1, ok := riscvVecMem(ops[0]) + if !ok { + return nil, true, fmt.Errorf("%s: invalid memory operand", mnem) + } + vd := reg(ops[1]) + if vd < 0 { + return nil, true, fmt.Errorf("%s: invalid vector register", mnem) + } + return wordLE(riscvVLSType(0x07, 0, 0, 0, 0, rs1, vd)), true, nil + + case "VSE8V", "VSE32V": + // Unit-stride store: INSTR vs3, (base). + if len(ops) != 2 { + return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + vs3 := reg(ops[0]) + rs1, ok := riscvVecMem(ops[1]) + if !ok { + return nil, true, fmt.Errorf("%s: invalid memory operand", mnem) + } + if vs3 < 0 { + return nil, true, fmt.Errorf("%s: invalid vector register", mnem) + } + width := 0 + if mnem == "VSE32V" { + width = 6 + } + return wordLE(riscvVLSType(0x27, 0, 0, width, 0, rs1, vs3)), true, nil + + case "VLSSEG4E32V", "VLSSEG8E32V": + // Constant-stride segmented load: INSTR (base), stride, vd. + if len(ops) != 3 { + return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + rs1, ok := riscvVecMem(ops[0]) + if !ok { + return nil, true, fmt.Errorf("%s: invalid memory operand", mnem) + } + rs2 := reg(ops[1]) + vd := reg(ops[2]) + if rs2 < 0 || vd < 0 { + return nil, true, fmt.Errorf("%s: invalid register operand", mnem) + } + nf := 3 // 4 fields + if mnem == "VLSSEG8E32V" { + nf = 7 // 8 fields + } + return wordLE(riscvVLSType(0x07, nf, 2, 6, int32(rs2), rs1, vd)), true, nil + + case "VADDVV", "VXORVV", "VMSNEVV": + // Vector-vector: INSTR vs1, vs2, vd. + if len(ops) != 3 { + return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + vs1, vs2, vd := reg(ops[0]), reg(ops[1]), reg(ops[2]) + if vs1 < 0 || vs2 < 0 || vd < 0 { + return nil, true, fmt.Errorf("%s: invalid vector register", mnem) + } + funct6 := map[string]int{"VADDVV": 0x00, "VXORVV": 0x0B, "VMSNEVV": 0x19}[mnem] + return wordLE(riscvVVInstr(funct6, riscvVf3VV, int32(vs1), vs2, vd)), true, nil + + case "VADDVX", "VMSEQVX": + // Vector-scalar: INSTR rs1, vs2, vd (the scalar in the rs1 field). + if len(ops) != 3 { + return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + rs1, vs2, vd := reg(ops[0]), reg(ops[1]), reg(ops[2]) + if rs1 < 0 || vs2 < 0 || vd < 0 { + return nil, true, fmt.Errorf("%s: invalid register operand", mnem) + } + funct6 := 0x00 + if mnem == "VMSEQVX" { + funct6 = 0x18 + } + return wordLE(riscvVVInstr(funct6, riscvVf3VX, int32(rs1), vs2, vd)), true, nil + + case "VSLLVI", "VSRLVI": + // Vector-immediate shift: INSTR $uimm, vs2, vd. + if len(ops) != 3 { + return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + imm := int(immFromOperand(ops[0])) + if imm < 0 || imm > 31 { + return nil, true, fmt.Errorf("%s: immediate out of range [0, 31]", mnem) + } + vs2, vd := reg(ops[1]), reg(ops[2]) + if vs2 < 0 || vd < 0 { + return nil, true, fmt.Errorf("%s: invalid vector register", mnem) + } + funct6 := 0x25 // vsll.vi + if mnem == "VSRLVI" { + funct6 = 0x28 // vsrl.vi + } + return wordLE(riscvVVInstr(funct6, riscvVf3VI, int32(imm), vs2, vd)), true, nil + + case "VFIRSTM": + // vmfirst.m rd, vs2: the unmasked form carries 0x11 in the rs1 field + // and sets the mask bit (funct7 = 0x20 | 1). + if len(ops) != 2 { + return nil, true, fmt.Errorf("VFIRSTM expects 2 operands, got %d", len(ops)) + } + vs2, rd := reg(ops[0]), reg(ops[1]) + if vs2 < 0 || rd < 0 { + return nil, true, fmt.Errorf("VFIRSTM: invalid register operand") + } + return wordLE(riscvVUnaryInstr(0x10, riscvVf3MV, 0x11, vs2, rd)), true, nil + + case "VIDV": + // vid.v vd (vs2 must be v0; the unmasked form sets the mask bit). + if len(ops) != 1 { + return nil, true, fmt.Errorf("VIDV expects 1 operand, got %d", len(ops)) + } + vd := reg(ops[0]) + if vd < 0 { + return nil, true, fmt.Errorf("VIDV: invalid vector register") + } + return wordLE(riscvVUnaryInstr(0x14, riscvVf3MV, 0x11, 0, vd)), true, nil + + case "VMV4RV": + // vmv4r.v vd, vs2: whole-register group move. + if len(ops) != 2 { + return nil, true, fmt.Errorf("VMV4RV expects 2 operands, got %d", len(ops)) + } + vs2, vd := reg(ops[0]), reg(ops[1]) + if vs2 < 0 || vd < 0 { + return nil, true, fmt.Errorf("VMV4RV: invalid vector register") + } + return wordLE(riscvVUnaryInstr(0x27, 0x3, 0x3, vs2, vd)), true, nil + } + return nil, false, nil +} + +// riscvVTypeToken parses a vsetvli configuration token (E8, M8, MF2 and +// friends): the letter prefix selects the field and the suffix its value +// through the given table. +func riscvVTypeToken(name, prefix string, codes map[string]int) (int, error) { + if len(name) <= len(prefix) || name[:len(prefix)] != prefix { + return 0, fmt.Errorf("invalid vtype token %q (want %s)", name, prefix) + } + code, ok := codes[name[len(prefix):]] + if !ok { + return 0, fmt.Errorf("invalid vtype token %q", name) + } + return code, nil +} + +// riscvVecMem reads a vector memory operand: a bare base register, the only +// addressing form the vector loads and stores carry. Frame-pseudo bases are +// rejected: the toolchain resolves no frame reference on the vector forms. +func riscvVecMem(op *ast.Operand) (rs1 int, ok bool) { + if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { + return -1, false + } + if op.Addr.Base == "" || op.Addr.Offset != 0 { + return -1, false + } + rs1 = riscvRegNum(op.Addr.Base) + return rs1, rs1 >= 0 +} + // Instruction type classifiers. func isRTypeInstr(m string) bool { switch m { @@ -1547,7 +2111,7 @@ func isFPArithInstr(m string) bool { switch m { case "FADDS", "FSUBS", "FMULS", "FDIVS", "FADDD", "FSUBD", "FMULD", "FDIVD", - "FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD": + "FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD": return true } return false diff --git a/asm/riscv_encode.go b/asm/riscv_encode.go index 23e8b26..eae74d0 100644 --- a/asm/riscv_encode.go +++ b/asm/riscv_encode.go @@ -141,10 +141,36 @@ func riscvRegNum(name string) int { case "F31", "FT11": return 31 default: + // Vector registers V0-V31 (the "V" extension). They share the + // register numbering with the integer file: a bare number 0-31. + if len(name) >= 2 && name[0] == 'V' { + if n, ok := parseRegDigits(name[1:], 31); ok { + return n + } + } return -1 } } +// parseRegDigits parses a decimal register suffix and reports whether it is +// within [0, max]. +func parseRegDigits(digits string, max int) (int, bool) { + if digits == "" { + return 0, false + } + n := 0 + for i := 0; i < len(digits); i++ { + if digits[i] < '0' || digits[i] > '9' { + return 0, false + } + n = n*10 + int(digits[i]-'0') + if n > max { + return 0, false + } + } + return n, true +} + // RISC-V instruction encoding parameters. type riscvEnc struct { opcode uint32 // bits [6:0] @@ -232,25 +258,28 @@ var riscvInstrTable = map[string]riscvEnc{ "JALR": {0x67, 0x0, 0x00}, // RV64A, atomics (AMO opcode 0x2F). - // funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27]. - "AMOSWAPW": {0x2F, 0x2, 0x01 << 2}, - "AMOSWAPD": {0x2F, 0x3, 0x01 << 2}, - "AMOADDW": {0x2F, 0x2, 0x00 << 2}, - "AMOADDD": {0x2F, 0x3, 0x00 << 2}, - "AMOANDW": {0x2F, 0x2, 0x0C << 2}, - "AMOANDD": {0x2F, 0x3, 0x0C << 2}, - "AMOORW": {0x2F, 0x2, 0x06 << 2}, - "AMOORD": {0x2F, 0x3, 0x06 << 2}, - "AMOXORW": {0x2F, 0x2, 0x04 << 2}, - "AMOXORD": {0x2F, 0x3, 0x04 << 2}, - "AMOMAXW": {0x2F, 0x2, 0x14 << 2}, - "AMOMAXD": {0x2F, 0x3, 0x14 << 2}, - "AMOMINW": {0x2F, 0x2, 0x10 << 2}, - "AMOMIND": {0x2F, 0x3, 0x10 << 2}, - "AMOMAXUW": {0x2F, 0x2, 0x1C << 2}, - "AMOMAXUD": {0x2F, 0x3, 0x1C << 2}, - "AMOMINUW": {0x2F, 0x2, 0x18 << 2}, - "AMOMINUD": {0x2F, 0x3, 0x18 << 2}, + // funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full + // 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in + // the lower two, exactly as the toolchain writes them: every AMO sets both + // aq and rl (funct7 |= 3). + "AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3}, + "AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3}, + "AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3}, + "AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3}, + "AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3}, + "AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3}, + "AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3}, + "AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3}, + "AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3}, + "AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3}, + "AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3}, + "AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3}, + "AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3}, + "AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3}, + "AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3}, + "AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3}, + "AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3}, + "AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3}, // RV64F/D, floating-point arithmetic. "FADDS": {0x53, 0x0, 0x00}, @@ -273,12 +302,16 @@ var riscvInstrTable = map[string]riscvEnc{ "FMAXS": {0x53, 0x1, 0x14}, "FMIND": {0x53, 0x0, 0x15}, "FMAXD": {0x53, 0x1, 0x15}, + // FP sign injection (double): rs2 carries the sign source. + "FSGNJD": {0x53, 0x0, 0x11}, // RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03). - "LRW": {0x2F, 0x2, 0x02 << 2}, - "LRD": {0x2F, 0x3, 0x02 << 2}, - "SCW": {0x2F, 0x2, 0x03 << 2}, - "SCD": {0x2F, 0x3, 0x03 << 2}, + // The toolchain gives LR acquire ordering (aq = 1) and SC release + // ordering (rl = 1). + "LRW": {0x2F, 0x2, 0x02<<2 | 0x2}, + "LRD": {0x2F, 0x3, 0x02<<2 | 0x2}, + "SCW": {0x2F, 0x2, 0x03<<2 | 0x1}, + "SCD": {0x2F, 0x3, 0x03<<2 | 0x1}, // FP compare, result in integer register (funct7 0x50/0x51). "FEQS": {0x53, 0x2, 0x50}, @@ -296,11 +329,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 { } // riscvAMOType encodes an atomic (AMO) instruction. -// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode. -// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2). +// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the +// funct5 in its upper five bits and the aq/rl ordering bits in the lower two +// (the table stores the full field, so the word needs no reassembly). func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 { - funct5 := enc.funct7 >> 2 // extract funct5 from the stored value - return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) | + return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) | (enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode } @@ -441,6 +474,71 @@ func riscvJType(rd int, offset int32) uint32 { 0x6F // JAL opcode } +// ---- RVV ("V" extension) encoding helpers ---- + +// The OP-V major opcode and its funct3 subclasses. +const ( + riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*) + // funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX, + // 5 OPFVF, 6 OPMVX, 7 vsetvli. + riscvVf3VV = 0x0 // vector-vector + riscvVf3MV = 0x2 // vector mask + riscvVf3VI = 0x3 // vector-immediate + riscvVf3VX = 0x4 // vector-scalar + riscvVf3Cfg = 0x7 // vsetvli +) + +// riscvVType composes the vsetvli/vsetivli vtype immediate: the register +// group multiplier in [2:0], the selected element width in [5:3] and the +// tail-agnostic and mask-agnostic policies in bits 6 and 7. +func riscvVType(vsew, vlmul, vta, vma int) int { + return vlmul | vsew<<3 | vta<<6 | vma<<7 +} + +// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the +// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a +// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00 +// the toolchain writes above its 10-bit vtype). +func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 { + imm := vtype & 0x3FF + if vsetivli { + imm |= 0xC00 + } + return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 | + uint32(rd)<<7 | riscvOpV +} + +// riscvVLSType encodes a vector load or store: the full 32-bit word with the +// segment count in bits [31:29], the addressing mode in bits [28:26], the +// unmasked bit at 25 and the width in funct3. width follows the load +// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit). +func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 { + return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 | + uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 | + uint32(rd)<<7 | op +} + +// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in +// funct7's upper bits, bit 25 as the unmasked flag and the three registers in +// the standard positions. vs1 may name an integer register for the *VX forms +// (the scalar sits in the rs1 field) or an immediate for the *VI forms. +func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 { + return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 | + uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV +} + +// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed +// fields live where the second source register would be: rs1Field and vs2 are +// written verbatim (the oracle writes fixed non-zero constants there for some +// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v). +func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 { + return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 | + uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV +} + +// riscvSegNF maps a segment count to the 3-bit nf field (count - 1). +func riscvSegNF(n int) int32 { return int32(n - 1) } + // ---- RVC (compressed) encoding helpers ---- // isRVCIntReg reports whether a register number can be encoded in the 3-bit diff --git a/asm/riscv_encode_test.go b/asm/riscv_encode_test.go index 0d1b920..f6dea7e 100644 --- a/asm/riscv_encode_test.go +++ b/asm/riscv_encode_test.go @@ -5,6 +5,8 @@ package asm import ( "bytes" + "encoding/binary" + "encoding/hex" "strings" "testing" @@ -971,3 +973,224 @@ TEXT ·edge(SB), NOSPLIT, $0 t.Errorf("int32-span immediates must assemble: %v", err) } } + +// riscvWants decodes code as little-endian words and pins each one; the +// expected values below were read off GOARCH=riscv64 go tool objdump of +// kernels assembled with go tool asm (the toolchain's riscv64.s testdata +// cross-checks the same words). +func riscvWants(t *testing.T, code []byte, want ...uint32) { + t.Helper() + got := make([]uint32, 0, len(code)/4) + for i := 0; i+4 <= len(code); i += 4 { + got = append(got, binary.LittleEndian.Uint32(code[i:])) + } + if len(got) < len(want) { + t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code) + } + // The RET (JALR) ends the sequence; only the pinned prefix is compared. + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// riscvWantsHex pins the exact hex encoding of a function's instruction +// bytes, including any 2-byte compressed instructions in the stream; the +// expected strings were read off GOARCH=riscv64 go tool objdump of kernels +// assembled with go tool asm (the toolchain's riscv64.s testdata +// cross-checks the same words). +func riscvWantsHex(t *testing.T, code []byte, wantHex string) { + t.Helper() + got := hex.EncodeToString(code) + if got != wantHex { + t.Errorf("code = %s, want %s", got, wantHex) + } +} + +// TestRISCV_extendedPseudos pins the toolchain-synthesised instructions: +// ANDN/ORN (XORI + AND/OR through the destination or TMP), the five-word +// MIN/MAX expansion, the four-word rotate, ROR's compressed reverse shift +// (C.SLLI when rd == rs1, both non-zero, 1 <= sll <= 63), the identical- +// input MIN/MAX fold to C.MV, FABSD (FSGNJX.D), SEQZ and RDTIME (csrrs with +// the time CSR). +func TestRISCV_extendedPseudos(t *testing.T) { + t.Run("logic and minmax", func(t *testing.T) { + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·l(SB), NOSPLIT, $0 + ANDN X19, X20, X21 + ANDN X19, X20 + ORN X20, X19 + MAX X26, X28, X29 + MIN X29, X30, X5 + MAX X5, X5 + MAX X5, X5, X6 + SEQZ X5, X6 + NEG X5, X6 + NOT X5 + RDTIME X5 + RET +`) + code := assembleRISCVHelper(t, fn) + // Words 0-10 up to the folded C.MV pair (halfwords 96 82 and 16 83), + // then SEQZ, NEG, NOT and RDTIME. + riscvWantsHex(t, code, + "93caf9ffb37a5a01"+"93cff9ff337afa01"+"934ffaffb3e9f901"+ + "b32fae01b30ff041b34eae01b3fedf01b34ede01"+ + "b3afee01b30ff041b342df01b3f25f00b3425f00"+ + "9682"+"1683"+ + "13b31200"+"33035040"+"93c2f2ff"+"f32210c0"+"67800000") + }) + + t.Run("rotate", func(t *testing.T) { + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·r(SB), NOSPLIT, $0 + ROR X10, X11, X12 + ROR X10, X11 + ROR $63, X11 + RORIW $31, X13, X14 + RORIW $1, X14, X15 + RORIW $3, X14 + RORW X15, X16, X17 + RORW $31, X13 + RET +`) + code := assembleRISCVHelper(t, fn) + // The third ROR carries the compressed C.SLLI (05 86) in mid-stream. + riscvWantsHex(t, code, + "b30fa040b39ff50133d6a50033e6cf00"+ + "b30fa040b39ff501b3d5a500b3e5bf00"+ + "93dff5038605b3e5bf00"+ + "9bdff6011b97160033e7ef00"+ + "9b5f17009b17f701b3e7ff00"+ + "9b5f37001b17d70133e7ef00"+ + "b30ff040bb1ff801bb58f800b3e81f01"+ + "9bdff6019b961600b3e6df00"+"67800000") + }) + + t.Run("fp and branches", func(t *testing.T) { + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·f(SB), NOSPLIT, $0 + FABSD F1, F2 + FSGNJD F1, F0, F2 + FMADDD F1, F2, F3, F4 + FMSUBD F1, F2, F3, F4 + FNMSUBD F1, F2, F3, F4 + BGT X5, X6, tgt + BLE X5, X6, tgt + BGTU X5, X6, tgt + BLEU X5, X6, tgt +tgt: + RDTIME X5 + RET +`) + code := assembleRISCVHelper(t, fn) + riscvWantsHex(t, code, + "53a11022"+"53011022"+"4382201a4782201a4b82201a"+ + "63485300635653006364530063725300"+ // blt/bge/bltu/bgeu x6, x5 + "f32210c0"+"67800000") + }) +} + +// TestRISCV_amoWords pins the full AMO family: every AMO carries aq and rl +// (funct7 |= 3), LR is acquire (funct7 |= 2) and SC release (funct7 |= 1), +// exactly as GOARCH=riscv64 go tool asm encodes them. +func TestRISCV_amoWords(t *testing.T) { + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·amo(SB), NOSPLIT, $0 + AMOSWAPW X5, (X6), X7 + AMOSWAPD X5, (X6), X7 + AMOADDW X5, (X6), X7 + AMOADDD X5, (X6), X7 + AMOANDW X5, (X6), X7 + AMOANDD X5, (X6), X7 + AMOORW X5, (X6), X7 + AMOORD X5, (X6), X7 + AMOXORW X5, (X6), X7 + AMOXORD X5, (X6), X7 + AMOMAXW X5, (X6), X7 + AMOMAXD X5, (X6), X7 + AMOMAXUW X5, (X6), X7 + AMOMAXUD X5, (X6), X7 + AMOMINUW X5, (X6), X7 + AMOMINUD X5, (X6), X7 + LRW (X5), X6 + LRD (X5), X6 + SCW X5, (X6), X7 + SCD X5, (X6), X7 + RET +`) + code := assembleRISCVHelper(t, fn) + riscvWants(t, code, + 0x0E5323AF, // amoswap.w + 0x0E5333AF, // amoswap.d + 0x065323AF, // amoaddd.w + 0x065333AF, // amoadd.d + 0x665323AF, // amoand.w + 0x665333AF, // amoand.d + 0x465323AF, // amoor.w + 0x465333AF, // amoor.d + 0x265323AF, // amoxor.w + 0x265333AF, // amoxor.d + 0xA65323AF, // amomax.w + 0xA65333AF, // amomax.d + 0xE65323AF, // amomaxu.w + 0xE65333AF, // amomaxu.d + 0xC65323AF, // amominu.w + 0xC65333AF, // amominu.d + 0x1402A32F, // lr.w (aq) + 0x1402B32F, // lr.d + 0x1A5323AF, // sc.w (rl) + 0x1A5333AF, // sc.d + ) +} + +// TestRISCV_vectorWords pins the RVV slice and the VSET* encodings. The +// toolchain canonicalises an immediate avl to vsetivli even under the +// VSETVLI spelling (`VSETVLI $15` and `VSETIVLI $15` come out byte- +// identical), which is what the 0xC00 bit of the first word carries. +func TestRISCV_vectorWords(t *testing.T) { + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·v(SB), NOSPLIT, $0 + VSETVLI X5, E8, M8, TA, MA, X6 + VSETIVLI $4, E32, M1, TA, MA, X0 + VSETVLI $15, E32, M1, TA, MA, X12 + VADDVV V1, V2, V3 + VADDVX X12, V12, V12 + VXORVV V8, V16, V24 + VMSEQVX X12, V8, V0 + VMSNEVV V8, V16, V0 + VSLLVI $8, V28, V30 + VSRLVI $25, V29, V29 + VFIRSTM V0, X6 + VIDV V12 + VMV4RV V8, V24 + VLE8V (X10), V8 + VSE8V V24, (X10) + VSE32V V9, (X11) + VLSSEG4E32V (X14), X0, V0 + VLSSEG8E32V (X10), X0, V4 + RET +`) + code := assembleRISCVHelper(t, fn) + riscvWants(t, code, + 0x0C32F357, // vsetvli x6, x5, vtype 0xc3 (E8, M8, TA, MA) + 0xCD027057, // vsetivli x0, 4 + 0xCD07F657, // vsetivli x12, 15: VSETVLI $15 canonicalises to the same word + 0x022081D7, // vadd.vv v3, v2, v1 + 0x02C64657, // vadd.vx v12, v12, x12 + 0x2F040C57, // vxor.vv v24, v16, v8 + 0x62864057, // vmseq.vx v0, v8, x12 + 0x67040057, // vmsne.vv v0, v16, v8 + 0x97C43F57, // vsll.vi v30, v28, 8 + 0xA3DCBED7, // vsrl.vi v29, v29, 25 + 0x4208A357, // vmfirst.m x6, v0 + 0x5208A657, // vid.v v12 + 0x9E81BC57, // vmv4r.v v24, v8 + 0x02050407, // vle8.v v8, (x10) + 0x02050C27, // vse8.v v24, (x10) + 0x0205E4A7, // vse32.v v9, (x11) + 0x6A076007, // vlsseg4e32.v v0, (x14), x0 + 0xEA056207, // vlsseg8e32.v v4, (x10), x0 + ) +} diff --git a/testdata/verify/atomics_loong64.s b/testdata/verify/atomics_loong64.s new file mode 100644 index 0000000..b09b285 --- /dev/null +++ b/testdata/verify/atomics_loong64.s @@ -0,0 +1,49 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the loong64 atomics: the AM* family in its plain +// and _dbar (acquire/release) forms, spelled as the runtime's +// atomic_loong64.s spells them. Every AM* takes three operands: +// value, (address), result. + +#include "textflag.h" + +TEXT ·plain(SB), NOSPLIT, $0-0 + AMSWAPB R14, (R13), R12 + AMSWAPH R14, (R13), R12 + AMSWAPW R5, (R4), R6 + AMSWAPV R5, (R4), R0 + AMCASB R14, (R13), R12 + AMCASH R6, (R4), R5 + AMCASW R6, (R4), R5 + AMCASV R6, (R4), R5 + AMADDW R5, (R4), R0 + AMADDV R14, (R13), R12 + AMANDW R5, (R4), R6 + AMANDV R5, (R4), R6 + AMORW R5, (R4), R0 + AMORV R5, (R4), R6 + AMXORW R5, (R4), R6 + AMXORV R5, (R4), R6 + AMMAXW R5, (R4), R6 + AMMAXV R5, (R4), R6 + AMMINW R5, (R4), R6 + AMMINV R5, (R4), R6 + AMMAXWU R5, (R4), R6 + AMMAXVU R5, (R4), R6 + AMMINWU R5, (R4), R6 + AMMINVU R5, (R4), R6 + RET + +TEXT ·dbar(SB), NOSPLIT, $0-0 + AMADDDBW R5, (R4), R6 + AMADDDBV R5, (R4), R6 + AMANDDBW R5, (R6), R0 + AMANDDBV R5, (R4), R6 + AMORDBW R5, (R6), R0 + AMORDBV R5, (R4), R6 + AMSWAPDBW R5, (R4), R6 + AMSWAPDBV R5, (R4), R0 + AMCASDBW R6, (R4), R5 + AMCASDBV R6, (R4), R5 + RET diff --git a/testdata/verify/atomics_riscv64.s b/testdata/verify/atomics_riscv64.s new file mode 100644 index 0000000..5cc9abd --- /dev/null +++ b/testdata/verify/atomics_riscv64.s @@ -0,0 +1,35 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the riscv64 atomics: the RV64A AMO family and the +// load-reserved / store-conditional pair, in the toolchain's spelling +// (value, (address), result). Both orderings sit in the encodings: the +// table gives every AMO aq and rl, LR acquire and SC release. + +#include "textflag.h" + +TEXT ·amo(SB), NOSPLIT, $0-0 + AMOSWAPW X5, (X6), X7 + AMOSWAPD X5, (X6), X7 + AMOADDW X5, (X6), X7 + AMOADDD X5, (X6), X7 + AMOANDW X5, (X6), X7 + AMOANDD X5, (X6), X7 + AMOORW X5, (X6), X7 + AMOORD X5, (X6), X7 + AMOXORW X5, (X6), X7 + AMOXORD X5, (X6), X7 + AMOMAXW X5, (X6), X7 + AMOMAXD X5, (X6), X7 + AMOMAXUW X5, (X6), X7 + AMOMAXUD X5, (X6), X7 + AMOMINUW X5, (X6), X7 + AMOMINUD X5, (X6), X7 + RET + +TEXT ·lrsc(SB), NOSPLIT, $0-0 + LRW (X5), X6 + LRD (X5), X6 + SCW X5, (X6), X7 + SCD X5, (X6), X7 + RET diff --git a/testdata/verify/bitmanip_riscv64.s b/testdata/verify/bitmanip_riscv64.s new file mode 100644 index 0000000..7ed32bf --- /dev/null +++ b/testdata/verify/bitmanip_riscv64.s @@ -0,0 +1,64 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the riscv64 toolchain-synthesised instructions: +// the Zbb-style pseudos the assembler expands instruction-for-instruction +// (ANDN/ORN, MIN/MAX, ROR and friends, the reversed branches, FABSD), the +// CSR read RDTIME and the FP sign-injection and fused-multiply-add forms. + +#include "textflag.h" + +TEXT ·logic(SB), NOSPLIT, $0-0 + ANDN X19, X20, X21 + ANDN X19, X20 + ANDN X21, X19, X21 + ORN X20, X19 + ORN X20, X19, X21 + MAX X26, X28, X29 + MAX X26, X28 + MAXU X28, X29, X30 + MAXU X28, X29 + MIN X29, X30, X5 + MIN X29, X30 + MINU X30, X5, X6 + MINU X30, X5 + MAX X5, X5 + MAX X5, X5, X6 + SEQZ X5, X6 + NEG X5, X6 + NEG X5 + NOT X5 + NOT X5, X6 + NOP + RET + +TEXT ·rotate(SB), NOSPLIT, $0-0 + ROR X10, X11, X12 + ROR X10, X11 + ROR $63, X11 + RORIW $31, X13, X14 + RORIW $1, X14, X15 + RORIW $3, X14 + RORW X15, X16, X17 + RORW $31, X13 + RET + +TEXT ·fp(SB), NOSPLIT, $0-0 + FABSD F1, F2 + FSGNJD F1, F0, F2 + FMADDD F1, F2, F3, F4 + FMSUBD F1, F2, F3, F4 + FNMSUBD F1, F2, F3, F4 + FMADDS F1, F2, F3, F4 + FNMADDS F1, F2, F3, F4 + RET + +TEXT ·branches(SB), NOSPLIT, $0-0 + BGT X5, X6, tgt + BLE X5, X6, tgt + BGTU X5, X6, tgt + BLEU X5, X6, tgt + +tgt: + RDTIME X5 + RET diff --git a/testdata/verify/vector_loong64.s b/testdata/verify/vector_loong64.s new file mode 100644 index 0000000..1577483 --- /dev/null +++ b/testdata/verify/vector_loong64.s @@ -0,0 +1,85 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the loong64 LSX/LASX slice: every function pairs +// with the same instructions in the go tool asm ground truth. + +#include "textflag.h" + +TEXT ·threeReg(SB), NOSPLIT, $0-0 + VADDV V1, V2, V3 + VADDW V1, V2, V3 + VADDV V2, V1 + VANDV V1, V2, V3 + VANDV V1, V2 + VXORV V1, V2, V3 + VXORV V1, V2 + VSEQB V1, V2, V3 + VSEQV V1, V2, V3 + VSRAB V1, V2, V3 + VROTRW V1, V2, V3 + VPCNTV V1, V2 + XVADDV X1, X2, X3 + XVADDV X2, X1 + XVANDV X1, X2, X3 + XVXORV X1, X2, X3 + XVSEQB X1, X2, X3 + XVSEQV X1, X2, X3 + XVPCNTV X1, X2 + RET + +TEXT ·immediates(SB), NOSPLIT, $0-0 + VANDB $0, V2, V3 + VANDB $255, V2 + VSEQB $3, V2, V3 + VSEQV $15, V2, V3 + VSRAB $0, V1, V2 + VSRAB $7, V1, V2 + VSRAB $6, V1 + VROTRW $0, V1, V2 + VROTRW $16, V1, V2 + VROTRW $16, V1 + XVANDB $1, X2, X2 + RET + +TEXT ·conditions(SB), NOSPLIT, $0-0 + VSETNEV V1, FCC0 + VSETANYEQB V1, FCC0 + VSETANYEQV V2, FCC0 + VSETALLNEV V0, FCC0 + XVSETNEV X1, FCC0 + XVSETANYEQB X1, FCC0 + XVSETANYEQV X1, FCC0 + XVSETALLNEV X1, FCC0 + RET + +TEXT ·fpConvert(SB), NOSPLIT, $0-0 + FFINTDV F0, F1 + FSEL FCC0, F3, F4, F3 + FSEL FCC1, F1, F2 + RET + +TEXT ·memMoves(SB), NOSPLIT, $0-0 + VMOVQ V1, V9 + VMOVQ (R4), V2 + VMOVQ 16(R4), V2 + VMOVQ V0, (R4) + VMOVQ V0, 32(R4) + VMOVQ V0,-16(R6) + VMOVQ (R4)(R7), V3 + VMOVQ V3, (R4)(R7) + XVMOVQ X3, X7 + XVMOVQ (R4), X2 + XVMOVQ X0, (R4) + XVMOVQ (R4)(R7), X4 + XVMOVQ X0, (R4)(R7) + RET + +TEXT ·elements(SB), NOSPLIT, $0-0 + VMOVQ R6, V0.B16 + VMOVQ R6, V12.W4 + XVMOVQ R6, X0.B32 + VMOVQ (R4), V4.W4 + VMOVQ (R10), V0.W4 + XVMOVQ (R4), X0.B32 + RET diff --git a/testdata/verify/vector_riscv64.s b/testdata/verify/vector_riscv64.s new file mode 100644 index 0000000..006ba50 --- /dev/null +++ b/testdata/verify/vector_riscv64.s @@ -0,0 +1,53 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the riscv64 RVV slice: the instructions GOROOT's +// vector kernels use (crypto/internal/fips140/subtle/xor_riscv64.s, +// internal/bytealg and internal/chacha8rand), spelled as they spell them. + +#include "textflag.h" + +TEXT ·config(SB), NOSPLIT, $0-0 + VSETVLI X5, E8, M8, TA, MA, X6 + VSETVLI X11, E8, M8, TA, MA, X5 + VSETVLI X12, E8, M8, TA, MA, X5 + VSETVLI X13, E8, M8, TU, MU, X15 + VSETIVLI $4, E32, M1, TA, MA, X0 + VSETIVLI $15, E32, M1, TA, MA, X12 + VSETVLI $15, E32, M1, TA, MA, X12 + VSETVLI X10, E16, M1, TU, MU, X12 + VSETVLI X10, E32, M2, TA, MA, X12 + VSETVLI X10, E64, M8, TU, MU, X12 + VSETIVLI $31, E32, M1, TA, MA, X12 + RET + +TEXT ·loadsStores(SB), NOSPLIT, $0-0 + VLE8V (X10), V8 + VLE8V (X11), V16 + VLE8V (X12), V16 + VIDV V12 + VMV4RV V8, V24 + VSE8V V24, (X10) + VSE32V V0, (X11) + VSE32V V8, (X11) + VSE32V V15, (X11) + RET + +TEXT ·segmented(SB), NOSPLIT, $0-0 + VLSSEG4E32V (X14), X0, V0 + VLSSEG8E32V (X10), X0, V4 + RET + +TEXT ·crypto(SB), NOSPLIT, $0-0 + VADDVV V20, V4, V4 + VADDVV V27, V11, V11 + VADDVX X12, V12, V12 + VXORVV V8, V16, V24 + VXORVV V13, V13, V13 + VMSEQVX X12, V8, V0 + VMSNEVV V8, V16, V0 + VFIRSTM V0, X6 + VFIRSTM V0, X7 + VSLLVI $8, V28, V30 + VSRLVI $25, V29, V29 + RET