feat(riscv64,loong64): encode AMO atomics, vector slices and bit ops
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+171
-6
@@ -30,7 +30,10 @@ package asm
|
||||
// of the immediate and register fields), mirroring the toolchain's OP_*
|
||||
// helpers, so each l64* function only ORs its fields in.
|
||||
|
||||
import "maps"
|
||||
import (
|
||||
"maps"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
||||
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
|
||||
@@ -103,7 +106,12 @@ func loong64RegNum(name string) int {
|
||||
case "R31", "S8":
|
||||
return 31
|
||||
}
|
||||
// F0-F31, FCC0-FCC7, FCSR0-FCSR31.
|
||||
// F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31,
|
||||
// X0-X31) are deliberately NOT accepted here: they are a separate
|
||||
// register class, and the toolchain rejects V/X names wherever an
|
||||
// integer or FP register is expected (GOARCH=loong64 go tool asm reports
|
||||
// "unrecognized instruction" for `BEQZ X0`). Vector operands are
|
||||
// resolved only through loong64VecRegNum.
|
||||
if len(name) >= 4 && name[:4] == "FCSR" {
|
||||
return loong64RegSpecial(name[4:], 31)
|
||||
}
|
||||
@@ -148,6 +156,19 @@ func loong64RegSpecial(digits string, max int) int {
|
||||
return -1
|
||||
}
|
||||
|
||||
// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or
|
||||
// X0-X31) to its 5-bit number, or -1. The vector banks are a register class
|
||||
// of their own: the toolchain accepts them only in the vector operands of the
|
||||
// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1,
|
||||
// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X
|
||||
// spellings never reach the integer/FP resolver.
|
||||
func loong64VecRegNum(name string) int {
|
||||
if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') {
|
||||
return -1
|
||||
}
|
||||
return loong64RegSpecial(name[1:], 31)
|
||||
}
|
||||
|
||||
// ---- format helpers ----
|
||||
|
||||
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
||||
@@ -247,7 +268,7 @@ const (
|
||||
l64Firr14 // 2RI14 (ldptr/stptr)
|
||||
l64Firr16 // 2RI16 (addu16i.d)
|
||||
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
||||
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
|
||||
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel)
|
||||
l64Firir // bstrins/bstrpick
|
||||
l64Firrr // alsl
|
||||
l64Fi15 // syscall/break/dbar
|
||||
@@ -255,6 +276,8 @@ const (
|
||||
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
||||
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
||||
l64Fpreld // preld (2RI12 + 5-bit hint)
|
||||
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
|
||||
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
|
||||
)
|
||||
|
||||
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
||||
@@ -277,10 +300,68 @@ type l64DualEnc struct {
|
||||
var l64DualTable = map[string]l64DualEnc{}
|
||||
|
||||
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
||||
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
|
||||
// yet; the base integer, memory and floating-point ISA is complete.
|
||||
// to their encoding.
|
||||
var l64InstrTable = map[string]l64Enc{}
|
||||
|
||||
// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX
|
||||
// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per
|
||||
// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and
|
||||
// `XVADDV X1, X2, X3`, and rejects the crossed spellings.
|
||||
type l64Vec3Enc struct {
|
||||
op uint32
|
||||
lasx bool
|
||||
}
|
||||
|
||||
// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic:
|
||||
// the opcode, the bank, the accepted immediate range, the bias the toolchain
|
||||
// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b
|
||||
// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one).
|
||||
type l64VecImmEnc struct {
|
||||
op uint32
|
||||
lasx bool
|
||||
min, max int
|
||||
bias int
|
||||
mask int
|
||||
}
|
||||
|
||||
// l64VecBank marks the LSX/LASX mnemonics and records which register bank
|
||||
// each accepts; presence in the map routes the mnemonic through the vector
|
||||
// dispatcher rather than the integer/FP formats.
|
||||
var l64VecBank = map[string]bool{}
|
||||
|
||||
// l64VecImmInfo mirrors l64VecImmTable for the dispatcher.
|
||||
var l64VecImmInfo = map[string]l64VecImmEnc{}
|
||||
|
||||
// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as
|
||||
// vpcnt.v).
|
||||
var l64Vec2R = map[string]bool{}
|
||||
|
||||
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
|
||||
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
|
||||
type l64VmovqEnc struct {
|
||||
ld, st, ldx, stx uint32 // plain and indexed load/store
|
||||
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
|
||||
pickS, pickU uint32 // vpickve2gr.{,u} element extract
|
||||
ins uint32 // vinsgr2vr element insert
|
||||
dup uint32 // vreplgr2vr duplicate (width in [11:10])
|
||||
move uint32 // vori.b/xvori.b $0 register move
|
||||
}
|
||||
|
||||
var l64VmovqTable = map[bool]l64VmovqEnc{
|
||||
false: { // VMOVQ, the LSX (V) bank
|
||||
ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15,
|
||||
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
|
||||
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
|
||||
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
|
||||
},
|
||||
true: { // XVMOVQ, the LASX (X) bank
|
||||
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
|
||||
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
|
||||
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
|
||||
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
|
||||
},
|
||||
}
|
||||
|
||||
func init() {
|
||||
// 3R, integer.
|
||||
rrr := map[string]uint32{
|
||||
@@ -360,6 +441,10 @@ func init() {
|
||||
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
||||
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
||||
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
||||
// LSX: convert a 64-bit integer lane to a double float. The operand
|
||||
// bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`),
|
||||
// so the entry stays on the 2R integer/FP format.
|
||||
"FFINTDV": 0x474a << 10,
|
||||
}
|
||||
for m, op := range rr {
|
||||
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
||||
@@ -416,12 +501,14 @@ func init() {
|
||||
// LUI is the Plan 9 spelling of lu12i.w.
|
||||
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
||||
|
||||
// 4R, fused multiply-add.
|
||||
// 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC
|
||||
// condition flag, the layout matches the 4R shape).
|
||||
rrrr := map[string]uint32{
|
||||
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
||||
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
||||
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
||||
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
||||
"FSEL": 0x340 << 18,
|
||||
}
|
||||
for m, op := range rrrr {
|
||||
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
||||
@@ -455,6 +542,10 @@ func init() {
|
||||
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
||||
|
||||
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
|
||||
// The toolchain's form is three operands, `AMADDW rk, (rj), rd`
|
||||
// (cmd/asm/internal/asm/testdata/loong64enc1.s and
|
||||
// internal/runtime/atomic/atomic_loong64.s); the two-register spelling
|
||||
// is rejected by the oracle.
|
||||
am := map[string]uint32{
|
||||
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
||||
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
||||
@@ -472,10 +563,84 @@ func init() {
|
||||
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
||||
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
||||
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
||||
// The _dbar (acquire/release) add, and, or variants: opcodes read off
|
||||
// `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends.
|
||||
"AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15,
|
||||
"AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15,
|
||||
"AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15,
|
||||
}
|
||||
for m, op := range am {
|
||||
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
||||
}
|
||||
|
||||
// ---- LSX/LASX (V*/XV*) ----
|
||||
// Every opcode below was read off `go tool objdump` of a GOARCH=loong64
|
||||
// `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks
|
||||
// most of them), not assumed from the LoongArch manual.
|
||||
|
||||
// Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with
|
||||
// vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain
|
||||
// accepts: LSX spellings take V0-V31, LASX spellings X0-X31.
|
||||
vec3 := map[string]l64Vec3Enc{
|
||||
"VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false},
|
||||
"VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false},
|
||||
"VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false},
|
||||
"VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false},
|
||||
"XVADDV": {0xE817 << 15, true},
|
||||
"XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true},
|
||||
"XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true},
|
||||
}
|
||||
for m, e := range vec3 {
|
||||
l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op}
|
||||
l64VecBank[m] = e.lasx
|
||||
}
|
||||
|
||||
// Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate
|
||||
// range, bias and field mask are the ones the toolchain encodes: vandi.b
|
||||
// stores the raw 8-bit constant, vsrai.b stores imm+8 (byte-lane bias),
|
||||
// vseqi.b and vseqi.d store 5-bit and 7-bit two's-complement values.
|
||||
// The mnemonics that also have a register form (VSEQB, VSEQV, VSRAB,
|
||||
// VROTRW) keep their three-register entry in l64InstrTable; the
|
||||
// dispatcher picks the immediate opcode from l64VecImmInfo by operand
|
||||
// kind, so the immediate entries must not overwrite the table.
|
||||
vecImm := map[string]l64VecImmEnc{
|
||||
"VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF},
|
||||
"XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF},
|
||||
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
|
||||
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
|
||||
"VSEQV": {0xE503 << 15, false, -64, 63, 0, 0x7F},
|
||||
"XVSEQV": {0xE903 << 15, true, -64, 63, 0, 0x7F},
|
||||
"VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F},
|
||||
"VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F},
|
||||
}
|
||||
for m, e := range vecImm {
|
||||
l64VecImmInfo[m] = e
|
||||
l64VecBank[m] = e.lasx
|
||||
}
|
||||
|
||||
// Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*,
|
||||
// vsetallnez.*): the sub-op rides in the rk field.
|
||||
vecCf := map[string]uint32{
|
||||
"VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10,
|
||||
"VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10,
|
||||
"VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10,
|
||||
"VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10,
|
||||
}
|
||||
for m, op := range vecCf {
|
||||
l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op}
|
||||
l64VecBank[m] = strings.HasPrefix(m, "XV")
|
||||
}
|
||||
|
||||
// Lane popcount: INSTR vj, vd (the 2R layout with the opcode extending
|
||||
// over the unused vk field).
|
||||
vec2r := map[string]l64Vec3Enc{
|
||||
"VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true},
|
||||
}
|
||||
for m, e := range vec2r {
|
||||
l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op}
|
||||
l64VecBank[m] = e.lasx
|
||||
l64Vec2R[m] = true
|
||||
}
|
||||
}
|
||||
|
||||
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
||||
|
||||
Reference in New Issue
Block a user