feat(asm): encode the riscv64 quad-precision family and fix the fp cvt paths

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 00:47:27 +02:00
1 parent 955bc6643e
commit 9b751f6e58
4 files changed
+331 -50

No files matched your search

+83 -44
View File
@@ -344,22 +344,46 @@ var riscvInstrTable = map[string]riscvEnc{
// FP loads/stores.
"FLW": {0x07, 0x2, 0x00},
"FLD": {0x07, 0x3, 0x00},
"FLQ": {0x07, 0x4, 0x00},
"FSW": {0x27, 0x2, 0x00},
"FSD": {0x27, 0x3, 0x00},
"FSQ": {0x27, 0x4, 0x00},
// FP min/max.
"FMINS": {0x53, 0x0, 0x14},
"FMAXS": {0x53, 0x1, 0x14},
"FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15},
// FP sign injection (double): rs2 carries the sign source.
// FP compare: the integer destination rides in rd (the 3-op R-type
// path writes it there).
"FEQS": {0x53, 0x2, 0x50},
"FLTS": {0x53, 0x1, 0x50},
"FLES": {0x53, 0x0, 0x50},
"FEQD": {0x53, 0x2, 0x51},
"FLTD": {0x53, 0x1, 0x51},
"FLED": {0x53, 0x0, 0x51},
"FEQQ": {0x53, 0x2, 0x53},
"FLTQ": {0x53, 0x1, 0x53},
"FLEQ": {0x53, 0x0, 0x53},
// FP sign injection: the sign source rides in rs2; the XOR form's
// funct3 is 2 in every width.
"FSGNJD": {0x53, 0x0, 0x11},
"FSGNJS": {0x53, 0x0, 0x10},
"FSGNJX": {0x53, 0x0, 0x14},
"FSGNJXD": {0x53, 0x0, 0x15},
"FSGNJXS": {0x53, 0x0, 0x14},
"FSGNJXD": {0x53, 0x2, 0x11},
"FSGNJXS": {0x53, 0x2, 0x10},
"FSGNJND": {0x53, 0x1, 0x11},
"FSGNJNS": {0x53, 0x1, 0x10},
"FSGNJNX": {0x53, 0x1, 0x14},
"FSGNJQ": {0x53, 0x0, 0x13},
"FSGNJNQ": {0x53, 0x1, 0x13},
"FSGNJXQ": {0x53, 0x2, 0x13},
// RV64Q, quad-precision arithmetic (the fmt field rides in funct7's
// low bits: 11 for quad).
"FADDQ": {0x53, 0x0, 0x03},
"FSUBQ": {0x53, 0x0, 0x07},
"FMULQ": {0x53, 0x0, 0x0B},
"FDIVQ": {0x53, 0x0, 0x0F},
"FSQRTQ": {0x53, 0x0, 0x2F},
"FMINQ": {0x53, 0x0, 0x17},
"FMAXQ": {0x53, 0x1, 0x17},
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
// The toolchain gives LR acquire ordering (aq = 1) and SC release
@@ -368,14 +392,6 @@ var riscvInstrTable = map[string]riscvEnc{
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
// FP compare, result in integer register (funct7 0x50/0x51).
"FEQS": {0x53, 0x2, 0x50},
"FLTS": {0x53, 0x1, 0x50},
"FLES": {0x53, 0x0, 0x50},
"FEQD": {0x53, 0x2, 0x51},
"FLTD": {0x53, 0x1, 0x51},
"FLED": {0x53, 0x0, 0x51},
}
// riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode.
@@ -399,49 +415,68 @@ func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
type riscvCvtEnc struct {
funct7 uint32 // bits [31:25]
rs2 uint32 // conversion-type code in bits [24:20]
funct3 uint32 // the rounding mode or the fclass marker, bits [14:12]
opcode uint32 // always 0x53 (OP-FP)
}
var riscvCvtTable = map[string]riscvCvtEnc{
// float → int (rs2 selects the integer width/sign).
"FCVTWS": {0x60, 0x0, 0x53}, // float32 → int32
"FCVTWUS": {0x60, 0x1, 0x53}, // float32 → uint32
"FCVTLS": {0x60, 0x2, 0x53}, // float32 → int64
"FCVTLUS": {0x60, 0x3, 0x53}, // float32 → uint64
"FCVTWD": {0x61, 0x0, 0x53}, // float64 → int32
"FCVTWUD": {0x61, 0x1, 0x53}, // float64 → uint32
"FCVTLD": {0x61, 0x2, 0x53}, // float64 → int64
"FCVTLUD": {0x61, 0x3, 0x53}, // float64 → uint64
// int → float (rs2 selects the integer width/sign).
"FCVTSW": {0x68, 0x0, 0x53}, // int32 → float32
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
"FCVTDLU": {0x69, 0x3, 0x53}, // uint64 → float64
// float → int (rs2 selects the integer width/sign). The bare forms
// carry the specification's default rounding mode RTZ in funct3; the
// suffixed spellings override it.
"FCVTWS": {0x60, 0x0, 0x1, 0x53}, // float32 → int32
"FCVTWUS": {0x60, 0x1, 0x1, 0x53}, // float32 → uint32
"FCVTLS": {0x60, 0x2, 0x1, 0x53}, // float32 → int64
"FCVTLUS": {0x60, 0x3, 0x1, 0x53}, // float32 → uint64
"FCVTWD": {0x61, 0x0, 0x1, 0x53}, // float64 → int32
"FCVTWUD": {0x61, 0x1, 0x1, 0x53}, // float64 → uint32
"FCVTLD": {0x61, 0x2, 0x1, 0x53}, // float64 → int64
"FCVTLUD": {0x61, 0x3, 0x1, 0x53}, // float64 → uint64
// int → float (rs2 selects the integer width/sign): the default
// rounding mode RNE keeps funct3 0.
"FCVTSW": {0x68, 0x0, 0x0, 0x53}, // int32 → float32
"FCVTSWU": {0x68, 0x1, 0x0, 0x53}, // uint32 → float32
"FCVTSL": {0x68, 0x2, 0x0, 0x53}, // int64 → float32
"FCVTSLU": {0x68, 0x3, 0x0, 0x53}, // uint64 → float32
"FCLASSS": {0x70, 0x0, 0x1, 0x53}, // classify float32 → GPR mask
"FCLASSD": {0x71, 0x0, 0x1, 0x53}, // classify float64 → GPR mask
"FCVTDW": {0x69, 0x0, 0x0, 0x53}, // int32 → float64
"FCVTDWU": {0x69, 0x1, 0x0, 0x53}, // uint32 → float64
"FCVTDL": {0x69, 0x2, 0x0, 0x53}, // int64 → float64
"FCVTDLU": {0x69, 0x3, 0x0, 0x53}, // uint64 → float64
// float → float width conversion.
"FCVTSD": {0x20, 0x1, 0x53}, // float64 → float32
"FCVTDS": {0x21, 0x0, 0x53}, // float32 → float64
"FCVTSD": {0x20, 0x1, 0x0, 0x53}, // float64 → float32
"FCVTDS": {0x21, 0x0, 0x0, 0x53}, // float32 → float64
// Quad-precision conversions: the fmt field rides in funct7's low
// bits (11 for quad), the other width in rs2 where one is needed.
"FCVTSQ": {0x20, 0x3, 0x0, 0x53}, // quad → float32
"FCVTDQ": {0x21, 0x3, 0x0, 0x53}, // quad → float64
"FCVTQS": {0x23, 0x0, 0x0, 0x53}, // float32 → quad
"FCVTQD": {0x23, 0x1, 0x0, 0x53}, // float64 → quad
"FCVTWQ": {0x63, 0x0, 0x1, 0x53}, // quad → int32
"FCVTWUQ": {0x63, 0x1, 0x1, 0x53}, // quad → uint32
"FCVTLQ": {0x63, 0x2, 0x1, 0x53}, // quad → int64
"FCVTLUQ": {0x63, 0x3, 0x1, 0x53}, // quad → uint64
"FCVTQW": {0x6B, 0x0, 0x0, 0x53}, // int32 → quad
"FCVTQWU": {0x6B, 0x1, 0x0, 0x53}, // uint32 → quad
"FCVTQL": {0x6B, 0x2, 0x0, 0x53}, // int64 → quad
"FCVTQLU": {0x6B, 0x3, 0x0, 0x53}, // uint64 → quad
"FCLASSQ": {0x73, 0x0, 0x1, 0x53}, // classify quad → GPR mask
// Bit moves between integer and FP registers (no conversion).
"FMVXD": {0x71, 0x0, 0x53}, // float64 → int64 (bit move)
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
"FMVXD": {0x71, 0x0, 0x0, 0x53}, // float64 → int64 (bit move)
"FMVDX": {0x79, 0x0, 0x0, 0x53}, // int64 → float64 (bit move)
"FMVXW": {0x70, 0x0, 0x0, 0x53}, // float32 → int32 (bit move)
"FMVWX": {0x78, 0x0, 0x0, 0x53}, // int32 → float32 (bit move)
// The toolchain's W/D suffix spellings of the same moves.
"FMVXS": {0x70, 0x0, 0x53},
"FMVFS": {0x78, 0x0, 0x53},
"FMVSX": {0x79, 0x0, 0x53},
"FMVXS": {0x70, 0x0, 0x0, 0x53},
"FMVFS": {0x78, 0x0, 0x0, 0x53},
"FMVSX": {0x78, 0x0, 0x0, 0x53},
}
// riscvCvtType encodes an FP conversion instruction.
// Layout: funct7 | rs2(convtype) | rs1 | funct3(0) | rd | opcode.
// Layout: funct7 | rs2(convtype) | rs1 | funct3(rm) | rd | opcode.
func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 {
return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) |
(uint32(rd) << 7) | enc.opcode
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
}
// R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD).
@@ -455,12 +490,16 @@ type riscvFmaEnc struct {
var riscvFmaTable = map[string]riscvFmaEnc{
"FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3
"FMADDD": {0x1, 0x43},
"FMADDQ": {0x3, 0x43},
"FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3
"FMSUBD": {0x1, 0x47},
"FMSUBQ": {0x3, 0x47},
"FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3
"FNMSUBD": {0x1, 0x4B},
"FNMSUBQ": {0x3, 0x4B},
"FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3
"FNMADDD": {0x1, 0x4F},
"FNMADDQ": {0x3, 0x4F},
}
// riscvFmaType encodes an R4-type fused multiply-add instruction.