feat(asm): encode the riscv64 quad-precision family and fix the fp cvt paths
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
955bc6643e
commit
9b751f6e58
4 files changed
+331
-50
No files matched your search
+83
-44
@@ -344,22 +344,46 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
// FP loads/stores.
|
||||
"FLW": {0x07, 0x2, 0x00},
|
||||
"FLD": {0x07, 0x3, 0x00},
|
||||
"FLQ": {0x07, 0x4, 0x00},
|
||||
"FSW": {0x27, 0x2, 0x00},
|
||||
"FSD": {0x27, 0x3, 0x00},
|
||||
"FSQ": {0x27, 0x4, 0x00},
|
||||
// FP min/max.
|
||||
"FMINS": {0x53, 0x0, 0x14},
|
||||
"FMAXS": {0x53, 0x1, 0x14},
|
||||
"FMIND": {0x53, 0x0, 0x15},
|
||||
"FMAXD": {0x53, 0x1, 0x15},
|
||||
// FP sign injection (double): rs2 carries the sign source.
|
||||
// FP compare: the integer destination rides in rd (the 3-op R-type
|
||||
// path writes it there).
|
||||
"FEQS": {0x53, 0x2, 0x50},
|
||||
"FLTS": {0x53, 0x1, 0x50},
|
||||
"FLES": {0x53, 0x0, 0x50},
|
||||
"FEQD": {0x53, 0x2, 0x51},
|
||||
"FLTD": {0x53, 0x1, 0x51},
|
||||
"FLED": {0x53, 0x0, 0x51},
|
||||
"FEQQ": {0x53, 0x2, 0x53},
|
||||
"FLTQ": {0x53, 0x1, 0x53},
|
||||
"FLEQ": {0x53, 0x0, 0x53},
|
||||
// FP sign injection: the sign source rides in rs2; the XOR form's
|
||||
// funct3 is 2 in every width.
|
||||
"FSGNJD": {0x53, 0x0, 0x11},
|
||||
"FSGNJS": {0x53, 0x0, 0x10},
|
||||
"FSGNJX": {0x53, 0x0, 0x14},
|
||||
"FSGNJXD": {0x53, 0x0, 0x15},
|
||||
"FSGNJXS": {0x53, 0x0, 0x14},
|
||||
"FSGNJXD": {0x53, 0x2, 0x11},
|
||||
"FSGNJXS": {0x53, 0x2, 0x10},
|
||||
"FSGNJND": {0x53, 0x1, 0x11},
|
||||
"FSGNJNS": {0x53, 0x1, 0x10},
|
||||
"FSGNJNX": {0x53, 0x1, 0x14},
|
||||
"FSGNJQ": {0x53, 0x0, 0x13},
|
||||
"FSGNJNQ": {0x53, 0x1, 0x13},
|
||||
"FSGNJXQ": {0x53, 0x2, 0x13},
|
||||
// RV64Q, quad-precision arithmetic (the fmt field rides in funct7's
|
||||
// low bits: 11 for quad).
|
||||
"FADDQ": {0x53, 0x0, 0x03},
|
||||
"FSUBQ": {0x53, 0x0, 0x07},
|
||||
"FMULQ": {0x53, 0x0, 0x0B},
|
||||
"FDIVQ": {0x53, 0x0, 0x0F},
|
||||
"FSQRTQ": {0x53, 0x0, 0x2F},
|
||||
"FMINQ": {0x53, 0x0, 0x17},
|
||||
"FMAXQ": {0x53, 0x1, 0x17},
|
||||
|
||||
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||
// The toolchain gives LR acquire ordering (aq = 1) and SC release
|
||||
@@ -368,14 +392,6 @@ var riscvInstrTable = map[string]riscvEnc{
|
||||
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
|
||||
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
|
||||
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
|
||||
|
||||
// FP compare, result in integer register (funct7 0x50/0x51).
|
||||
"FEQS": {0x53, 0x2, 0x50},
|
||||
"FLTS": {0x53, 0x1, 0x50},
|
||||
"FLES": {0x53, 0x0, 0x50},
|
||||
"FEQD": {0x53, 0x2, 0x51},
|
||||
"FLTD": {0x53, 0x1, 0x51},
|
||||
"FLED": {0x53, 0x0, 0x51},
|
||||
}
|
||||
|
||||
// riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode.
|
||||
@@ -399,49 +415,68 @@ func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||
type riscvCvtEnc struct {
|
||||
funct7 uint32 // bits [31:25]
|
||||
rs2 uint32 // conversion-type code in bits [24:20]
|
||||
funct3 uint32 // the rounding mode or the fclass marker, bits [14:12]
|
||||
opcode uint32 // always 0x53 (OP-FP)
|
||||
}
|
||||
|
||||
var riscvCvtTable = map[string]riscvCvtEnc{
|
||||
// float → int (rs2 selects the integer width/sign).
|
||||
"FCVTWS": {0x60, 0x0, 0x53}, // float32 → int32
|
||||
"FCVTWUS": {0x60, 0x1, 0x53}, // float32 → uint32
|
||||
"FCVTLS": {0x60, 0x2, 0x53}, // float32 → int64
|
||||
"FCVTLUS": {0x60, 0x3, 0x53}, // float32 → uint64
|
||||
"FCVTWD": {0x61, 0x0, 0x53}, // float64 → int32
|
||||
"FCVTWUD": {0x61, 0x1, 0x53}, // float64 → uint32
|
||||
"FCVTLD": {0x61, 0x2, 0x53}, // float64 → int64
|
||||
"FCVTLUD": {0x61, 0x3, 0x53}, // float64 → uint64
|
||||
// int → float (rs2 selects the integer width/sign).
|
||||
"FCVTSW": {0x68, 0x0, 0x53}, // int32 → float32
|
||||
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
||||
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
||||
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
||||
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
|
||||
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
|
||||
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
||||
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
||||
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
||||
"FCVTDLU": {0x69, 0x3, 0x53}, // uint64 → float64
|
||||
// float → int (rs2 selects the integer width/sign). The bare forms
|
||||
// carry the specification's default rounding mode RTZ in funct3; the
|
||||
// suffixed spellings override it.
|
||||
"FCVTWS": {0x60, 0x0, 0x1, 0x53}, // float32 → int32
|
||||
"FCVTWUS": {0x60, 0x1, 0x1, 0x53}, // float32 → uint32
|
||||
"FCVTLS": {0x60, 0x2, 0x1, 0x53}, // float32 → int64
|
||||
"FCVTLUS": {0x60, 0x3, 0x1, 0x53}, // float32 → uint64
|
||||
"FCVTWD": {0x61, 0x0, 0x1, 0x53}, // float64 → int32
|
||||
"FCVTWUD": {0x61, 0x1, 0x1, 0x53}, // float64 → uint32
|
||||
"FCVTLD": {0x61, 0x2, 0x1, 0x53}, // float64 → int64
|
||||
"FCVTLUD": {0x61, 0x3, 0x1, 0x53}, // float64 → uint64
|
||||
// int → float (rs2 selects the integer width/sign): the default
|
||||
// rounding mode RNE keeps funct3 0.
|
||||
"FCVTSW": {0x68, 0x0, 0x0, 0x53}, // int32 → float32
|
||||
"FCVTSWU": {0x68, 0x1, 0x0, 0x53}, // uint32 → float32
|
||||
"FCVTSL": {0x68, 0x2, 0x0, 0x53}, // int64 → float32
|
||||
"FCVTSLU": {0x68, 0x3, 0x0, 0x53}, // uint64 → float32
|
||||
"FCLASSS": {0x70, 0x0, 0x1, 0x53}, // classify float32 → GPR mask
|
||||
"FCLASSD": {0x71, 0x0, 0x1, 0x53}, // classify float64 → GPR mask
|
||||
"FCVTDW": {0x69, 0x0, 0x0, 0x53}, // int32 → float64
|
||||
"FCVTDWU": {0x69, 0x1, 0x0, 0x53}, // uint32 → float64
|
||||
"FCVTDL": {0x69, 0x2, 0x0, 0x53}, // int64 → float64
|
||||
"FCVTDLU": {0x69, 0x3, 0x0, 0x53}, // uint64 → float64
|
||||
// float → float width conversion.
|
||||
"FCVTSD": {0x20, 0x1, 0x53}, // float64 → float32
|
||||
"FCVTDS": {0x21, 0x0, 0x53}, // float32 → float64
|
||||
"FCVTSD": {0x20, 0x1, 0x0, 0x53}, // float64 → float32
|
||||
"FCVTDS": {0x21, 0x0, 0x0, 0x53}, // float32 → float64
|
||||
// Quad-precision conversions: the fmt field rides in funct7's low
|
||||
// bits (11 for quad), the other width in rs2 where one is needed.
|
||||
"FCVTSQ": {0x20, 0x3, 0x0, 0x53}, // quad → float32
|
||||
"FCVTDQ": {0x21, 0x3, 0x0, 0x53}, // quad → float64
|
||||
"FCVTQS": {0x23, 0x0, 0x0, 0x53}, // float32 → quad
|
||||
"FCVTQD": {0x23, 0x1, 0x0, 0x53}, // float64 → quad
|
||||
"FCVTWQ": {0x63, 0x0, 0x1, 0x53}, // quad → int32
|
||||
"FCVTWUQ": {0x63, 0x1, 0x1, 0x53}, // quad → uint32
|
||||
"FCVTLQ": {0x63, 0x2, 0x1, 0x53}, // quad → int64
|
||||
"FCVTLUQ": {0x63, 0x3, 0x1, 0x53}, // quad → uint64
|
||||
"FCVTQW": {0x6B, 0x0, 0x0, 0x53}, // int32 → quad
|
||||
"FCVTQWU": {0x6B, 0x1, 0x0, 0x53}, // uint32 → quad
|
||||
"FCVTQL": {0x6B, 0x2, 0x0, 0x53}, // int64 → quad
|
||||
"FCVTQLU": {0x6B, 0x3, 0x0, 0x53}, // uint64 → quad
|
||||
"FCLASSQ": {0x73, 0x0, 0x1, 0x53}, // classify quad → GPR mask
|
||||
// Bit moves between integer and FP registers (no conversion).
|
||||
"FMVXD": {0x71, 0x0, 0x53}, // float64 → int64 (bit move)
|
||||
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
||||
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
||||
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
||||
"FMVXD": {0x71, 0x0, 0x0, 0x53}, // float64 → int64 (bit move)
|
||||
"FMVDX": {0x79, 0x0, 0x0, 0x53}, // int64 → float64 (bit move)
|
||||
"FMVXW": {0x70, 0x0, 0x0, 0x53}, // float32 → int32 (bit move)
|
||||
"FMVWX": {0x78, 0x0, 0x0, 0x53}, // int32 → float32 (bit move)
|
||||
// The toolchain's W/D suffix spellings of the same moves.
|
||||
"FMVXS": {0x70, 0x0, 0x53},
|
||||
"FMVFS": {0x78, 0x0, 0x53},
|
||||
"FMVSX": {0x79, 0x0, 0x53},
|
||||
"FMVXS": {0x70, 0x0, 0x0, 0x53},
|
||||
"FMVFS": {0x78, 0x0, 0x0, 0x53},
|
||||
"FMVSX": {0x78, 0x0, 0x0, 0x53},
|
||||
}
|
||||
|
||||
// riscvCvtType encodes an FP conversion instruction.
|
||||
// Layout: funct7 | rs2(convtype) | rs1 | funct3(0) | rd | opcode.
|
||||
// Layout: funct7 | rs2(convtype) | rs1 | funct3(rm) | rd | opcode.
|
||||
func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 {
|
||||
return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) |
|
||||
(uint32(rd) << 7) | enc.opcode
|
||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||
}
|
||||
|
||||
// R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD).
|
||||
@@ -455,12 +490,16 @@ type riscvFmaEnc struct {
|
||||
var riscvFmaTable = map[string]riscvFmaEnc{
|
||||
"FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3
|
||||
"FMADDD": {0x1, 0x43},
|
||||
"FMADDQ": {0x3, 0x43},
|
||||
"FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3
|
||||
"FMSUBD": {0x1, 0x47},
|
||||
"FMSUBQ": {0x3, 0x47},
|
||||
"FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3
|
||||
"FNMSUBD": {0x1, 0x4B},
|
||||
"FNMSUBQ": {0x3, 0x4B},
|
||||
"FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3
|
||||
"FNMADDD": {0x1, 0x4F},
|
||||
"FNMADDQ": {0x3, 0x4F},
|
||||
}
|
||||
|
||||
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
||||
|
||||
Reference in new issue
Block a user