feat(asm): encode the riscv64 quad-precision family and fix the fp cvt paths
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
955bc6643e
commit
9b751f6e58
4 files changed
+331
-50
No files matched your search
+10
-6
@@ -1368,7 +1368,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
|||||||
|
|
||||||
// FP conversions with an explicit rounding mode: FCVTWS.RNE and friends
|
// FP conversions with an explicit rounding mode: FCVTWS.RNE and friends
|
||||||
// suffix the base mnemonic with RNE/RTZ/RDN/RUP/RMM, which lands in the
|
// suffix the base mnemonic with RNE/RTZ/RDN/RUP/RMM, which lands in the
|
||||||
// low three bits of the funct7 field.
|
// funct3 field, replacing the bare form's default.
|
||||||
if i := strings.IndexByte(mnem, '.'); i > 0 {
|
if i := strings.IndexByte(mnem, '.'); i > 0 {
|
||||||
if base, ok := riscvCvtTable[mnem[:i]]; ok {
|
if base, ok := riscvCvtTable[mnem[:i]]; ok {
|
||||||
rm, ok := riscvRoundModes[mnem[i+1:]]
|
rm, ok := riscvRoundModes[mnem[i+1:]]
|
||||||
@@ -1383,7 +1383,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
|||||||
if rd < 0 || rs1 < 0 {
|
if rd < 0 || rs1 < 0 {
|
||||||
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
||||||
}
|
}
|
||||||
base.funct7 = (base.funct7 &^ 7) | rm
|
base.funct3 = uint32(rm)
|
||||||
word := riscvCvtType(base, rd, rs1)
|
word := riscvCvtType(base, rd, rs1)
|
||||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||||
}
|
}
|
||||||
@@ -4653,18 +4653,21 @@ func isFPArithInstr(m string) bool {
|
|||||||
case "FADDS", "FSUBS", "FMULS", "FDIVS",
|
case "FADDS", "FSUBS", "FMULS", "FDIVS",
|
||||||
"FADDD", "FSUBD", "FMULD", "FDIVD",
|
"FADDD", "FSUBD", "FMULD", "FDIVD",
|
||||||
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD",
|
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD",
|
||||||
"FSGNJS", "FSGNJX", "FSGNJXD", "FSGNJXS", "FSGNJND", "FSGNJNS", "FSGNJNX":
|
"FSGNJS", "FSGNJXD", "FSGNJXS", "FSGNJND", "FSGNJNS",
|
||||||
|
"FADDQ", "FSUBQ", "FMULQ", "FDIVQ",
|
||||||
|
"FSQRTQ", "FMINQ", "FMAXQ", "FSGNJQ",
|
||||||
|
"FSGNJXQ", "FSGNJNQ":
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func isFPLoadInstr(m string) bool {
|
func isFPLoadInstr(m string) bool {
|
||||||
return m == "FLW" || m == "FLD"
|
return m == "FLW" || m == "FLD" || m == "FLQ"
|
||||||
}
|
}
|
||||||
|
|
||||||
func isFPStoreInstr(m string) bool {
|
func isFPStoreInstr(m string) bool {
|
||||||
return m == "FSW" || m == "FSD"
|
return m == "FSW" || m == "FSD" || m == "FSQ"
|
||||||
}
|
}
|
||||||
|
|
||||||
func isLRInstr(m string) bool {
|
func isLRInstr(m string) bool {
|
||||||
@@ -4677,7 +4680,8 @@ func isSCInstr(m string) bool {
|
|||||||
|
|
||||||
func isFPCmpInstr(m string) bool {
|
func isFPCmpInstr(m string) bool {
|
||||||
switch m {
|
switch m {
|
||||||
case "FEQS", "FLTS", "FLES", "FEQD", "FLTD", "FLED":
|
case "FEQS", "FLTS", "FLES", "FEQD", "FLTD", "FLED",
|
||||||
|
"FEQQ", "FLTQ", "FLEQ":
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
return false
|
return false
|
||||||
|
|||||||
+83
-44
@@ -344,22 +344,46 @@ var riscvInstrTable = map[string]riscvEnc{
|
|||||||
// FP loads/stores.
|
// FP loads/stores.
|
||||||
"FLW": {0x07, 0x2, 0x00},
|
"FLW": {0x07, 0x2, 0x00},
|
||||||
"FLD": {0x07, 0x3, 0x00},
|
"FLD": {0x07, 0x3, 0x00},
|
||||||
|
"FLQ": {0x07, 0x4, 0x00},
|
||||||
"FSW": {0x27, 0x2, 0x00},
|
"FSW": {0x27, 0x2, 0x00},
|
||||||
"FSD": {0x27, 0x3, 0x00},
|
"FSD": {0x27, 0x3, 0x00},
|
||||||
|
"FSQ": {0x27, 0x4, 0x00},
|
||||||
// FP min/max.
|
// FP min/max.
|
||||||
"FMINS": {0x53, 0x0, 0x14},
|
"FMINS": {0x53, 0x0, 0x14},
|
||||||
"FMAXS": {0x53, 0x1, 0x14},
|
"FMAXS": {0x53, 0x1, 0x14},
|
||||||
"FMIND": {0x53, 0x0, 0x15},
|
"FMIND": {0x53, 0x0, 0x15},
|
||||||
"FMAXD": {0x53, 0x1, 0x15},
|
"FMAXD": {0x53, 0x1, 0x15},
|
||||||
// FP sign injection (double): rs2 carries the sign source.
|
// FP compare: the integer destination rides in rd (the 3-op R-type
|
||||||
|
// path writes it there).
|
||||||
|
"FEQS": {0x53, 0x2, 0x50},
|
||||||
|
"FLTS": {0x53, 0x1, 0x50},
|
||||||
|
"FLES": {0x53, 0x0, 0x50},
|
||||||
|
"FEQD": {0x53, 0x2, 0x51},
|
||||||
|
"FLTD": {0x53, 0x1, 0x51},
|
||||||
|
"FLED": {0x53, 0x0, 0x51},
|
||||||
|
"FEQQ": {0x53, 0x2, 0x53},
|
||||||
|
"FLTQ": {0x53, 0x1, 0x53},
|
||||||
|
"FLEQ": {0x53, 0x0, 0x53},
|
||||||
|
// FP sign injection: the sign source rides in rs2; the XOR form's
|
||||||
|
// funct3 is 2 in every width.
|
||||||
"FSGNJD": {0x53, 0x0, 0x11},
|
"FSGNJD": {0x53, 0x0, 0x11},
|
||||||
"FSGNJS": {0x53, 0x0, 0x10},
|
"FSGNJS": {0x53, 0x0, 0x10},
|
||||||
"FSGNJX": {0x53, 0x0, 0x14},
|
"FSGNJXD": {0x53, 0x2, 0x11},
|
||||||
"FSGNJXD": {0x53, 0x0, 0x15},
|
"FSGNJXS": {0x53, 0x2, 0x10},
|
||||||
"FSGNJXS": {0x53, 0x0, 0x14},
|
|
||||||
"FSGNJND": {0x53, 0x1, 0x11},
|
"FSGNJND": {0x53, 0x1, 0x11},
|
||||||
"FSGNJNS": {0x53, 0x1, 0x10},
|
"FSGNJNS": {0x53, 0x1, 0x10},
|
||||||
"FSGNJNX": {0x53, 0x1, 0x14},
|
"FSGNJQ": {0x53, 0x0, 0x13},
|
||||||
|
"FSGNJNQ": {0x53, 0x1, 0x13},
|
||||||
|
"FSGNJXQ": {0x53, 0x2, 0x13},
|
||||||
|
// RV64Q, quad-precision arithmetic (the fmt field rides in funct7's
|
||||||
|
// low bits: 11 for quad).
|
||||||
|
"FADDQ": {0x53, 0x0, 0x03},
|
||||||
|
"FSUBQ": {0x53, 0x0, 0x07},
|
||||||
|
"FMULQ": {0x53, 0x0, 0x0B},
|
||||||
|
"FDIVQ": {0x53, 0x0, 0x0F},
|
||||||
|
"FSQRTQ": {0x53, 0x0, 0x2F},
|
||||||
|
"FMINQ": {0x53, 0x0, 0x17},
|
||||||
|
"FMAXQ": {0x53, 0x1, 0x17},
|
||||||
|
|
||||||
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||||
// The toolchain gives LR acquire ordering (aq = 1) and SC release
|
// The toolchain gives LR acquire ordering (aq = 1) and SC release
|
||||||
@@ -368,14 +392,6 @@ var riscvInstrTable = map[string]riscvEnc{
|
|||||||
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
|
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
|
||||||
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
|
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
|
||||||
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
|
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
|
||||||
|
|
||||||
// FP compare, result in integer register (funct7 0x50/0x51).
|
|
||||||
"FEQS": {0x53, 0x2, 0x50},
|
|
||||||
"FLTS": {0x53, 0x1, 0x50},
|
|
||||||
"FLES": {0x53, 0x0, 0x50},
|
|
||||||
"FEQD": {0x53, 0x2, 0x51},
|
|
||||||
"FLTD": {0x53, 0x1, 0x51},
|
|
||||||
"FLED": {0x53, 0x0, 0x51},
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode.
|
// riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode.
|
||||||
@@ -399,49 +415,68 @@ func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
|||||||
type riscvCvtEnc struct {
|
type riscvCvtEnc struct {
|
||||||
funct7 uint32 // bits [31:25]
|
funct7 uint32 // bits [31:25]
|
||||||
rs2 uint32 // conversion-type code in bits [24:20]
|
rs2 uint32 // conversion-type code in bits [24:20]
|
||||||
|
funct3 uint32 // the rounding mode or the fclass marker, bits [14:12]
|
||||||
opcode uint32 // always 0x53 (OP-FP)
|
opcode uint32 // always 0x53 (OP-FP)
|
||||||
}
|
}
|
||||||
|
|
||||||
var riscvCvtTable = map[string]riscvCvtEnc{
|
var riscvCvtTable = map[string]riscvCvtEnc{
|
||||||
// float → int (rs2 selects the integer width/sign).
|
// float → int (rs2 selects the integer width/sign). The bare forms
|
||||||
"FCVTWS": {0x60, 0x0, 0x53}, // float32 → int32
|
// carry the specification's default rounding mode RTZ in funct3; the
|
||||||
"FCVTWUS": {0x60, 0x1, 0x53}, // float32 → uint32
|
// suffixed spellings override it.
|
||||||
"FCVTLS": {0x60, 0x2, 0x53}, // float32 → int64
|
"FCVTWS": {0x60, 0x0, 0x1, 0x53}, // float32 → int32
|
||||||
"FCVTLUS": {0x60, 0x3, 0x53}, // float32 → uint64
|
"FCVTWUS": {0x60, 0x1, 0x1, 0x53}, // float32 → uint32
|
||||||
"FCVTWD": {0x61, 0x0, 0x53}, // float64 → int32
|
"FCVTLS": {0x60, 0x2, 0x1, 0x53}, // float32 → int64
|
||||||
"FCVTWUD": {0x61, 0x1, 0x53}, // float64 → uint32
|
"FCVTLUS": {0x60, 0x3, 0x1, 0x53}, // float32 → uint64
|
||||||
"FCVTLD": {0x61, 0x2, 0x53}, // float64 → int64
|
"FCVTWD": {0x61, 0x0, 0x1, 0x53}, // float64 → int32
|
||||||
"FCVTLUD": {0x61, 0x3, 0x53}, // float64 → uint64
|
"FCVTWUD": {0x61, 0x1, 0x1, 0x53}, // float64 → uint32
|
||||||
// int → float (rs2 selects the integer width/sign).
|
"FCVTLD": {0x61, 0x2, 0x1, 0x53}, // float64 → int64
|
||||||
"FCVTSW": {0x68, 0x0, 0x53}, // int32 → float32
|
"FCVTLUD": {0x61, 0x3, 0x1, 0x53}, // float64 → uint64
|
||||||
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
// int → float (rs2 selects the integer width/sign): the default
|
||||||
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
// rounding mode RNE keeps funct3 0.
|
||||||
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
"FCVTSW": {0x68, 0x0, 0x0, 0x53}, // int32 → float32
|
||||||
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
|
"FCVTSWU": {0x68, 0x1, 0x0, 0x53}, // uint32 → float32
|
||||||
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
|
"FCVTSL": {0x68, 0x2, 0x0, 0x53}, // int64 → float32
|
||||||
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
"FCVTSLU": {0x68, 0x3, 0x0, 0x53}, // uint64 → float32
|
||||||
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
"FCLASSS": {0x70, 0x0, 0x1, 0x53}, // classify float32 → GPR mask
|
||||||
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
"FCLASSD": {0x71, 0x0, 0x1, 0x53}, // classify float64 → GPR mask
|
||||||
"FCVTDLU": {0x69, 0x3, 0x53}, // uint64 → float64
|
"FCVTDW": {0x69, 0x0, 0x0, 0x53}, // int32 → float64
|
||||||
|
"FCVTDWU": {0x69, 0x1, 0x0, 0x53}, // uint32 → float64
|
||||||
|
"FCVTDL": {0x69, 0x2, 0x0, 0x53}, // int64 → float64
|
||||||
|
"FCVTDLU": {0x69, 0x3, 0x0, 0x53}, // uint64 → float64
|
||||||
// float → float width conversion.
|
// float → float width conversion.
|
||||||
"FCVTSD": {0x20, 0x1, 0x53}, // float64 → float32
|
"FCVTSD": {0x20, 0x1, 0x0, 0x53}, // float64 → float32
|
||||||
"FCVTDS": {0x21, 0x0, 0x53}, // float32 → float64
|
"FCVTDS": {0x21, 0x0, 0x0, 0x53}, // float32 → float64
|
||||||
|
// Quad-precision conversions: the fmt field rides in funct7's low
|
||||||
|
// bits (11 for quad), the other width in rs2 where one is needed.
|
||||||
|
"FCVTSQ": {0x20, 0x3, 0x0, 0x53}, // quad → float32
|
||||||
|
"FCVTDQ": {0x21, 0x3, 0x0, 0x53}, // quad → float64
|
||||||
|
"FCVTQS": {0x23, 0x0, 0x0, 0x53}, // float32 → quad
|
||||||
|
"FCVTQD": {0x23, 0x1, 0x0, 0x53}, // float64 → quad
|
||||||
|
"FCVTWQ": {0x63, 0x0, 0x1, 0x53}, // quad → int32
|
||||||
|
"FCVTWUQ": {0x63, 0x1, 0x1, 0x53}, // quad → uint32
|
||||||
|
"FCVTLQ": {0x63, 0x2, 0x1, 0x53}, // quad → int64
|
||||||
|
"FCVTLUQ": {0x63, 0x3, 0x1, 0x53}, // quad → uint64
|
||||||
|
"FCVTQW": {0x6B, 0x0, 0x0, 0x53}, // int32 → quad
|
||||||
|
"FCVTQWU": {0x6B, 0x1, 0x0, 0x53}, // uint32 → quad
|
||||||
|
"FCVTQL": {0x6B, 0x2, 0x0, 0x53}, // int64 → quad
|
||||||
|
"FCVTQLU": {0x6B, 0x3, 0x0, 0x53}, // uint64 → quad
|
||||||
|
"FCLASSQ": {0x73, 0x0, 0x1, 0x53}, // classify quad → GPR mask
|
||||||
// Bit moves between integer and FP registers (no conversion).
|
// Bit moves between integer and FP registers (no conversion).
|
||||||
"FMVXD": {0x71, 0x0, 0x53}, // float64 → int64 (bit move)
|
"FMVXD": {0x71, 0x0, 0x0, 0x53}, // float64 → int64 (bit move)
|
||||||
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
"FMVDX": {0x79, 0x0, 0x0, 0x53}, // int64 → float64 (bit move)
|
||||||
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
"FMVXW": {0x70, 0x0, 0x0, 0x53}, // float32 → int32 (bit move)
|
||||||
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
"FMVWX": {0x78, 0x0, 0x0, 0x53}, // int32 → float32 (bit move)
|
||||||
// The toolchain's W/D suffix spellings of the same moves.
|
// The toolchain's W/D suffix spellings of the same moves.
|
||||||
"FMVXS": {0x70, 0x0, 0x53},
|
"FMVXS": {0x70, 0x0, 0x0, 0x53},
|
||||||
"FMVFS": {0x78, 0x0, 0x53},
|
"FMVFS": {0x78, 0x0, 0x0, 0x53},
|
||||||
"FMVSX": {0x79, 0x0, 0x53},
|
"FMVSX": {0x78, 0x0, 0x0, 0x53},
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvCvtType encodes an FP conversion instruction.
|
// riscvCvtType encodes an FP conversion instruction.
|
||||||
// Layout: funct7 | rs2(convtype) | rs1 | funct3(0) | rd | opcode.
|
// Layout: funct7 | rs2(convtype) | rs1 | funct3(rm) | rd | opcode.
|
||||||
func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 {
|
func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 {
|
||||||
return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) |
|
return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) |
|
||||||
(uint32(rd) << 7) | enc.opcode
|
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||||
}
|
}
|
||||||
|
|
||||||
// R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD).
|
// R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD).
|
||||||
@@ -455,12 +490,16 @@ type riscvFmaEnc struct {
|
|||||||
var riscvFmaTable = map[string]riscvFmaEnc{
|
var riscvFmaTable = map[string]riscvFmaEnc{
|
||||||
"FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3
|
"FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3
|
||||||
"FMADDD": {0x1, 0x43},
|
"FMADDD": {0x1, 0x43},
|
||||||
|
"FMADDQ": {0x3, 0x43},
|
||||||
"FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3
|
"FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3
|
||||||
"FMSUBD": {0x1, 0x47},
|
"FMSUBD": {0x1, 0x47},
|
||||||
|
"FMSUBQ": {0x3, 0x47},
|
||||||
"FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3
|
"FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3
|
||||||
"FNMSUBD": {0x1, 0x4B},
|
"FNMSUBD": {0x1, 0x4B},
|
||||||
|
"FNMSUBQ": {0x3, 0x4B},
|
||||||
"FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3
|
"FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3
|
||||||
"FNMADDD": {0x1, 0x4F},
|
"FNMADDD": {0x1, 0x4F},
|
||||||
|
"FNMADDQ": {0x3, 0x4F},
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
||||||
|
|||||||
@@ -0,0 +1,86 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import "testing"
|
||||||
|
|
||||||
|
// TestRISCVQuadFP_golden pins the quad-precision ("Q") family on golden
|
||||||
|
// vectors from the specification. The Go toolchain's object table carries
|
||||||
|
// the encodings but its assembler accepts no Q mnemonic, so no oracle run
|
||||||
|
// is possible: the words below follow the Q extension's encoding table with
|
||||||
|
// the Plan 9 operand order the other widths use (first operand in rs2, the
|
||||||
|
// bare float-to-integer conversions carrying the RTZ rounding mode).
|
||||||
|
func TestRISCVQuadFP_golden(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·q(SB), NOSPLIT, $0
|
||||||
|
FLQ 8(X5), F3
|
||||||
|
FSQ F3, 8(X5)
|
||||||
|
FADDQ F1, F2, F3
|
||||||
|
FSUBQ F1, F2, F3
|
||||||
|
FMULQ F1, F2, F3
|
||||||
|
FDIVQ F1, F2, F3
|
||||||
|
FSQRTQ F2, F1
|
||||||
|
FMINQ F1, F2, F3
|
||||||
|
FMAXQ F1, F2, F3
|
||||||
|
FEQQ F3, F2, X5
|
||||||
|
FLTQ F3, F2, X5
|
||||||
|
FLEQ F3, F2, X5
|
||||||
|
FSGNJQ F1, F2, F3
|
||||||
|
FSGNJNQ F1, F2, F3
|
||||||
|
FSGNJXQ F1, F2, F3
|
||||||
|
FMADDQ F1, F2, F3, F4
|
||||||
|
FMSUBQ F1, F2, F3, F4
|
||||||
|
FNMSUBQ F1, F2, F3, F4
|
||||||
|
FNMADDQ F1, F2, F3, F4
|
||||||
|
FCVTSQ F5, F2
|
||||||
|
FCVTDQ F5, F2
|
||||||
|
FCVTQS F2, F5
|
||||||
|
FCVTQD F2, F5
|
||||||
|
FCVTWQ F2, X5
|
||||||
|
FCVTWUQ F2, X5
|
||||||
|
FCVTLQ F2, X5
|
||||||
|
FCVTLUQ F2, X5
|
||||||
|
FCVTQW X5, F2
|
||||||
|
FCVTQWU X5, F2
|
||||||
|
FCVTQL X5, F2
|
||||||
|
FCVTQLU X5, F2
|
||||||
|
FCLASSQ F2, X5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
riscvWants(t, code,
|
||||||
|
0x0082C187, // flq f3, 8(x5)
|
||||||
|
0x0032C427, // fsq f3, 8(x5)
|
||||||
|
0x061101D3, // fadd.q f3, f2, f1
|
||||||
|
0x0E1101D3, // fsub.q
|
||||||
|
0x161101D3, // fmul.q
|
||||||
|
0x1E1101D3, // fdiv.q
|
||||||
|
0x5E0100D3, // fsqrt.q f1, f2
|
||||||
|
0x2E1101D3, // fmin.q f3, f2, f1
|
||||||
|
0x2E1111D3, // fmax.q
|
||||||
|
0xA63122D3, // feq.q x5, f2, f3
|
||||||
|
0xA63112D3, // flt.q
|
||||||
|
0xA63102D3, // fle.q
|
||||||
|
0x261101D3, // fsgnj.q
|
||||||
|
0x261111D3, // fsgnjn.q
|
||||||
|
0x261121D3, // fsgnjx.q
|
||||||
|
0x1E208243, // fmadd.q f4, f1, f2, f3
|
||||||
|
0x1E208247, // fmsub.q
|
||||||
|
0x1E20824B, // fnmsub.q
|
||||||
|
0x1E20824F, // fnmadd.q
|
||||||
|
0x40328153, // fcvt.s.q f2, f5
|
||||||
|
0x42328153, // fcvt.d.q f2, f5
|
||||||
|
0x460102D3, // fcvt.q.s f5, f2
|
||||||
|
0x461102D3, // fcvt.q.d f5, f2
|
||||||
|
0xC60112D3, // fcvt.w.q x5, f2
|
||||||
|
0xC61112D3, // fcvt.wu.q
|
||||||
|
0xC62112D3, // fcvt.l.q
|
||||||
|
0xC63112D3, // fcvt.lu.q
|
||||||
|
0xD6028153, // fcvt.q.w f2, x5
|
||||||
|
0xD6128153, // fcvt.q.wu
|
||||||
|
0xD6228153, // fcvt.q.l
|
||||||
|
0xD6328153, // fcvt.q.lu
|
||||||
|
0xE60112D3, // fclass.q x5, f2
|
||||||
|
)
|
||||||
|
}
|
||||||
@@ -0,0 +1,152 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestRISCVScalarFP_Differential proves the scalar single- and
|
||||||
|
// double-precision families against the toolchain: sections 21.6 through
|
||||||
|
// 22.7 of the "F" and "D" specifications' computational, conversion, move,
|
||||||
|
// sign-injection, compare and classify instructions in the toolchain's own
|
||||||
|
// testdata wording, assembled by gasm and by go tool asm must agree word
|
||||||
|
// for word.
|
||||||
|
func TestRISCVScalarFP_Differential(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·fp(SB), NOSPLIT, $0
|
||||||
|
|
||||||
|
// 21.6: Single-Precision Floating-Point Computational Instructions
|
||||||
|
FADDS F1, F0, F2 // 53011000
|
||||||
|
FSUBS F1, F0, F2 // 53011008
|
||||||
|
FMULS F1, F0, F2 // 53011010
|
||||||
|
FDIVS F1, F0, F2 // 53011018
|
||||||
|
FMINS F1, F0, F2 // 53011028
|
||||||
|
FMAXS F1, F0, F2 // 53111028
|
||||||
|
FSQRTS F0, F1 // d3000058
|
||||||
|
|
||||||
|
// 21.7: Single-Precision Floating-Point Conversion and Move Instructions
|
||||||
|
FCVTWS F0, X5 // d31200c0
|
||||||
|
FCVTWS.RNE F0, X5 // d30200c0
|
||||||
|
FCVTWS.RTZ F0, X5 // d31200c0
|
||||||
|
FCVTWS.RDN F0, X5 // d32200c0
|
||||||
|
FCVTWS.RUP F0, X5 // d33200c0
|
||||||
|
FCVTWS.RMM F0, X5 // d34200c0
|
||||||
|
FCVTLS F0, X5 // d31220c0
|
||||||
|
FCVTLS.RNE F0, X5 // d30220c0
|
||||||
|
FCVTLS.RTZ F0, X5 // d31220c0
|
||||||
|
FCVTLS.RDN F0, X5 // d32220c0
|
||||||
|
FCVTLS.RUP F0, X5 // d33220c0
|
||||||
|
FCVTLS.RMM F0, X5 // d34220c0
|
||||||
|
FCVTSW X5, F0 // 538002d0
|
||||||
|
FCVTSL X5, F0 // 538022d0
|
||||||
|
FCVTWUS F0, X5 // d31210c0
|
||||||
|
FCVTWUS.RNE F0, X5 // d30210c0
|
||||||
|
FCVTWUS.RTZ F0, X5 // d31210c0
|
||||||
|
FCVTWUS.RDN F0, X5 // d32210c0
|
||||||
|
FCVTWUS.RUP F0, X5 // d33210c0
|
||||||
|
FCVTWUS.RMM F0, X5 // d34210c0
|
||||||
|
FCVTLUS F0, X5 // d31230c0
|
||||||
|
FCVTLUS.RNE F0, X5 // d30230c0
|
||||||
|
FCVTLUS.RTZ F0, X5 // d31230c0
|
||||||
|
FCVTLUS.RDN F0, X5 // d32230c0
|
||||||
|
FCVTLUS.RUP F0, X5 // d33230c0
|
||||||
|
FCVTLUS.RMM F0, X5 // d34230c0
|
||||||
|
FCVTSWU X5, F0 // 538012d0
|
||||||
|
FCVTSLU X5, F0 // 538032d0
|
||||||
|
FSGNJS F1, F0, F2 // 53011020
|
||||||
|
FSGNJNS F1, F0, F2 // 53111020
|
||||||
|
FSGNJXS F1, F0, F2 // 53211020
|
||||||
|
FMVXS F0, X5 // d30200e0
|
||||||
|
FMVSX X5, F0 // 538002f0
|
||||||
|
FMVXW F0, X5 // d30200e0
|
||||||
|
FMVWX X5, F0 // 538002f0
|
||||||
|
FMADDS F1, F2, F3, F4 // 43822018
|
||||||
|
FMSUBS F1, F2, F3, F4 // 47822018
|
||||||
|
FNMSUBS F1, F2, F3, F4 // 4b822018
|
||||||
|
FNMADDS F1, F2, F3, F4 // 4f822018
|
||||||
|
|
||||||
|
// 21.8: Single-Precision Floating-Point Compare Instructions
|
||||||
|
FEQS F0, F1, X7 // d3a300a0
|
||||||
|
FLTS F0, F1, X7 // d39300a0
|
||||||
|
FLES F0, F1, X7 // d38300a0
|
||||||
|
|
||||||
|
// 21.9: Single-Precision Floating-Point Classify Instruction
|
||||||
|
FCLASSS F0, X5 // d31200e0
|
||||||
|
|
||||||
|
// 22.3: Double-Precision Load and Store Instructions
|
||||||
|
FLD (X5), F0 // 07b00200
|
||||||
|
FLD 4(X5), F0 // 07b04200
|
||||||
|
FSD F0, (X5) // 27b00200
|
||||||
|
FSD F0, 4(X5) // 27b20200
|
||||||
|
|
||||||
|
// 22.4: Double-Precision Floating-Point Computational Instructions
|
||||||
|
FADDD F1, F0, F2 // 53011002
|
||||||
|
FSUBD F1, F0, F2 // 5301100a
|
||||||
|
FMULD F1, F0, F2 // 53011012
|
||||||
|
FDIVD F1, F0, F2 // 5301101a
|
||||||
|
FMIND F1, F0, F2 // 5301102a
|
||||||
|
FMAXD F1, F0, F2 // 5311102a
|
||||||
|
FSQRTD F0, F1 // d300005a
|
||||||
|
|
||||||
|
// 22.5: Double-Precision Floating-Point Conversion and Move Instructions
|
||||||
|
FCVTWD F0, X5 // d31200c2
|
||||||
|
FCVTWD.RNE F0, X5 // d30200c2
|
||||||
|
FCVTWD.RTZ F0, X5 // d31200c2
|
||||||
|
FCVTWD.RDN F0, X5 // d32200c2
|
||||||
|
FCVTWD.RUP F0, X5 // d33200c2
|
||||||
|
FCVTWD.RMM F0, X5 // d34200c2
|
||||||
|
FCVTLD F0, X5 // d31220c2
|
||||||
|
FCVTLD.RNE F0, X5 // d30220c2
|
||||||
|
FCVTLD.RTZ F0, X5 // d31220c2
|
||||||
|
FCVTLD.RDN F0, X5 // d32220c2
|
||||||
|
FCVTLD.RUP F0, X5 // d33220c2
|
||||||
|
FCVTLD.RMM F0, X5 // d34220c2
|
||||||
|
FCVTDW X5, F0 // 538002d2
|
||||||
|
FCVTDL X5, F0 // 538022d2
|
||||||
|
FCVTWUD F0, X5 // d31210c2
|
||||||
|
FCVTWUD.RNE F0, X5 // d30210c2
|
||||||
|
FCVTWUD.RTZ F0, X5 // d31210c2
|
||||||
|
FCVTWUD.RDN F0, X5 // d32210c2
|
||||||
|
FCVTWUD.RUP F0, X5 // d33210c2
|
||||||
|
FCVTWUD.RMM F0, X5 // d34210c2
|
||||||
|
FCVTLUD F0, X5 // d31230c2
|
||||||
|
FCVTLUD.RNE F0, X5 // d30230c2
|
||||||
|
FCVTLUD.RTZ F0, X5 // d31230c2
|
||||||
|
FCVTLUD.RDN F0, X5 // d32230c2
|
||||||
|
FCVTLUD.RUP F0, X5 // d33230c2
|
||||||
|
FCVTLUD.RMM F0, X5 // d34230c2
|
||||||
|
FCVTDWU X5, F0 // 538012d2
|
||||||
|
FCVTDLU X5, F0 // 538032d2
|
||||||
|
FCVTSD F0, F1 // d3001040
|
||||||
|
FCVTDS F0, F1 // d3000042
|
||||||
|
FSGNJD F1, F0, F2 // 53011022
|
||||||
|
FSGNJND F1, F0, F2 // 53111022
|
||||||
|
FSGNJXD F1, F0, F2 // 53211022
|
||||||
|
FMVXD F0, X5 // d30200e2
|
||||||
|
FMVDX X5, F0 // 538002f2
|
||||||
|
FMADDD F1, F2, F3, F4 // 4382201a
|
||||||
|
FMSUBD F1, F2, F3, F4 // 4782201a
|
||||||
|
FNMSUBD F1, F2, F3, F4 // 4b82201a
|
||||||
|
FNMADDD F1, F2, F3, F4 // 4f82201a
|
||||||
|
|
||||||
|
// 22.6: Double-Precision Floating-Point Compare Instructions
|
||||||
|
FEQD F0, F1, X7 // d3a300a2
|
||||||
|
FLTD F0, F1, X7 // d39300a2
|
||||||
|
FLED F0, F1, X7 // d38300a2
|
||||||
|
|
||||||
|
// 22.7: Double-Precision Floating-Point Classify Instruction
|
||||||
|
FCLASSD F0, X5 // d31200e2
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
dir := t.TempDir()
|
||||||
|
path := filepath.Join(dir, "fp_riscv64.s")
|
||||||
|
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
assertRISCVDifferential(t, path, src, "fp")
|
||||||
|
}
|
||||||
Reference in new issue
Block a user