From 9b751f6e5811384963105ac65de406e2170748b2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 00:40:43 +0200 Subject: [PATCH] feat(asm): encode the riscv64 quad-precision family and fix the fp cvt paths Assisted-by: GLM 5.3 Flash --- asm/riscv_assemble.go | 16 ++-- asm/riscv_encode.go | 127 ++++++++++++++++++++----------- asm/riscv_quadfp_test.go | 86 +++++++++++++++++++++ asm/riscv_scalarfp_test.go | 152 +++++++++++++++++++++++++++++++++++++ 4 files changed, 331 insertions(+), 50 deletions(-) create mode 100644 asm/riscv_quadfp_test.go create mode 100644 asm/riscv_scalarfp_test.go diff --git a/asm/riscv_assemble.go b/asm/riscv_assemble.go index e5954cb..b3f3ab3 100644 --- a/asm/riscv_assemble.go +++ b/asm/riscv_assemble.go @@ -1368,7 +1368,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv // FP conversions with an explicit rounding mode: FCVTWS.RNE and friends // suffix the base mnemonic with RNE/RTZ/RDN/RUP/RMM, which lands in the - // low three bits of the funct7 field. + // funct3 field, replacing the bare form's default. if i := strings.IndexByte(mnem, '.'); i > 0 { if base, ok := riscvCvtTable[mnem[:i]]; ok { rm, ok := riscvRoundModes[mnem[i+1:]] @@ -1383,7 +1383,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } - base.funct7 = (base.funct7 &^ 7) | rm + base.funct3 = uint32(rm) word := riscvCvtType(base, rd, rs1) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } @@ -4653,18 +4653,21 @@ func isFPArithInstr(m string) bool { case "FADDS", "FSUBS", "FMULS", "FDIVS", "FADDD", "FSUBD", "FMULD", "FDIVD", "FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD", - "FSGNJS", "FSGNJX", "FSGNJXD", "FSGNJXS", "FSGNJND", "FSGNJNS", "FSGNJNX": + "FSGNJS", "FSGNJXD", "FSGNJXS", "FSGNJND", "FSGNJNS", + "FADDQ", "FSUBQ", "FMULQ", "FDIVQ", + "FSQRTQ", "FMINQ", "FMAXQ", "FSGNJQ", + "FSGNJXQ", "FSGNJNQ": return true } return false } func isFPLoadInstr(m string) bool { - return m == "FLW" || m == "FLD" + return m == "FLW" || m == "FLD" || m == "FLQ" } func isFPStoreInstr(m string) bool { - return m == "FSW" || m == "FSD" + return m == "FSW" || m == "FSD" || m == "FSQ" } func isLRInstr(m string) bool { @@ -4677,7 +4680,8 @@ func isSCInstr(m string) bool { func isFPCmpInstr(m string) bool { switch m { - case "FEQS", "FLTS", "FLES", "FEQD", "FLTD", "FLED": + case "FEQS", "FLTS", "FLES", "FEQD", "FLTD", "FLED", + "FEQQ", "FLTQ", "FLEQ": return true } return false diff --git a/asm/riscv_encode.go b/asm/riscv_encode.go index a8b4f6f..f7a7600 100644 --- a/asm/riscv_encode.go +++ b/asm/riscv_encode.go @@ -344,22 +344,46 @@ var riscvInstrTable = map[string]riscvEnc{ // FP loads/stores. "FLW": {0x07, 0x2, 0x00}, "FLD": {0x07, 0x3, 0x00}, + "FLQ": {0x07, 0x4, 0x00}, "FSW": {0x27, 0x2, 0x00}, "FSD": {0x27, 0x3, 0x00}, + "FSQ": {0x27, 0x4, 0x00}, // FP min/max. "FMINS": {0x53, 0x0, 0x14}, "FMAXS": {0x53, 0x1, 0x14}, "FMIND": {0x53, 0x0, 0x15}, "FMAXD": {0x53, 0x1, 0x15}, - // FP sign injection (double): rs2 carries the sign source. + // FP compare: the integer destination rides in rd (the 3-op R-type + // path writes it there). + "FEQS": {0x53, 0x2, 0x50}, + "FLTS": {0x53, 0x1, 0x50}, + "FLES": {0x53, 0x0, 0x50}, + "FEQD": {0x53, 0x2, 0x51}, + "FLTD": {0x53, 0x1, 0x51}, + "FLED": {0x53, 0x0, 0x51}, + "FEQQ": {0x53, 0x2, 0x53}, + "FLTQ": {0x53, 0x1, 0x53}, + "FLEQ": {0x53, 0x0, 0x53}, + // FP sign injection: the sign source rides in rs2; the XOR form's + // funct3 is 2 in every width. "FSGNJD": {0x53, 0x0, 0x11}, "FSGNJS": {0x53, 0x0, 0x10}, - "FSGNJX": {0x53, 0x0, 0x14}, - "FSGNJXD": {0x53, 0x0, 0x15}, - "FSGNJXS": {0x53, 0x0, 0x14}, + "FSGNJXD": {0x53, 0x2, 0x11}, + "FSGNJXS": {0x53, 0x2, 0x10}, "FSGNJND": {0x53, 0x1, 0x11}, "FSGNJNS": {0x53, 0x1, 0x10}, - "FSGNJNX": {0x53, 0x1, 0x14}, + "FSGNJQ": {0x53, 0x0, 0x13}, + "FSGNJNQ": {0x53, 0x1, 0x13}, + "FSGNJXQ": {0x53, 0x2, 0x13}, + // RV64Q, quad-precision arithmetic (the fmt field rides in funct7's + // low bits: 11 for quad). + "FADDQ": {0x53, 0x0, 0x03}, + "FSUBQ": {0x53, 0x0, 0x07}, + "FMULQ": {0x53, 0x0, 0x0B}, + "FDIVQ": {0x53, 0x0, 0x0F}, + "FSQRTQ": {0x53, 0x0, 0x2F}, + "FMINQ": {0x53, 0x0, 0x17}, + "FMAXQ": {0x53, 0x1, 0x17}, // RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03). // The toolchain gives LR acquire ordering (aq = 1) and SC release @@ -368,14 +392,6 @@ var riscvInstrTable = map[string]riscvEnc{ "LRD": {0x2F, 0x3, 0x02<<2 | 0x2}, "SCW": {0x2F, 0x2, 0x03<<2 | 0x1}, "SCD": {0x2F, 0x3, 0x03<<2 | 0x1}, - - // FP compare, result in integer register (funct7 0x50/0x51). - "FEQS": {0x53, 0x2, 0x50}, - "FLTS": {0x53, 0x1, 0x50}, - "FLES": {0x53, 0x0, 0x50}, - "FEQD": {0x53, 0x2, 0x51}, - "FLTD": {0x53, 0x1, 0x51}, - "FLED": {0x53, 0x0, 0x51}, } // riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode. @@ -399,49 +415,68 @@ func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 { type riscvCvtEnc struct { funct7 uint32 // bits [31:25] rs2 uint32 // conversion-type code in bits [24:20] + funct3 uint32 // the rounding mode or the fclass marker, bits [14:12] opcode uint32 // always 0x53 (OP-FP) } var riscvCvtTable = map[string]riscvCvtEnc{ - // float → int (rs2 selects the integer width/sign). - "FCVTWS": {0x60, 0x0, 0x53}, // float32 → int32 - "FCVTWUS": {0x60, 0x1, 0x53}, // float32 → uint32 - "FCVTLS": {0x60, 0x2, 0x53}, // float32 → int64 - "FCVTLUS": {0x60, 0x3, 0x53}, // float32 → uint64 - "FCVTWD": {0x61, 0x0, 0x53}, // float64 → int32 - "FCVTWUD": {0x61, 0x1, 0x53}, // float64 → uint32 - "FCVTLD": {0x61, 0x2, 0x53}, // float64 → int64 - "FCVTLUD": {0x61, 0x3, 0x53}, // float64 → uint64 - // int → float (rs2 selects the integer width/sign). - "FCVTSW": {0x68, 0x0, 0x53}, // int32 → float32 - "FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32 - "FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32 - "FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32 - "FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask - "FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask - "FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64 - "FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64 - "FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64 - "FCVTDLU": {0x69, 0x3, 0x53}, // uint64 → float64 + // float → int (rs2 selects the integer width/sign). The bare forms + // carry the specification's default rounding mode RTZ in funct3; the + // suffixed spellings override it. + "FCVTWS": {0x60, 0x0, 0x1, 0x53}, // float32 → int32 + "FCVTWUS": {0x60, 0x1, 0x1, 0x53}, // float32 → uint32 + "FCVTLS": {0x60, 0x2, 0x1, 0x53}, // float32 → int64 + "FCVTLUS": {0x60, 0x3, 0x1, 0x53}, // float32 → uint64 + "FCVTWD": {0x61, 0x0, 0x1, 0x53}, // float64 → int32 + "FCVTWUD": {0x61, 0x1, 0x1, 0x53}, // float64 → uint32 + "FCVTLD": {0x61, 0x2, 0x1, 0x53}, // float64 → int64 + "FCVTLUD": {0x61, 0x3, 0x1, 0x53}, // float64 → uint64 + // int → float (rs2 selects the integer width/sign): the default + // rounding mode RNE keeps funct3 0. + "FCVTSW": {0x68, 0x0, 0x0, 0x53}, // int32 → float32 + "FCVTSWU": {0x68, 0x1, 0x0, 0x53}, // uint32 → float32 + "FCVTSL": {0x68, 0x2, 0x0, 0x53}, // int64 → float32 + "FCVTSLU": {0x68, 0x3, 0x0, 0x53}, // uint64 → float32 + "FCLASSS": {0x70, 0x0, 0x1, 0x53}, // classify float32 → GPR mask + "FCLASSD": {0x71, 0x0, 0x1, 0x53}, // classify float64 → GPR mask + "FCVTDW": {0x69, 0x0, 0x0, 0x53}, // int32 → float64 + "FCVTDWU": {0x69, 0x1, 0x0, 0x53}, // uint32 → float64 + "FCVTDL": {0x69, 0x2, 0x0, 0x53}, // int64 → float64 + "FCVTDLU": {0x69, 0x3, 0x0, 0x53}, // uint64 → float64 // float → float width conversion. - "FCVTSD": {0x20, 0x1, 0x53}, // float64 → float32 - "FCVTDS": {0x21, 0x0, 0x53}, // float32 → float64 + "FCVTSD": {0x20, 0x1, 0x0, 0x53}, // float64 → float32 + "FCVTDS": {0x21, 0x0, 0x0, 0x53}, // float32 → float64 + // Quad-precision conversions: the fmt field rides in funct7's low + // bits (11 for quad), the other width in rs2 where one is needed. + "FCVTSQ": {0x20, 0x3, 0x0, 0x53}, // quad → float32 + "FCVTDQ": {0x21, 0x3, 0x0, 0x53}, // quad → float64 + "FCVTQS": {0x23, 0x0, 0x0, 0x53}, // float32 → quad + "FCVTQD": {0x23, 0x1, 0x0, 0x53}, // float64 → quad + "FCVTWQ": {0x63, 0x0, 0x1, 0x53}, // quad → int32 + "FCVTWUQ": {0x63, 0x1, 0x1, 0x53}, // quad → uint32 + "FCVTLQ": {0x63, 0x2, 0x1, 0x53}, // quad → int64 + "FCVTLUQ": {0x63, 0x3, 0x1, 0x53}, // quad → uint64 + "FCVTQW": {0x6B, 0x0, 0x0, 0x53}, // int32 → quad + "FCVTQWU": {0x6B, 0x1, 0x0, 0x53}, // uint32 → quad + "FCVTQL": {0x6B, 0x2, 0x0, 0x53}, // int64 → quad + "FCVTQLU": {0x6B, 0x3, 0x0, 0x53}, // uint64 → quad + "FCLASSQ": {0x73, 0x0, 0x1, 0x53}, // classify quad → GPR mask // Bit moves between integer and FP registers (no conversion). - "FMVXD": {0x71, 0x0, 0x53}, // float64 → int64 (bit move) - "FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move) - "FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move) - "FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move) + "FMVXD": {0x71, 0x0, 0x0, 0x53}, // float64 → int64 (bit move) + "FMVDX": {0x79, 0x0, 0x0, 0x53}, // int64 → float64 (bit move) + "FMVXW": {0x70, 0x0, 0x0, 0x53}, // float32 → int32 (bit move) + "FMVWX": {0x78, 0x0, 0x0, 0x53}, // int32 → float32 (bit move) // The toolchain's W/D suffix spellings of the same moves. - "FMVXS": {0x70, 0x0, 0x53}, - "FMVFS": {0x78, 0x0, 0x53}, - "FMVSX": {0x79, 0x0, 0x53}, + "FMVXS": {0x70, 0x0, 0x0, 0x53}, + "FMVFS": {0x78, 0x0, 0x0, 0x53}, + "FMVSX": {0x78, 0x0, 0x0, 0x53}, } // riscvCvtType encodes an FP conversion instruction. -// Layout: funct7 | rs2(convtype) | rs1 | funct3(0) | rd | opcode. +// Layout: funct7 | rs2(convtype) | rs1 | funct3(rm) | rd | opcode. func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 { return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) | - (uint32(rd) << 7) | enc.opcode + (enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode } // R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD). @@ -455,12 +490,16 @@ type riscvFmaEnc struct { var riscvFmaTable = map[string]riscvFmaEnc{ "FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3 "FMADDD": {0x1, 0x43}, + "FMADDQ": {0x3, 0x43}, "FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3 "FMSUBD": {0x1, 0x47}, + "FMSUBQ": {0x3, 0x47}, "FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3 "FNMSUBD": {0x1, 0x4B}, + "FNMSUBQ": {0x3, 0x4B}, "FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3 "FNMADDD": {0x1, 0x4F}, + "FNMADDQ": {0x3, 0x4F}, } // riscvFmaType encodes an R4-type fused multiply-add instruction. diff --git a/asm/riscv_quadfp_test.go b/asm/riscv_quadfp_test.go new file mode 100644 index 0000000..c5c83f3 --- /dev/null +++ b/asm/riscv_quadfp_test.go @@ -0,0 +1,86 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import "testing" + +// TestRISCVQuadFP_golden pins the quad-precision ("Q") family on golden +// vectors from the specification. The Go toolchain's object table carries +// the encodings but its assembler accepts no Q mnemonic, so no oracle run +// is possible: the words below follow the Q extension's encoding table with +// the Plan 9 operand order the other widths use (first operand in rs2, the +// bare float-to-integer conversions carrying the RTZ rounding mode). +func TestRISCVQuadFP_golden(t *testing.T) { + fn := firstTextRISCV(t, `#include "textflag.h" +TEXT ·q(SB), NOSPLIT, $0 + FLQ 8(X5), F3 + FSQ F3, 8(X5) + FADDQ F1, F2, F3 + FSUBQ F1, F2, F3 + FMULQ F1, F2, F3 + FDIVQ F1, F2, F3 + FSQRTQ F2, F1 + FMINQ F1, F2, F3 + FMAXQ F1, F2, F3 + FEQQ F3, F2, X5 + FLTQ F3, F2, X5 + FLEQ F3, F2, X5 + FSGNJQ F1, F2, F3 + FSGNJNQ F1, F2, F3 + FSGNJXQ F1, F2, F3 + FMADDQ F1, F2, F3, F4 + FMSUBQ F1, F2, F3, F4 + FNMSUBQ F1, F2, F3, F4 + FNMADDQ F1, F2, F3, F4 + FCVTSQ F5, F2 + FCVTDQ F5, F2 + FCVTQS F2, F5 + FCVTQD F2, F5 + FCVTWQ F2, X5 + FCVTWUQ F2, X5 + FCVTLQ F2, X5 + FCVTLUQ F2, X5 + FCVTQW X5, F2 + FCVTQWU X5, F2 + FCVTQL X5, F2 + FCVTQLU X5, F2 + FCLASSQ F2, X5 + RET +`) + code := assembleRISCVHelper(t, fn) + riscvWants(t, code, + 0x0082C187, // flq f3, 8(x5) + 0x0032C427, // fsq f3, 8(x5) + 0x061101D3, // fadd.q f3, f2, f1 + 0x0E1101D3, // fsub.q + 0x161101D3, // fmul.q + 0x1E1101D3, // fdiv.q + 0x5E0100D3, // fsqrt.q f1, f2 + 0x2E1101D3, // fmin.q f3, f2, f1 + 0x2E1111D3, // fmax.q + 0xA63122D3, // feq.q x5, f2, f3 + 0xA63112D3, // flt.q + 0xA63102D3, // fle.q + 0x261101D3, // fsgnj.q + 0x261111D3, // fsgnjn.q + 0x261121D3, // fsgnjx.q + 0x1E208243, // fmadd.q f4, f1, f2, f3 + 0x1E208247, // fmsub.q + 0x1E20824B, // fnmsub.q + 0x1E20824F, // fnmadd.q + 0x40328153, // fcvt.s.q f2, f5 + 0x42328153, // fcvt.d.q f2, f5 + 0x460102D3, // fcvt.q.s f5, f2 + 0x461102D3, // fcvt.q.d f5, f2 + 0xC60112D3, // fcvt.w.q x5, f2 + 0xC61112D3, // fcvt.wu.q + 0xC62112D3, // fcvt.l.q + 0xC63112D3, // fcvt.lu.q + 0xD6028153, // fcvt.q.w f2, x5 + 0xD6128153, // fcvt.q.wu + 0xD6228153, // fcvt.q.l + 0xD6328153, // fcvt.q.lu + 0xE60112D3, // fclass.q x5, f2 + ) +} diff --git a/asm/riscv_scalarfp_test.go b/asm/riscv_scalarfp_test.go new file mode 100644 index 0000000..1f2e989 --- /dev/null +++ b/asm/riscv_scalarfp_test.go @@ -0,0 +1,152 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "os" + "path/filepath" + "testing" +) + +// TestRISCVScalarFP_Differential proves the scalar single- and +// double-precision families against the toolchain: sections 21.6 through +// 22.7 of the "F" and "D" specifications' computational, conversion, move, +// sign-injection, compare and classify instructions in the toolchain's own +// testdata wording, assembled by gasm and by go tool asm must agree word +// for word. +func TestRISCVScalarFP_Differential(t *testing.T) { + src := `#include "textflag.h" + +TEXT ·fp(SB), NOSPLIT, $0 + + // 21.6: Single-Precision Floating-Point Computational Instructions + FADDS F1, F0, F2 // 53011000 + FSUBS F1, F0, F2 // 53011008 + FMULS F1, F0, F2 // 53011010 + FDIVS F1, F0, F2 // 53011018 + FMINS F1, F0, F2 // 53011028 + FMAXS F1, F0, F2 // 53111028 + FSQRTS F0, F1 // d3000058 + + // 21.7: Single-Precision Floating-Point Conversion and Move Instructions + FCVTWS F0, X5 // d31200c0 + FCVTWS.RNE F0, X5 // d30200c0 + FCVTWS.RTZ F0, X5 // d31200c0 + FCVTWS.RDN F0, X5 // d32200c0 + FCVTWS.RUP F0, X5 // d33200c0 + FCVTWS.RMM F0, X5 // d34200c0 + FCVTLS F0, X5 // d31220c0 + FCVTLS.RNE F0, X5 // d30220c0 + FCVTLS.RTZ F0, X5 // d31220c0 + FCVTLS.RDN F0, X5 // d32220c0 + FCVTLS.RUP F0, X5 // d33220c0 + FCVTLS.RMM F0, X5 // d34220c0 + FCVTSW X5, F0 // 538002d0 + FCVTSL X5, F0 // 538022d0 + FCVTWUS F0, X5 // d31210c0 + FCVTWUS.RNE F0, X5 // d30210c0 + FCVTWUS.RTZ F0, X5 // d31210c0 + FCVTWUS.RDN F0, X5 // d32210c0 + FCVTWUS.RUP F0, X5 // d33210c0 + FCVTWUS.RMM F0, X5 // d34210c0 + FCVTLUS F0, X5 // d31230c0 + FCVTLUS.RNE F0, X5 // d30230c0 + FCVTLUS.RTZ F0, X5 // d31230c0 + FCVTLUS.RDN F0, X5 // d32230c0 + FCVTLUS.RUP F0, X5 // d33230c0 + FCVTLUS.RMM F0, X5 // d34230c0 + FCVTSWU X5, F0 // 538012d0 + FCVTSLU X5, F0 // 538032d0 + FSGNJS F1, F0, F2 // 53011020 + FSGNJNS F1, F0, F2 // 53111020 + FSGNJXS F1, F0, F2 // 53211020 + FMVXS F0, X5 // d30200e0 + FMVSX X5, F0 // 538002f0 + FMVXW F0, X5 // d30200e0 + FMVWX X5, F0 // 538002f0 + FMADDS F1, F2, F3, F4 // 43822018 + FMSUBS F1, F2, F3, F4 // 47822018 + FNMSUBS F1, F2, F3, F4 // 4b822018 + FNMADDS F1, F2, F3, F4 // 4f822018 + + // 21.8: Single-Precision Floating-Point Compare Instructions + FEQS F0, F1, X7 // d3a300a0 + FLTS F0, F1, X7 // d39300a0 + FLES F0, F1, X7 // d38300a0 + + // 21.9: Single-Precision Floating-Point Classify Instruction + FCLASSS F0, X5 // d31200e0 + + // 22.3: Double-Precision Load and Store Instructions + FLD (X5), F0 // 07b00200 + FLD 4(X5), F0 // 07b04200 + FSD F0, (X5) // 27b00200 + FSD F0, 4(X5) // 27b20200 + + // 22.4: Double-Precision Floating-Point Computational Instructions + FADDD F1, F0, F2 // 53011002 + FSUBD F1, F0, F2 // 5301100a + FMULD F1, F0, F2 // 53011012 + FDIVD F1, F0, F2 // 5301101a + FMIND F1, F0, F2 // 5301102a + FMAXD F1, F0, F2 // 5311102a + FSQRTD F0, F1 // d300005a + + // 22.5: Double-Precision Floating-Point Conversion and Move Instructions + FCVTWD F0, X5 // d31200c2 + FCVTWD.RNE F0, X5 // d30200c2 + FCVTWD.RTZ F0, X5 // d31200c2 + FCVTWD.RDN F0, X5 // d32200c2 + FCVTWD.RUP F0, X5 // d33200c2 + FCVTWD.RMM F0, X5 // d34200c2 + FCVTLD F0, X5 // d31220c2 + FCVTLD.RNE F0, X5 // d30220c2 + FCVTLD.RTZ F0, X5 // d31220c2 + FCVTLD.RDN F0, X5 // d32220c2 + FCVTLD.RUP F0, X5 // d33220c2 + FCVTLD.RMM F0, X5 // d34220c2 + FCVTDW X5, F0 // 538002d2 + FCVTDL X5, F0 // 538022d2 + FCVTWUD F0, X5 // d31210c2 + FCVTWUD.RNE F0, X5 // d30210c2 + FCVTWUD.RTZ F0, X5 // d31210c2 + FCVTWUD.RDN F0, X5 // d32210c2 + FCVTWUD.RUP F0, X5 // d33210c2 + FCVTWUD.RMM F0, X5 // d34210c2 + FCVTLUD F0, X5 // d31230c2 + FCVTLUD.RNE F0, X5 // d30230c2 + FCVTLUD.RTZ F0, X5 // d31230c2 + FCVTLUD.RDN F0, X5 // d32230c2 + FCVTLUD.RUP F0, X5 // d33230c2 + FCVTLUD.RMM F0, X5 // d34230c2 + FCVTDWU X5, F0 // 538012d2 + FCVTDLU X5, F0 // 538032d2 + FCVTSD F0, F1 // d3001040 + FCVTDS F0, F1 // d3000042 + FSGNJD F1, F0, F2 // 53011022 + FSGNJND F1, F0, F2 // 53111022 + FSGNJXD F1, F0, F2 // 53211022 + FMVXD F0, X5 // d30200e2 + FMVDX X5, F0 // 538002f2 + FMADDD F1, F2, F3, F4 // 4382201a + FMSUBD F1, F2, F3, F4 // 4782201a + FNMSUBD F1, F2, F3, F4 // 4b82201a + FNMADDD F1, F2, F3, F4 // 4f82201a + + // 22.6: Double-Precision Floating-Point Compare Instructions + FEQD F0, F1, X7 // d3a300a2 + FLTD F0, F1, X7 // d39300a2 + FLED F0, F1, X7 // d38300a2 + + // 22.7: Double-Precision Floating-Point Classify Instruction + FCLASSD F0, X5 // d31200e2 + RET +` + dir := t.TempDir() + path := filepath.Join(dir, "fp_riscv64.s") + if err := os.WriteFile(path, []byte(src), 0o644); err != nil { + t.Fatal(err) + } + assertRISCVDifferential(t, path, src, "fp") +}