From eee7a6d4a4d8104cd1a29a8d344f6d690d3d428c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sun, 2 Aug 2026 11:45:00 +0200 Subject: [PATCH] feat(riscv): add RISC-V RV64A, FP and CSR instruction support Assisted-by: Kimi K3 --- asm/riscv_assemble.go | 187 ++++++++++++++++++++++++++++++++++++ asm/riscv_encode.go | 160 ++++++++++++++++++++++++++++++ testdata/amo_fp_riscv64.s | 20 ++++ testdata/csr_riscv64.s | 27 ++++++ testdata/fma_riscv64.s | 24 +++++ testdata/lrsc_cvt_riscv64.s | 37 +++++++ 6 files changed, 455 insertions(+) create mode 100644 testdata/amo_fp_riscv64.s create mode 100644 testdata/csr_riscv64.s create mode 100644 testdata/fma_riscv64.s create mode 100644 testdata/lrsc_cvt_riscv64.s diff --git a/asm/riscv_assemble.go b/asm/riscv_assemble.go index ee4368a..403888a 100644 --- a/asm/riscv_assemble.go +++ b/asm/riscv_assemble.go @@ -90,6 +90,66 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } + // FP conversion / move instructions use a separate table (rs2 encodes + // the conversion type, not a register). Handle them before the main + // table lookup. + if cvtEnc, ok := riscvCvtTable[mnem]; ok { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rs1 := regFromOperand(ops[0]) + rd := regFromOperand(ops[1]) + if rd < 0 || rs1 < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + word := riscvCvtType(cvtEnc, rd, rs1) + return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil + } + + // R4-type fused multiply-add: INSTR rs1, rs2, rs3, rd (destination last). + if fmaEnc, ok := riscvFmaTable[mnem]; ok { + if len(ops) != 4 { + return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) + } + rs1 := regFromOperand(ops[0]) + rs2 := regFromOperand(ops[1]) + rs3 := regFromOperand(ops[2]) + rd := regFromOperand(ops[3]) + if rd < 0 || rs1 < 0 || rs2 < 0 || rs3 < 0 { + return nil, fmt.Errorf("invalid FP register in %s", mnem) + } + word := riscvFmaType(fmaEnc, rd, rs1, rs2, rs3) + return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil + } + + // CSR instructions: INSTR csr, rs1|uimm, rd (destination last). + if csrEnc, ok := riscvCsrTable[mnem]; ok { + if len(ops) != 3 { + return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + csr := immFromOperand(ops[0]) // CSR address (12-bit) + rd := regFromOperand(ops[2]) // destination register + if rd < 0 { + return nil, fmt.Errorf("invalid destination register in %s", mnem) + } + var src int + if csrEnc.imm { + // Immediate variant: ops[1] is a 5-bit unsigned immediate. + src = int(immFromOperand(ops[1])) + if src < 0 || src > 31 { + return nil, fmt.Errorf("%s: uimm out of range 0-31", mnem) + } + } else { + // Register variant: ops[1] is a register. + src = regFromOperand(ops[1]) + if src < 0 { + return nil, fmt.Errorf("invalid source register in %s", mnem) + } + } + word := riscvCsrType(csrEnc, rd, src, csr) + return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil + } + enc, ok := riscvInstrTable[mnem] if !ok { return nil, fmt.Errorf("unsupported RISC-V instruction %q", mnem) @@ -106,6 +166,82 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv } word = riscvRType(enc, rd, rs1, rs2) + // AMO atomics: Plan 9 order is INSTR src, (addr), dst. + case len(ops) == 3 && isAMOInstr(mnem): + rs2 := regFromOperand(ops[0]) // source value + rs1, _ := memFromOperandWithFrame(ops[1], fi) // memory address + rd := regFromOperand(ops[2]) // destination (old value) + if rd < 0 || rs1 < 0 || rs2 < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + word = riscvAMOType(enc, rd, rs1, rs2) + + // FP arithmetic: Plan 9 order is INSTR src1, src2, dst. + case len(ops) == 3 && isFPArithInstr(mnem): + rs1 := regFromOperand(ops[0]) + rs2 := regFromOperand(ops[1]) + rd := regFromOperand(ops[2]) + if rd < 0 || rs1 < 0 || rs2 < 0 { + return nil, fmt.Errorf("invalid FP register in %s", mnem) + } + word = riscvRType(enc, rd, rs1, rs2) + + // FP arithmetic (2-operand): FSQRT src, dst. + case len(ops) == 2 && isFPArithInstr(mnem): + rs1 := regFromOperand(ops[0]) + rd := regFromOperand(ops[1]) + if rd < 0 || rs1 < 0 { + return nil, fmt.Errorf("invalid FP register in %s", mnem) + } + word = riscvRType(enc, rd, rs1, 0) + + // FP loads: INSTR addr, freg (Plan 9: source first). + case len(ops) == 2 && isFPLoadInstr(mnem): + rd := regFromOperand(ops[1]) + rs1, imm := memFromOperandWithFrame(ops[0], fi) + if rd < 0 || rs1 < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + word = riscvIType(enc, rd, rs1, imm) + + // FP stores: INSTR freg, addr (Plan 9: source first). + case len(ops) == 2 && isFPStoreInstr(mnem): + rs2 := regFromOperand(ops[0]) + rs1, imm := memFromOperandWithFrame(ops[1], fi) + if rs2 < 0 || rs1 < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + word = riscvSType(enc, rs1, rs2, imm) + + // LR (load-reserved): INSTR (addr), dst — 2 operands. + case len(ops) == 2 && isLRInstr(mnem): + rs1, _ := memFromOperandWithFrame(ops[0], fi) + rd := regFromOperand(ops[1]) + if rd < 0 || rs1 < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + word = riscvAMOType(enc, rd, rs1, 0) // rs2=0 for LR + + // SC (store-conditional): INSTR src, (addr), dst — 3 operands. + case len(ops) == 3 && isSCInstr(mnem): + rs2 := regFromOperand(ops[0]) + rs1, _ := memFromOperandWithFrame(ops[1], fi) + rd := regFromOperand(ops[2]) + if rd < 0 || rs1 < 0 || rs2 < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + word = riscvAMOType(enc, rd, rs1, rs2) + + // FP compare: INSTR src1, src2, dst(int) — result in integer register. + case len(ops) == 3 && isFPCmpInstr(mnem): + rs1 := regFromOperand(ops[0]) + rs2 := regFromOperand(ops[1]) + rd := regFromOperand(ops[2]) + if rd < 0 || rs1 < 0 || rs2 < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + word = riscvRType(enc, rd, rs1, rs2) + // I-type with immediate: Plan 9 order is INSTR src, imm, dst. case len(ops) == 3 && isITypeInstr(mnem): rs1 := regFromOperand(ops[0]) // source register @@ -216,6 +352,57 @@ func isUTypeInstr(m string) bool { return m == "LUI" || m == "AUIPC" } +func isAMOInstr(m string) bool { + switch m { + case "AMOSWAPW", "AMOSWAPD", "AMOADDW", "AMOADDD", + "AMOANDW", "AMOANDD", "AMOORW", "AMOORD", + "AMOXORW", "AMOXORD", "AMOMAXW", "AMOMAXD", + "AMOMINW", "AMOMIND", "AMOMAXUW", "AMOMAXUD", + "AMOMINUW", "AMOMINUD": + return true + } + return false +} + +func isFPArithInstr(m string) bool { + switch m { + case "FADDS", "FSUBS", "FMULS", "FDIVS", + "FADDD", "FSUBD", "FMULD", "FDIVD", + "FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD": + return true + } + return false +} + +func isFPLoadInstr(m string) bool { + return m == "FLW" || m == "FLD" +} + +func isFPStoreInstr(m string) bool { + return m == "FSW" || m == "FSD" +} + +func isLRInstr(m string) bool { + return m == "LRW" || m == "LRD" +} + +func isSCInstr(m string) bool { + return m == "SCW" || m == "SCD" +} + +func isFPCmpInstr(m string) bool { + switch m { + case "FEQS", "FLTS", "FLES", "FEQD", "FLTD", "FLED": + return true + } + return false +} + +func isFPCvtInstr(m string) bool { + _, ok := riscvCvtTable[m] + return ok +} + // Operand helpers. func regFromOperand(op *ast.Operand) int { // Register is in Addr.Base (from (base) syntax) or Addr.Sym.Name (bare ident). diff --git a/asm/riscv_encode.go b/asm/riscv_encode.go index fde4241..e686653 100644 --- a/asm/riscv_encode.go +++ b/asm/riscv_encode.go @@ -221,6 +221,63 @@ var riscvInstrTable = map[string]riscvEnc{ "ECALL": {0x73, 0x0, 0x00}, "EBREAK": {0x73, 0x0, 0x00}, "FENCE": {0x0F, 0x0, 0x00}, + + // RV64A — atomics (AMO opcode 0x2F). + // funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27]. + "AMOSWAPW": {0x2F, 0x2, 0x01 << 2}, + "AMOSWAPD": {0x2F, 0x3, 0x01 << 2}, + "AMOADDW": {0x2F, 0x2, 0x00 << 2}, + "AMOADDD": {0x2F, 0x3, 0x00 << 2}, + "AMOANDW": {0x2F, 0x2, 0x0C << 2}, + "AMOANDD": {0x2F, 0x3, 0x0C << 2}, + "AMOORW": {0x2F, 0x2, 0x06 << 2}, + "AMOORD": {0x2F, 0x3, 0x06 << 2}, + "AMOXORW": {0x2F, 0x2, 0x04 << 2}, + "AMOXORD": {0x2F, 0x3, 0x04 << 2}, + "AMOMAXW": {0x2F, 0x2, 0x14 << 2}, + "AMOMAXD": {0x2F, 0x3, 0x14 << 2}, + "AMOMINW": {0x2F, 0x2, 0x10 << 2}, + "AMOMIND": {0x2F, 0x3, 0x10 << 2}, + "AMOMAXUW": {0x2F, 0x2, 0x1C << 2}, + "AMOMAXUD": {0x2F, 0x3, 0x1C << 2}, + "AMOMINUW": {0x2F, 0x2, 0x18 << 2}, + "AMOMINUD": {0x2F, 0x3, 0x18 << 2}, + + // RV64F/D — floating-point arithmetic. + "FADDS": {0x53, 0x0, 0x00}, + "FSUBS": {0x53, 0x0, 0x04}, + "FMULS": {0x53, 0x0, 0x08}, + "FDIVS": {0x53, 0x0, 0x0C}, + "FADDD": {0x53, 0x0, 0x01}, + "FSUBD": {0x53, 0x0, 0x05}, + "FMULD": {0x53, 0x0, 0x09}, + "FDIVD": {0x53, 0x0, 0x0D}, + "FSQRTS": {0x53, 0x0, 0x2C}, + "FSQRTD": {0x53, 0x0, 0x2D}, + // FP loads/stores. + "FLW": {0x07, 0x2, 0x00}, + "FLD": {0x07, 0x3, 0x00}, + "FSW": {0x27, 0x2, 0x00}, + "FSD": {0x27, 0x3, 0x00}, + // FP min/max. + "FMINS": {0x53, 0x0, 0x14}, + "FMAXS": {0x53, 0x1, 0x14}, + "FMIND": {0x53, 0x0, 0x15}, + "FMAXD": {0x53, 0x1, 0x15}, + + // RV64A — load-reserved / store-conditional (funct5 0x02 / 0x03). + "LRW": {0x2F, 0x2, 0x02 << 2}, + "LRD": {0x2F, 0x3, 0x02 << 2}, + "SCW": {0x2F, 0x2, 0x03 << 2}, + "SCD": {0x2F, 0x3, 0x03 << 2}, + + // FP compare — result in integer register (funct7 0x50/0x51). + "FEQS": {0x53, 0x2, 0x50}, + "FLTS": {0x53, 0x1, 0x50}, + "FLES": {0x53, 0x0, 0x50}, + "FEQD": {0x53, 0x2, 0x51}, + "FLTD": {0x53, 0x1, 0x51}, + "FLED": {0x53, 0x0, 0x51}, } // riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode. @@ -229,6 +286,109 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 { (enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode } +// riscvAMOType encodes an atomic (AMO) instruction. +// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode. +// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2). +func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 { + funct5 := enc.funct7 >> 2 // extract funct5 from the stored value + return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) | + (enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode +} + +// FP conversion instructions (FCVT, FMV). These use the rs2 field to +// encode the conversion type rather than a register, so they are handled +// separately from the general instruction table. +type riscvCvtEnc struct { + funct7 uint32 // bits [31:25] + rs2 uint32 // conversion-type code in bits [24:20] + opcode uint32 // always 0x53 (OP-FP) +} + +var riscvCvtTable = map[string]riscvCvtEnc{ + // float → int (rs2 selects the integer width/sign). + "FCVTWS": {0x60, 0x0, 0x53}, // float32 → int32 + "FCVTWUS": {0x60, 0x1, 0x53}, // float32 → uint32 + "FCVTLS": {0x60, 0x2, 0x53}, // float32 → int64 + "FCVTLUS": {0x60, 0x3, 0x53}, // float32 → uint64 + "FCVTWD": {0x61, 0x0, 0x53}, // float64 → int32 + "FCVTWUD": {0x61, 0x1, 0x53}, // float64 → uint32 + "FCVTLD": {0x61, 0x2, 0x53}, // float64 → int64 + "FCVTLUD": {0x61, 0x3, 0x53}, // float64 → uint64 + // int → float (rs2 selects the integer width/sign). + "FCVTSW": {0x68, 0x0, 0x53}, // int32 → float32 + "FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32 + "FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32 + "FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32 + "FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64 + "FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64 + "FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64 + "FCVTDLU": {0x69, 0x3, 0x53}, // uint64 → float64 + // float → float width conversion. + "FCVTSD": {0x20, 0x1, 0x53}, // float64 → float32 + "FCVTDS": {0x21, 0x0, 0x53}, // float32 → float64 + // Bit moves between integer and FP registers (no conversion). + "FMVXD": {0x71, 0x0, 0x53}, // float64 → int64 (bit move) + "FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move) + "FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move) + "FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move) +} + +// riscvCvtType encodes an FP conversion instruction. +// Layout: funct7 | rs2(convtype) | rs1 | funct3(0) | rd | opcode. +func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 { + return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) | + (uint32(rd) << 7) | enc.opcode +} + +// R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD). +// These take 4 register operands: rs1, rs2, rs3, rd. +// Layout: rs3 | fmt | rs2 | rs1 | rm | rd | opcode. +type riscvFmaEnc struct { + fmt uint32 // bits [26:25]: 0x0 = single, 0x1 = double + opcode uint32 // bits [6:0] +} + +var riscvFmaTable = map[string]riscvFmaEnc{ + "FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3 + "FMADDD": {0x1, 0x43}, + "FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3 + "FMSUBD": {0x1, 0x47}, + "FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3 + "FNMSUBD": {0x1, 0x4B}, + "FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3 + "FNMADDD": {0x1, 0x4F}, +} + +// riscvFmaType encodes an R4-type fused multiply-add instruction. +func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 { + return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) | + (uint32(rs1) << 15) | (0x0 << 12) /* rm=dynamic */ | (uint32(rd) << 7) | enc.opcode +} + +// CSR (Control and Status Register) instructions. +// Format: csr[11:0] | rs1/zimm | funct3 | rd | opcode (0x73). +type riscvCsrEnc struct { + funct3 uint32 // bits [14:12] + imm bool // true for CSRRWI/CSRRSI/CSRRCI (5-bit uimm variant) +} + +var riscvCsrTable = map[string]riscvCsrEnc{ + "CSRRW": {0x1, false}, // rd=CSR, CSR=rs1 + "CSRRS": {0x2, false}, // rd=CSR, CSR |= rs1 + "CSRRC": {0x3, false}, // rd=CSR, CSR &= ~rs1 + "CSRRWI": {0x5, true}, // rd=CSR, CSR=uimm + "CSRRSI": {0x6, true}, // rd=CSR, CSR |= uimm + "CSRRCI": {0x7, true}, // rd=CSR, CSR &= ~uimm +} + +// riscvCsrType encodes a CSR instruction. +// csr is the 12-bit CSR address; src is either a register number or a 5-bit +// unsigned immediate (depending on enc.imm). +func riscvCsrType(enc riscvCsrEnc, rd, src int, csr int32) uint32 { + return (uint32(csr&0xFFF) << 20) | (uint32(src&0x1F) << 15) | + (enc.funct3 << 12) | (uint32(rd) << 7) | 0x73 +} + // riscvIType encodes an I-type instruction: imm[11:0] | rs1 | funct3 | rd | opcode. func riscvIType(enc riscvEnc, rd, rs1 int, imm int32) uint32 { return (uint32(imm&0xFFF) << 20) | (uint32(rs1) << 15) | diff --git a/testdata/amo_fp_riscv64.s b/testdata/amo_fp_riscv64.s new file mode 100644 index 0000000..6bab541 --- /dev/null +++ b/testdata/amo_fp_riscv64.s @@ -0,0 +1,20 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +#include "textflag.h" + +// func atomicAdd(ptr *int64, val int64) int64 +TEXT ·atomicAdd(SB), NOSPLIT, $0-24 + LD a+0(FP), A0 + LD b+8(FP), A1 + AMOADDD A1, (A0), A2 + SD A2, ret+16(FP) + RET + +// func fpAdd(a, b float64) float64 +TEXT ·fpAdd(SB), NOSPLIT, $0-24 + FLD a+0(FP), FA0 + FLD b+8(FP), FA1 + FADDD FA0, FA1, FA2 + FSD FA2, ret+16(FP) + RET diff --git a/testdata/csr_riscv64.s b/testdata/csr_riscv64.s new file mode 100644 index 0000000..874ea66 --- /dev/null +++ b/testdata/csr_riscv64.s @@ -0,0 +1,27 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +#include "textflag.h" + +// func readCSR(csr int64) int64 +// Reads a CSR into the return value. +TEXT ·readCSR(SB), NOSPLIT, $0-16 + LD a+0(FP), A0 + CSRRS $0x300, X0, A1 + SD A1, ret+8(FP) + RET + +// func setCSRBit(csr, bit int64) int64 +TEXT ·setCSRBit(SB), NOSPLIT, $0-24 + LD a+0(FP), A0 + LD b+8(FP), A1 + CSRRS $0x304, A1, A2 + SD A2, ret+16(FP) + RET + +// func writeCSR(val int64) int64 +TEXT ·writeCSR(SB), NOSPLIT, $0-16 + LD a+0(FP), A0 + CSRRW $0x305, A0, A1 + SD A1, ret+8(FP) + RET diff --git a/testdata/fma_riscv64.s b/testdata/fma_riscv64.s new file mode 100644 index 0000000..97f64f3 --- /dev/null +++ b/testdata/fma_riscv64.s @@ -0,0 +1,24 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +#include "textflag.h" + +// func fma(a, b, c float64) float64 +// Computes a*b + c using fused multiply-add. +TEXT ·fma(SB), NOSPLIT, $0-32 + FLD a+0(FP), FA0 + FLD b+8(FP), FA1 + FLD c+16(FP), FA2 + FMADDD FA0, FA1, FA2, FA3 + FSD FA3, ret+24(FP) + RET + +// func fms(a, b, c float64) float64 +// Computes a*b - c using fused multiply-subtract. +TEXT ·fms(SB), NOSPLIT, $0-32 + FLD a+0(FP), FA0 + FLD b+8(FP), FA1 + FLD c+16(FP), FA2 + FMSUBD FA0, FA1, FA2, FA3 + FSD FA3, ret+24(FP) + RET diff --git a/testdata/lrsc_cvt_riscv64.s b/testdata/lrsc_cvt_riscv64.s new file mode 100644 index 0000000..4ca5284 --- /dev/null +++ b/testdata/lrsc_cvt_riscv64.s @@ -0,0 +1,37 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +#include "textflag.h" + +// func casLoop(ptr *int64, old, new int64) bool +// Compare-and-swap using LR/SC. +TEXT ·casLoop(SB), NOSPLIT, $0-32 +cas_retry: + LD a+0(FP), A0 + LRD (A0), A1 + LD b+8(FP), A2 + BNE A1, A2, cas_fail + LD c+16(FP), A3 + SCD A3, (A0), A4 + BNE A4, X0, cas_retry + ADDI X0, $1, A5 + SD A5, ret+24(FP) + RET +cas_fail: + SD X0, ret+24(FP) + RET + +// func intToFloat(x int64) float64 +TEXT ·intToFloat(SB), NOSPLIT, $0-16 + LD a+0(FP), A0 + FCVTDL A0, FA0 + FSD FA0, ret+8(FP) + RET + +// func compare(a, b float64) bool +TEXT ·compare(SB), NOSPLIT, $0-24 + FLD a+0(FP), FA0 + FLD b+8(FP), FA1 + FLTD FA0, FA1, A0 + SD A0, ret+16(FP) + RET