The Zba address generation, Zbb unary bit operations, Zbc carry-less multiplication and Zbs single-bit families were names the table carried and the encoder refused: thirty spellings plus RORI and XNOR fell over. The register and immediate forms now encode as the toolchain does, the unary operations carry their fixed rs2 constant, RORI lowers to ROR's expansion (its reverse shift compressing like ROR's), XNOR XORs and inverts in place, and ROL/ROLW rotate left through the same temporary the toolchain uses, taking a register amount only as its own expansion requires. The toolchain's whole testdata block for these families is now a differential test: every word must agree byte for byte. Assisted-by: GLM 5.3 Flash
727 lines
24 KiB
Go
727 lines
24 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
package asm
|
|
|
|
// RISC-V register encoding: maps register names to their 5-bit numbers.
|
|
// The Go assembler uses the standard RISC-V ABI naming.
|
|
|
|
// riscvRegNum returns the 5-bit register number for a RISC-V register name.
|
|
// Returns -1 if the register is not recognized.
|
|
func riscvRegNum(name string) int {
|
|
switch name {
|
|
// Numbered integer registers.
|
|
case "X0", "ZERO":
|
|
return 0
|
|
case "X1", "RA", "LR":
|
|
return 1
|
|
case "X2", "SP":
|
|
return 2
|
|
case "X3", "GP":
|
|
return 3
|
|
case "X4", "TP":
|
|
return 4
|
|
case "X5", "T0":
|
|
return 5
|
|
case "X6", "T1":
|
|
return 6
|
|
case "X7", "T2":
|
|
return 7
|
|
case "X8", "S0", "FP":
|
|
return 8
|
|
case "X9", "S1":
|
|
return 9
|
|
case "X10", "A0":
|
|
return 10
|
|
case "X11", "A1":
|
|
return 11
|
|
case "X12", "A2":
|
|
return 12
|
|
case "X13", "A3":
|
|
return 13
|
|
case "X14", "A4":
|
|
return 14
|
|
case "X15", "A5":
|
|
return 15
|
|
case "X16", "A6":
|
|
return 16
|
|
case "X17", "A7":
|
|
return 17
|
|
case "X18", "S2":
|
|
return 18
|
|
case "X19", "S3":
|
|
return 19
|
|
case "X20", "S4":
|
|
return 20
|
|
case "X21", "S5":
|
|
return 21
|
|
case "X22", "S6":
|
|
return 22
|
|
case "X23", "S7":
|
|
return 23
|
|
case "X24", "S8":
|
|
return 24
|
|
case "X25", "S9":
|
|
return 25
|
|
case "X26", "S10", "CTXT":
|
|
return 26
|
|
case "X27", "S11", "g":
|
|
return 27
|
|
case "X28", "T3":
|
|
return 28
|
|
case "X29", "T4":
|
|
return 29
|
|
case "X30", "T5":
|
|
return 30
|
|
case "X31", "T6", "TMP":
|
|
return 31
|
|
// Floating-point registers (F0-F31).
|
|
case "F0", "FT0":
|
|
return 0
|
|
case "F1", "FT1":
|
|
return 1
|
|
case "F2", "FT2":
|
|
return 2
|
|
case "F3", "FT3":
|
|
return 3
|
|
case "F4", "FT4":
|
|
return 4
|
|
case "F5", "FT5":
|
|
return 5
|
|
case "F6", "FT6":
|
|
return 6
|
|
case "F7", "FT7":
|
|
return 7
|
|
case "F8", "FS0":
|
|
return 8
|
|
case "F9", "FS1":
|
|
return 9
|
|
case "F10", "FA0":
|
|
return 10
|
|
case "F11", "FA1":
|
|
return 11
|
|
case "F12", "FA2":
|
|
return 12
|
|
case "F13", "FA3":
|
|
return 13
|
|
case "F14", "FA4":
|
|
return 14
|
|
case "F15", "FA5":
|
|
return 15
|
|
case "F16", "FA6":
|
|
return 16
|
|
case "F17", "FA7":
|
|
return 17
|
|
case "F18", "FS2":
|
|
return 18
|
|
case "F19", "FS3":
|
|
return 19
|
|
case "F20", "FS4":
|
|
return 20
|
|
case "F21", "FS5":
|
|
return 21
|
|
case "F22", "FS6":
|
|
return 22
|
|
case "F23", "FS7":
|
|
return 23
|
|
case "F24", "FS8":
|
|
return 24
|
|
case "F25", "FS9":
|
|
return 25
|
|
case "F26", "FS10":
|
|
return 26
|
|
case "F27", "FS11":
|
|
return 27
|
|
case "F28", "FT8":
|
|
return 28
|
|
case "F29", "FT9":
|
|
return 29
|
|
case "F30", "FT10":
|
|
return 30
|
|
case "F31", "FT11":
|
|
return 31
|
|
default:
|
|
// Vector registers V0-V31 (the "V" extension). They share the
|
|
// register numbering with the integer file: a bare number 0-31.
|
|
if len(name) >= 2 && name[0] == 'V' {
|
|
if n, ok := parseRegDigits(name[1:], 31); ok {
|
|
return n
|
|
}
|
|
}
|
|
return -1
|
|
}
|
|
}
|
|
|
|
// parseRegDigits parses a decimal register suffix and reports whether it is
|
|
// within [0, max].
|
|
func parseRegDigits(digits string, max int) (int, bool) {
|
|
if digits == "" {
|
|
return 0, false
|
|
}
|
|
n := 0
|
|
for i := 0; i < len(digits); i++ {
|
|
if digits[i] < '0' || digits[i] > '9' {
|
|
return 0, false
|
|
}
|
|
n = n*10 + int(digits[i]-'0')
|
|
if n > max {
|
|
return 0, false
|
|
}
|
|
}
|
|
return n, true
|
|
}
|
|
|
|
// RISC-V instruction encoding parameters.
|
|
type riscvEnc struct {
|
|
opcode uint32 // bits [6:0]
|
|
funct3 uint32 // bits [14:12]
|
|
funct7 uint32 // bits [31:25]
|
|
}
|
|
|
|
// riscvInstrTable maps RISC-V mnemonics to their encoding.
|
|
var riscvInstrTable = map[string]riscvEnc{
|
|
// RV64I, R-type arithmetic/logic.
|
|
"ADD": {0x33, 0x0, 0x00},
|
|
"SUB": {0x33, 0x0, 0x20},
|
|
"SLL": {0x33, 0x1, 0x00},
|
|
"SLT": {0x33, 0x2, 0x00},
|
|
"SLTU": {0x33, 0x3, 0x00},
|
|
"XOR": {0x33, 0x4, 0x00},
|
|
"SRL": {0x33, 0x5, 0x00},
|
|
"SRA": {0x33, 0x5, 0x20},
|
|
"OR": {0x33, 0x6, 0x00},
|
|
"AND": {0x33, 0x7, 0x00},
|
|
// RV64I, 32-bit variants (W suffix).
|
|
"ADDW": {0x3B, 0x0, 0x00},
|
|
"SUBW": {0x3B, 0x0, 0x20},
|
|
"SLLW": {0x3B, 0x1, 0x00},
|
|
"SRLW": {0x3B, 0x5, 0x00},
|
|
"SRAW": {0x3B, 0x5, 0x20},
|
|
// RV64I, I-type shift-immediate (shamt in rs2 field).
|
|
"SLLI": {0x13, 0x1, 0x00},
|
|
"SRLI": {0x13, 0x5, 0x00},
|
|
"SRAI": {0x13, 0x5, 0x20},
|
|
"SLLIW": {0x1B, 0x1, 0x00},
|
|
"SRLIW": {0x1B, 0x5, 0x00},
|
|
"SRAIW": {0x1B, 0x5, 0x20},
|
|
// Zba/Zbs shift-immediate forms: the shamt spans bits [25:20], so the
|
|
// funct7 field carries the operation's funct6 and bit 25 comes from the
|
|
// amount. SLLIUW (Zba) zeroes the upper 32 bits before the shift.
|
|
"BCLRI": {0x13, 0x1, 0x24},
|
|
"BEXTI": {0x13, 0x5, 0x24},
|
|
"BINVI": {0x13, 0x1, 0x34},
|
|
"BSETI": {0x13, 0x1, 0x14},
|
|
"SLLIUW": {0x1B, 0x1, 0x04},
|
|
// Zbb unary bit operations: one source register, the rs2 field fixed
|
|
// (the count or the position the operation works on).
|
|
"CLZ": {0x13, 0x1, 0x30},
|
|
"CLZW": {0x1B, 0x1, 0x30},
|
|
"CPOP": {0x13, 0x1, 0x30},
|
|
"CPOPW": {0x1B, 0x1, 0x30},
|
|
"CTZ": {0x13, 0x1, 0x30},
|
|
"CTZW": {0x1B, 0x1, 0x30},
|
|
"SEXTB": {0x13, 0x1, 0x30},
|
|
"SEXTH": {0x13, 0x1, 0x30},
|
|
"ORCB": {0x13, 0x5, 0x14},
|
|
"REV8": {0x13, 0x5, 0x35},
|
|
"ZEXTH": {0x3B, 0x4, 0x04},
|
|
// RV64M, multiply/divide.
|
|
"MUL": {0x33, 0x0, 0x01},
|
|
"MULH": {0x33, 0x1, 0x01},
|
|
"MULHSU": {0x33, 0x2, 0x01},
|
|
"MULHU": {0x33, 0x3, 0x01},
|
|
"DIV": {0x33, 0x4, 0x01},
|
|
"DIVU": {0x33, 0x5, 0x01},
|
|
"REM": {0x33, 0x6, 0x01},
|
|
"REMU": {0x33, 0x7, 0x01},
|
|
// RV64M, 32-bit variants.
|
|
"MULW": {0x3B, 0x0, 0x01},
|
|
"DIVW": {0x3B, 0x4, 0x01},
|
|
"DIVUW": {0x3B, 0x5, 0x01},
|
|
"REMW": {0x3B, 0x6, 0x01},
|
|
"REMUW": {0x3B, 0x7, 0x01},
|
|
// Zicond conditional zeroing.
|
|
"CZEROEQZ": {0x33, 0x5, 0x07},
|
|
"CZERONEZ": {0x33, 0x7, 0x07},
|
|
// Zba address generation and Zbc carry-less multiplication.
|
|
"ADDUW": {0x3B, 0x0, 0x04},
|
|
"SH1ADD": {0x33, 0x2, 0x10},
|
|
"SH1ADDUW": {0x3B, 0x2, 0x10},
|
|
"SH2ADD": {0x33, 0x4, 0x10},
|
|
"SH2ADDUW": {0x3B, 0x4, 0x10},
|
|
"SH3ADD": {0x33, 0x6, 0x10},
|
|
"SH3ADDUW": {0x3B, 0x6, 0x10},
|
|
"CLMUL": {0x33, 0x1, 0x05},
|
|
"CLMULH": {0x33, 0x3, 0x05},
|
|
"CLMULR": {0x33, 0x2, 0x05},
|
|
// Zbs single-bit: the register forms; the immediate spellings lower to
|
|
// the shift-immediate entries below (BCLR $n is BCLRI $n).
|
|
"BCLR": {0x33, 0x1, 0x24},
|
|
"BEXT": {0x33, 0x5, 0x24},
|
|
"BINV": {0x33, 0x1, 0x34},
|
|
"BSET": {0x33, 0x1, 0x14},
|
|
// RV64I, I-type arithmetic.
|
|
"ADDI": {0x13, 0x0, 0x00},
|
|
"ADDIW": {0x1B, 0x0, 0x00},
|
|
"SLTI": {0x13, 0x2, 0x00},
|
|
"SLTIU": {0x13, 0x3, 0x00},
|
|
"XORI": {0x13, 0x4, 0x00},
|
|
"ORI": {0x13, 0x6, 0x00},
|
|
"ANDI": {0x13, 0x7, 0x00},
|
|
// Loads (I-type).
|
|
"LB": {0x03, 0x0, 0x00},
|
|
"LH": {0x03, 0x1, 0x00},
|
|
"LW": {0x03, 0x2, 0x00},
|
|
"LD": {0x03, 0x3, 0x00},
|
|
"LBU": {0x03, 0x4, 0x00},
|
|
"LHU": {0x03, 0x5, 0x00},
|
|
"LWU": {0x03, 0x6, 0x00},
|
|
// Stores (S-type).
|
|
"SB": {0x23, 0x0, 0x00},
|
|
"SH": {0x23, 0x1, 0x00},
|
|
"SW": {0x23, 0x2, 0x00},
|
|
"SD": {0x23, 0x3, 0x00},
|
|
// Branches (B-type).
|
|
"BEQ": {0x63, 0x0, 0x00},
|
|
"BNE": {0x63, 0x1, 0x00},
|
|
"BLT": {0x63, 0x4, 0x00},
|
|
"BGE": {0x63, 0x5, 0x00},
|
|
"BLTU": {0x63, 0x6, 0x00},
|
|
"BGEU": {0x63, 0x7, 0x00},
|
|
// The swapped-spelling comparison forms: encoded as BLT/BGE/BLTU/BGEU
|
|
// with the register operands swapped.
|
|
"BGT": {0x63, 0x4, 0x00},
|
|
"BLE": {0x63, 0x5, 0x00},
|
|
"BGTU": {0x63, 0x6, 0x00},
|
|
"BLEU": {0x63, 0x7, 0x00},
|
|
// U-type.
|
|
"LUI": {0x37, 0x0, 0x00},
|
|
"AUIPC": {0x17, 0x0, 0x00},
|
|
// System.
|
|
"ECALL": {0x73, 0x0, 0x00},
|
|
"EBREAK": {0x73, 0x0, 0x00},
|
|
"FENCE": {0x0F, 0x0, 0x00},
|
|
"FENCE.TSO": {0x0F, 0x0, 0x00},
|
|
"PAUSE": {0x0F, 0x0, 0x00},
|
|
// JALR, indirect jump/call (I-type).
|
|
"JALR": {0x67, 0x0, 0x00},
|
|
|
|
// RV64A, atomics (AMO opcode 0x2F).
|
|
// funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
|
|
// 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
|
|
// the lower two, exactly as the toolchain writes them: every AMO sets both
|
|
// aq and rl (funct7 |= 3).
|
|
"AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
|
|
"AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
|
|
"AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
|
|
"AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
|
|
"AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
|
|
"AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
|
|
"AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
|
|
"AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
|
|
"AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
|
|
"AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
|
|
"AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
|
|
"AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
|
|
"AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
|
|
"AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
|
|
"AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
|
|
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
|
|
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
|
|
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
|
|
|
|
// RV64F/D, floating-point arithmetic.
|
|
"FADDS": {0x53, 0x0, 0x00},
|
|
"FSUBS": {0x53, 0x0, 0x04},
|
|
"FMULS": {0x53, 0x0, 0x08},
|
|
"FDIVS": {0x53, 0x0, 0x0C},
|
|
"FADDD": {0x53, 0x0, 0x01},
|
|
"FSUBD": {0x53, 0x0, 0x05},
|
|
"FMULD": {0x53, 0x0, 0x09},
|
|
"FDIVD": {0x53, 0x0, 0x0D},
|
|
"FSQRTS": {0x53, 0x0, 0x2C},
|
|
"FSQRTD": {0x53, 0x0, 0x2D},
|
|
// FP loads/stores.
|
|
"FLW": {0x07, 0x2, 0x00},
|
|
"FLD": {0x07, 0x3, 0x00},
|
|
"FSW": {0x27, 0x2, 0x00},
|
|
"FSD": {0x27, 0x3, 0x00},
|
|
// FP min/max.
|
|
"FMINS": {0x53, 0x0, 0x14},
|
|
"FMAXS": {0x53, 0x1, 0x14},
|
|
"FMIND": {0x53, 0x0, 0x15},
|
|
"FMAXD": {0x53, 0x1, 0x15},
|
|
// FP sign injection (double): rs2 carries the sign source.
|
|
"FSGNJD": {0x53, 0x0, 0x11},
|
|
"FSGNJS": {0x53, 0x0, 0x10},
|
|
"FSGNJX": {0x53, 0x0, 0x14},
|
|
"FSGNJXD": {0x53, 0x0, 0x15},
|
|
"FSGNJXS": {0x53, 0x0, 0x14},
|
|
"FSGNJND": {0x53, 0x1, 0x11},
|
|
"FSGNJNS": {0x53, 0x1, 0x10},
|
|
"FSGNJNX": {0x53, 0x1, 0x14},
|
|
|
|
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
|
// The toolchain gives LR acquire ordering (aq = 1) and SC release
|
|
// ordering (rl = 1).
|
|
"LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
|
|
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
|
|
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
|
|
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
|
|
|
|
// FP compare, result in integer register (funct7 0x50/0x51).
|
|
"FEQS": {0x53, 0x2, 0x50},
|
|
"FLTS": {0x53, 0x1, 0x50},
|
|
"FLES": {0x53, 0x0, 0x50},
|
|
"FEQD": {0x53, 0x2, 0x51},
|
|
"FLTD": {0x53, 0x1, 0x51},
|
|
"FLED": {0x53, 0x0, 0x51},
|
|
}
|
|
|
|
// riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode.
|
|
func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
|
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
|
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
|
}
|
|
|
|
// riscvAMOType encodes an atomic (AMO) instruction.
|
|
// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
|
|
// funct5 in its upper five bits and the aq/rl ordering bits in the lower two
|
|
// (the table stores the full field, so the word needs no reassembly).
|
|
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
|
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
|
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
|
}
|
|
|
|
// FP conversion instructions (FCVT, FMV). These use the rs2 field to
|
|
// encode the conversion type rather than a register, so they are handled
|
|
// separately from the general instruction table.
|
|
type riscvCvtEnc struct {
|
|
funct7 uint32 // bits [31:25]
|
|
rs2 uint32 // conversion-type code in bits [24:20]
|
|
opcode uint32 // always 0x53 (OP-FP)
|
|
}
|
|
|
|
var riscvCvtTable = map[string]riscvCvtEnc{
|
|
// float → int (rs2 selects the integer width/sign).
|
|
"FCVTWS": {0x60, 0x0, 0x53}, // float32 → int32
|
|
"FCVTWUS": {0x60, 0x1, 0x53}, // float32 → uint32
|
|
"FCVTLS": {0x60, 0x2, 0x53}, // float32 → int64
|
|
"FCVTLUS": {0x60, 0x3, 0x53}, // float32 → uint64
|
|
"FCVTWD": {0x61, 0x0, 0x53}, // float64 → int32
|
|
"FCVTWUD": {0x61, 0x1, 0x53}, // float64 → uint32
|
|
"FCVTLD": {0x61, 0x2, 0x53}, // float64 → int64
|
|
"FCVTLUD": {0x61, 0x3, 0x53}, // float64 → uint64
|
|
// int → float (rs2 selects the integer width/sign).
|
|
"FCVTSW": {0x68, 0x0, 0x53}, // int32 → float32
|
|
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
|
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
|
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
|
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
|
|
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
|
|
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
|
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
|
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
|
"FCVTDLU": {0x69, 0x3, 0x53}, // uint64 → float64
|
|
// float → float width conversion.
|
|
"FCVTSD": {0x20, 0x1, 0x53}, // float64 → float32
|
|
"FCVTDS": {0x21, 0x0, 0x53}, // float32 → float64
|
|
// Bit moves between integer and FP registers (no conversion).
|
|
"FMVXD": {0x71, 0x0, 0x53}, // float64 → int64 (bit move)
|
|
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
|
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
|
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
|
// The toolchain's W/D suffix spellings of the same moves.
|
|
"FMVXS": {0x70, 0x0, 0x53},
|
|
"FMVFS": {0x78, 0x0, 0x53},
|
|
"FMVSX": {0x79, 0x0, 0x53},
|
|
}
|
|
|
|
// riscvCvtType encodes an FP conversion instruction.
|
|
// Layout: funct7 | rs2(convtype) | rs1 | funct3(0) | rd | opcode.
|
|
func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 {
|
|
return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) |
|
|
(uint32(rd) << 7) | enc.opcode
|
|
}
|
|
|
|
// R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD).
|
|
// These take 4 register operands: rs1, rs2, rs3, rd.
|
|
// Layout: rs3 | fmt | rs2 | rs1 | rm | rd | opcode.
|
|
type riscvFmaEnc struct {
|
|
fmt uint32 // bits [26:25]: 0x0 = single, 0x1 = double
|
|
opcode uint32 // bits [6:0]
|
|
}
|
|
|
|
var riscvFmaTable = map[string]riscvFmaEnc{
|
|
"FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3
|
|
"FMADDD": {0x1, 0x43},
|
|
"FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3
|
|
"FMSUBD": {0x1, 0x47},
|
|
"FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3
|
|
"FNMSUBD": {0x1, 0x4B},
|
|
"FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3
|
|
"FNMADDD": {0x1, 0x4F},
|
|
}
|
|
|
|
// riscvFmaType encodes an R4-type fused multiply-add instruction.
|
|
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
|
|
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
|
|
(uint32(rs1) << 15) | (0x0 << 12) /* rm=RNE */ | (uint32(rd) << 7) | enc.opcode
|
|
}
|
|
|
|
// CSR (Control and Status Register) instructions.
|
|
// Format: csr[11:0] | rs1/zimm | funct3 | rd | opcode (0x73).
|
|
type riscvCsrEnc struct {
|
|
funct3 uint32 // bits [14:12]
|
|
imm bool // true for CSRRWI/CSRRSI/CSRRCI (5-bit uimm variant)
|
|
}
|
|
|
|
var riscvCsrTable = map[string]riscvCsrEnc{
|
|
"CSRRW": {0x1, false}, // rd=CSR, CSR=rs1
|
|
"CSRRS": {0x2, false}, // rd=CSR, CSR |= rs1
|
|
"CSRRC": {0x3, false}, // rd=CSR, CSR &= ~rs1
|
|
"CSRRWI": {0x5, true}, // rd=CSR, CSR=uimm
|
|
"CSRRSI": {0x6, true}, // rd=CSR, CSR |= uimm
|
|
"CSRRCI": {0x7, true}, // rd=CSR, CSR &= ~uimm
|
|
}
|
|
|
|
// riscvCsrType encodes a CSR instruction.
|
|
// csr is the 12-bit CSR address; src is either a register number or a 5-bit
|
|
// unsigned immediate (depending on enc.imm).
|
|
func riscvCsrType(enc riscvCsrEnc, rd, src int, csr int32) uint32 {
|
|
return (uint32(csr&0xFFF) << 20) | (uint32(src&0x1F) << 15) |
|
|
(enc.funct3 << 12) | (uint32(rd) << 7) | 0x73
|
|
}
|
|
|
|
// riscvIType encodes an I-type instruction: imm[11:0] | rs1 | funct3 | rd | opcode.
|
|
func riscvIType(enc riscvEnc, rd, rs1 int, imm int32) uint32 {
|
|
return (uint32(imm&0xFFF) << 20) | (uint32(rs1) << 15) |
|
|
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
|
}
|
|
|
|
// riscvSType encodes an S-type instruction: imm[11:5] | rs2 | rs1 | funct3 | imm[4:0] | opcode.
|
|
func riscvSType(enc riscvEnc, rs1, rs2 int, imm int32) uint32 {
|
|
immU := uint32(imm) & 0xFFF
|
|
return ((immU >> 5) << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
|
(enc.funct3 << 12) | ((immU & 0x1F) << 7) | enc.opcode
|
|
}
|
|
|
|
// riscvBType encodes a B-type instruction (branches).
|
|
func riscvBType(enc riscvEnc, rs1, rs2 int, offset int32) uint32 {
|
|
imm := uint32(offset) & 0x1FFE // bits [12:1], bit 0 is always 0
|
|
return (((imm >> 12) & 1) << 31) | // imm[12]
|
|
(((imm >> 5) & 0x3F) << 25) | // imm[10:5]
|
|
(uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
|
(enc.funct3 << 12) |
|
|
(((imm >> 1) & 0xF) << 8) | // imm[4:1]
|
|
(((imm >> 11) & 1) << 7) | // imm[11]
|
|
enc.opcode
|
|
}
|
|
|
|
// riscvUType encodes a U-type instruction: imm[31:12] | rd | opcode.
|
|
func riscvUType(enc riscvEnc, rd int, imm int32) uint32 {
|
|
return (uint32(imm) & 0xFFFFF000) | (uint32(rd) << 7) | enc.opcode
|
|
}
|
|
|
|
// riscvJType encodes a J-type instruction (JAL).
|
|
func riscvJType(rd int, offset int32) uint32 {
|
|
imm := uint32(offset) & 0x1FFFFE // bits [20:1]
|
|
return (((imm >> 20) & 1) << 31) | // imm[20]
|
|
(((imm >> 1) & 0x3FF) << 21) | // imm[10:1]
|
|
(((imm >> 11) & 1) << 20) | // imm[11]
|
|
(((imm >> 12) & 0xFF) << 12) | // imm[19:12]
|
|
(uint32(rd) << 7) |
|
|
0x6F // JAL opcode
|
|
}
|
|
|
|
// ---- RVV ("V" extension) encoding helpers ----
|
|
|
|
// The OP-V major opcode and its funct3 subclasses.
|
|
const (
|
|
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
|
|
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
|
|
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
|
|
riscvVf3VV = 0x0 // vector-vector
|
|
riscvVf3MV = 0x2 // vector mask
|
|
riscvVf3VI = 0x3 // vector-immediate
|
|
riscvVf3VX = 0x4 // vector-scalar
|
|
riscvVf3Cfg = 0x7 // vsetvli
|
|
)
|
|
|
|
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
|
|
// group multiplier in [2:0], the selected element width in [5:3] and the
|
|
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
|
|
func riscvVType(vsew, vlmul, vta, vma int) int {
|
|
return vlmul | vsew<<3 | vta<<6 | vma<<7
|
|
}
|
|
|
|
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
|
|
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
|
|
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
|
|
// the toolchain writes above its 10-bit vtype).
|
|
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
|
|
imm := vtype & 0x3FF
|
|
if vsetivli {
|
|
imm |= 0xC00
|
|
}
|
|
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
|
|
uint32(rd)<<7 | riscvOpV
|
|
}
|
|
|
|
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
|
|
// segment count in bits [31:29], the addressing mode in bits [28:26], the
|
|
// unmasked bit at 25 and the width in funct3. width follows the load
|
|
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
|
|
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
|
|
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
|
|
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
|
|
uint32(rd)<<7 | op
|
|
}
|
|
|
|
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
|
|
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
|
|
// the standard positions. vs1 may name an integer register for the *VX forms
|
|
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
|
|
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
|
|
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
|
|
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
|
|
}
|
|
|
|
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
|
|
// fields live where the second source register would be: rs1Field and vs2 are
|
|
// written verbatim (the oracle writes fixed non-zero constants there for some
|
|
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
|
|
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
|
|
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
|
|
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
|
|
}
|
|
|
|
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
|
|
func riscvSegNF(n int) int32 { return int32(n - 1) }
|
|
|
|
// ---- RVC (compressed) encoding helpers ----
|
|
|
|
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
|
// prime register field used by compressed instructions (x8-x15).
|
|
func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 }
|
|
|
|
// rvcReg3 returns the 3-bit encoding for registers x8-x15 (0-7).
|
|
func rvcReg3(r int) uint32 { return uint32(r - 8) }
|
|
|
|
// rvcCR encodes a CR-type (register) compressed instruction.
|
|
// Format: funct4 | rd/rs1 | rs2 | op=2.
|
|
func rvcCR(funct4, rd, rs2 uint32) uint16 {
|
|
return uint16((funct4 << 12) | (rd << 7) | (rs2 << 2) | 0x2)
|
|
}
|
|
|
|
// rvcCI encodes a CI-type (immediate) compressed instruction.
|
|
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW, linear 6-bit immediate.
|
|
func rvcCI(funct3, rd uint32, imm uint32) uint16 {
|
|
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
|
|
}
|
|
|
|
// rvcSLLI encodes C.SLLI, which shares funct3=0 with C.ADDI but lives in the
|
|
// op=10 quadrant (unlike C.ADDI's op=01).
|
|
func rvcSLLI(rd, shamt uint32) uint16 {
|
|
return uint16(((shamt>>5)&1)<<12 | (rd << 7) | (shamt&0x1F)<<2 | 0x2)
|
|
}
|
|
|
|
// encodeRVCPattern extracts the bits listed in pattern (MSB first) from imm
|
|
// into a packed value, matching cmd/internal/obj/riscv's encodeBitPattern.
|
|
func encodeRVCPattern(imm uint32, pattern []int) uint32 {
|
|
packed := uint32(0)
|
|
for _, bit := range pattern {
|
|
packed = packed<<1 | (imm>>bit)&1
|
|
}
|
|
return packed
|
|
}
|
|
|
|
// rvcLSP encodes a stack-relative compressed load (op=10 quadrant): C.LWSP
|
|
// (funct3=2, 4-byte scale), C.LDSP (funct3=3) or C.FLDSP (funct3=1, 8-byte
|
|
// scale). offset is the full byte offset.
|
|
func rvcLSP(funct3, rd uint32, offset uint32) uint16 {
|
|
pattern := []int{5, 4, 3, 8, 7, 6}
|
|
if funct3 == 0x2 {
|
|
pattern = []int{5, 4, 3, 2, 7, 6}
|
|
}
|
|
packed := uint32(0)
|
|
for i, b := range pattern {
|
|
packed |= ((offset >> b) & 1) << (5 - i)
|
|
}
|
|
return uint16((funct3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x2)
|
|
}
|
|
|
|
// rvcSSP encodes a stack-relative compressed store (op=10 quadrant): C.SWSP
|
|
// (funct3=6, 4-byte scale), C.SDSP (funct3=7) or C.FSDSP (funct3=5, 8-byte
|
|
// scale). offset is the full byte offset.
|
|
func rvcSSP(funct3, rs2 uint32, offset uint32) uint16 {
|
|
pattern := []int{5, 4, 3, 8, 7, 6}
|
|
if funct3 == 0x6 {
|
|
pattern = []int{5, 4, 3, 2, 7, 6}
|
|
}
|
|
packed := uint32(0)
|
|
for i, b := range pattern {
|
|
packed |= ((offset >> b) & 1) << (5 - i)
|
|
}
|
|
return uint16((funct3 << 13) | (packed << 7) | (rs2 << 2) | 0x2)
|
|
}
|
|
|
|
// rvcCL encodes a register-relative compressed load (op=00 quadrant): C.LW
|
|
// (funct3=2), C.LD (funct3=3) or C.FLD (funct3=1). imm is the full byte
|
|
// offset; the immediate bits are extracted per the RISC-V CL format.
|
|
func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
|
|
pattern := []int{5, 4, 3, 7, 6}
|
|
if funct3 == 0x2 {
|
|
pattern = []int{5, 4, 3, 2, 6}
|
|
}
|
|
packed := encodeRVCPattern(imm, pattern)
|
|
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rd << 2))
|
|
}
|
|
|
|
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
|
|
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
|
|
// offset; the immediate bits are extracted per the RISC-V CS format, with the
|
|
// same five-bit patterns as the load side ({5,4,3,7,6} and {5,4,3,2,6},
|
|
// matching the toolchain's encodeCS).
|
|
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
|
|
pattern := []int{5, 4, 3, 7, 6}
|
|
if funct3 == 0x6 {
|
|
pattern = []int{5, 4, 3, 2, 6}
|
|
}
|
|
packed := encodeRVCPattern(imm, pattern)
|
|
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
|
|
}
|
|
|
|
// rvcCIW encodes a CIW-type compressed immediate wide instruction: C.ADDI4SPN
|
|
// (funct3=0). imm is the raw byte offset.
|
|
func rvcCIW(funct3, rd uint32, imm uint32) uint16 {
|
|
packed := encodeRVCPattern(imm, []int{5, 4, 9, 8, 7, 6, 2, 3})
|
|
return uint16((funct3 << 13) | (packed << 5) | (rd << 2))
|
|
}
|
|
|
|
// rvcCA encodes a CA-type (arithmetic) compressed instruction.
|
|
// Format: funct6[15:10] | rd'/rs1'[9:7] | funct2[6:5] | rs2'[4:2] | op=01.
|
|
func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 {
|
|
return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1)
|
|
}
|
|
|
|
// rvcCBShift encodes a CB-type shift/immediate compressed instruction
|
|
// (C.SRLI, C.SRAI, C.ANDI). rd is the 3-bit prime-register index; imm is
|
|
// the 6-bit shamt/immediate; funct2 selects the operation (0=SRLI, 1=SRAI,
|
|
// 2=ANDI).
|
|
func rvcCBShift(funct2, rd, imm uint32) uint16 {
|
|
return uint16((0x4 << 13) | ((imm>>5)&1)<<12 | (funct2 << 10) | (rd << 7) | (imm&0x1F)<<2 | 0x1)
|
|
}
|
|
|
|
// rvcADDI16SP encodes C.ADDI16SP: ADDI rd, imm, rd for the stack pointer
|
|
// with a 10-bit signed, 16-byte-scaled immediate. imm is the raw byte
|
|
// offset; the immediate bits are extracted in the order [9|4|6|8:7|5].
|
|
func rvcADDI16SP(rd uint32, imm int32) uint16 {
|
|
u := uint32(imm)
|
|
packed := uint32(0)
|
|
for _, bit := range []uint{9, 4, 6, 8, 7, 5} {
|
|
packed = packed<<1 | (u>>bit)&1
|
|
}
|
|
return uint16((0x3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x1)
|
|
}
|