Files
gasm-sdk/asm/riscv_assemble.go
T

3047 lines
98 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"errors"
"fmt"
"math/bits"
"slices"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// assembleRISCV assembles a RISC-V TEXT function body into machine code.
// It handles the full RV64IMAFDC instruction set including RVC compression.
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, []RiscvLiteral, error) {
fi := riscvComputeFrame(t)
prologue := riscvPrologue(fi)
guardLen, err := riscvGuardLen(fi)
if err != nil {
return nil, nil, nil, nil, nil, nil, err
}
lits := &riscvLiterals{}
var relocs []Reloc
var spadj []SpadjStep
// The prologue raises the SP delta by autosize; the boundary is reported
// at the pc just past its ADDI, exactly as the toolchain's pctospadj does.
// The guard prefix shifts its PC.
if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: guardLen + riscvPrologueSpadjPC(fi), Value: fi.autosize})
}
// Pass 1: collect instructions and compute label offsets assuming 4 bytes
// per instruction (or 8 for MOV $large-imm). No encoding yet. PCALIGN
// contributes only its padding, which is attached to the following
// instruction and emitted ahead of it. A relaxed branch carries the
// inverted condition and is followed by an inserted JMP rec (jmpTo set)
// that carries the original target.
type instrRec struct {
instr *ast.Instr
compressed bool
code []byte
pad int
relaxed bool
jmpTo string
}
var recs []instrRec
offsets := map[string]int{}
pos := guardLen + len(prologue)
pendingPad := 0
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pendingPad += riscvPCAlignPad(pos, s)
pos += riscvPCAlignPad(pos, s)
continue
}
recs = append(recs, instrRec{instr: s, pad: pendingPad})
pendingPad = 0
pos += riscvInstrSize(s, fi)
}
}
// Pass 2: encode each instruction using Pass-1 offsets. A branch or
// jump the offsets prove overlong encodes to a 4-byte placeholder: the
// relaxation pass rewrites it before the final encoding. pcRelPcs is
// unavailable this early, so the N(PC) forms take the same placeholder
// path.
pc := len(prologue)
for i := range recs {
branchLike := isBranchLike(recs[i].instr.Mnemonic.Text) || riscvIsCondBranch(recs[i].instr.Mnemonic.Text)
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil, nil, lits) // no relocs in Pass 2
if err != nil && !(branchLike && riscvIsRangeError(err)) {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
}
if err != nil {
code = make([]byte, 4)
}
recs[i].code = code
pc += len(code)
}
// Pass 3: try RVC compression.
for i := range recs {
if c16, ok := tryCompressRVC(recs[i].instr, fi); ok {
recs[i].compressed = true
recs[i].code = []byte{byte(c16), byte(c16 >> 8)}
}
}
// Pass 4: recompute offsets with actual sizes. recs holds the
// instructions in emission order, so an index into it walks t.Body in
// lockstep (the same single pass Pass 1 uses) instead of rescanning the
// whole slice per statement. PCALIGN padding is recomputed here, since
// compression has shifted instruction sizes since Pass 1.
offsets = map[string]int{}
pos = guardLen + len(prologue)
ri := 0
pendingPad = 0
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pad := riscvPCAlignPad(pos, s)
pendingPad += pad
pos += pad
continue
}
recs[ri].pad = pendingPad
pendingPad = 0
pos += len(recs[ri].code)
ri++
}
}
// Pass 4b: relax overlong conditional branches exactly as the toolchain
// does: invert the branch condition, point it at the instruction after an
// inserted JMP, let the JMP carry the original target, and re-layout until
// a pass inserts nothing. Inserted JMP recs share their branch's source
// line and trail it in emission order, so the body walk flushes them
// before every statement and at the end.
var pcRelPcs map[*ast.Instr]int
for {
offsets = map[string]int{}
pos = guardLen + len(prologue)
ri := 0
pcs := make([]int, len(recs))
flushJmps := func() {
for ri < len(recs) && recs[ri].jmpTo != "" {
pcs[ri] = pos
pos += 4
ri++
}
}
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
flushJmps()
offsets[s.Name.Text] = pos
case *ast.Instr:
flushJmps()
if ri >= len(recs) {
continue
}
pcs[ri] = pos + recs[ri].pad
pos += recs[ri].pad + len(recs[ri].code)
ri++
}
}
flushJmps()
changed := false
for i := range recs {
r := &recs[i]
if r.relaxed || r.jmpTo != "" {
continue
}
mnem := strings.ToUpper(r.instr.Mnemonic.Text)
if !riscvIsCondBranch(mnem) || len(r.instr.Operands) == 0 {
continue
}
target := labelFromOperand(r.instr.Operands[len(r.instr.Operands)-1])
targetOff, ok := offsets[target]
if !ok {
continue
}
if delta := int32(targetOff - pcs[i]); delta < -4096 || delta >= 4096 {
r.relaxed = true
recs = slices.Insert(recs, i+1, instrRec{instr: r.instr, jmpTo: target})
changed = true
}
}
if !changed {
// Capture the final pcs for the N(PC) branch and jump forms: the
// target is the instruction N source slots away (N=0 the branch
// itself, N negative backwards), resolved by index against the
// final layout.
pcRelPcs = map[*ast.Instr]int{}
for i := range recs {
n, ok := riscvPCRelOffset(recs[i].instr)
if !ok {
continue
}
if i+n < 0 || i+n >= len(recs) {
continue
}
pcRelPcs[recs[i].instr] = pcs[i+n]
}
break
}
}
// Pass 5: re-encode branches with corrected offsets. Record relocations
// during this final pass (relocation offsets are relative to instruction
// start). The guard prefix precedes the prologue; its branches target
// the morestack block at the end of the function, which the previous
// passes have sized.
var out []byte
guardBytes, guardReloc, err := riscvGuard(fi)
if err != nil {
return nil, nil, nil, nil, nil, nil, err
}
if fi.needSplit {
out = append(out, guardBytes...)
}
out = append(out, prologue...)
pc = guardLen + len(prologue)
preCount := len(relocs)
var lines []LineEntry
for _, r := range recs {
// PCALIGN padding precedes the instruction it was attached to.
if r.pad > 0 {
out = append(out, riscvPadBytes(r.pad)...)
pc += r.pad
}
lines = append(lines, LineEntry{Offset: pc, Line: r.instr.Pos().Line})
var code []byte
switch {
case r.jmpTo != "":
// The JMP a relaxation inserted: JAL X0 to the original target.
targetOff, ok := offsets[r.jmpTo]
if !ok {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("undefined label %q", r.jmpTo)
}
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(r.jmpTo, offset); err != nil {
return nil, nil, nil, nil, nil, nil, err
}
word := riscvJType(0, offset)
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
case r.relaxed:
// The inverted half of a relaxed branch: it targets the inserted
// JMP, always the very next instruction (offset 4).
enc, rs1, rs2, ok := riscvInvertedBranchEnc(strings.ToUpper(r.instr.Mnemonic.Text), r.instr.Operands)
if !ok {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: cannot relax branch", r.instr.Mnemonic.Text)
}
word := riscvBType(enc, rs1, rs2, 4)
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
case r.compressed && !isBranchLike(r.instr.Mnemonic.Text):
code = r.code
default:
var err error
code, err = encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs, pcRelPcs, lits)
if err != nil {
return nil, nil, nil, nil, nil, nil, err
}
if c16, ok := tryCompressRVC(r.instr, fi); ok {
code = []byte{byte(c16), byte(c16 >> 8)}
}
2026-08-13 18:12:22 +02:00
// Make newly added relocation offsets function-relative. Each
// instruction records its reloc offset relative to its own start;
// the current pc is that instruction's offset from the function
// start (which includes the prologue). After is the address just
// past the relocated field, shifted by the same amount.
for j := preCount; j < len(relocs); j++ {
2026-08-13 18:12:22 +02:00
relocs[j].Off += pc
relocs[j].After += pc
}
preCount = len(relocs)
// The RET's epilogue closes the frame: the SP delta returns to zero
// after its ADDI (restore LR + ADDI).
if strings.ToUpper(r.instr.Mnemonic.Text) == "RET" && fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: pc + riscvReturnEpilogueLen(fi), Value: 0})
}
}
out = append(out, code...)
pc += len(code)
}
if fi.needSplit {
relocs = append(relocs, guardReloc)
}
return out, offsets, relocs, lines, spadj, lits.list(), nil
}
// riscvImmAlias maps the R-type ALU mnemonics onto their I-type immediate
// forms: the toolchain accepts ADD $imm, rj, rd and emits addi. Applied
// whenever the first operand is an immediate.
var riscvImmAlias = map[string]string{
"ADD": "ADDI",
"ADDW": "ADDIW",
"AND": "ANDI",
"OR": "ORI",
"XOR": "XORI",
"SLT": "SLTI",
"SLTU": "SLTIU",
"SLL": "SLLI",
"SRL": "SRLI",
"SRA": "SRAI",
"SLLW": "SLLIW",
"SRLW": "SRLIW",
"SRAW": "SRAIW",
}
// riscvNormaliseImmAlias rewrites the mnemonic to its immediate form when the
// first operand is an immediate: the toolchain accepts ADD $imm, rj, rd and
// emits addi, and SUB $imm becomes addi with the negated immediate. The
// second result reports that negation; the operand itself is left untouched
// because several passes normalise the same instruction.
func riscvNormaliseImmAlias(mnem string, ops []*ast.Operand) (string, bool) {
if len(ops) >= 2 && isImmOperand(ops[0]) {
switch strings.ToUpper(mnem) {
case "SUB":
return "ADDI", true
case "SUBW":
return "ADDIW", true
}
if alias, ok := riscvImmAlias[strings.ToUpper(mnem)]; ok {
return alias, false
}
}
return mnem, false
}
// riscvPCAlignPad returns the padding PCALIGN inserts before the next
// instruction so that it starts at the requested boundary relative to the
// function start. The boundary must be a power of two between 8 and 2048, as
// the toolchain requires; anything else pads nothing.
func riscvPCAlignPad(pos int, instr *ast.Instr) int {
if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) {
return 0
}
align := int(immFromOperand(instr.Operands[0]))
if align < 8 || align > 2048 || align&(align-1) != 0 {
return 0
}
return (align - pos%align) % align
}
// riscvPadBytes renders PCALIGN padding: 4-byte NOPs (addi $0, X0, X0) with a
// trailing 2-byte compressed NOP when the pad is 2 mod 4, exactly as the
// toolchain lays the bytes down.
func riscvPadBytes(pad int) []byte {
out := make([]byte, 0, pad)
for ; pad >= 4; pad -= 4 {
out = append(out, 0x13, 0x00, 0x00, 0x00)
}
if pad == 2 {
out = append(out, 0x01, 0x00)
}
return out
}
// riscvInstrSize returns the encoded size in bytes of a RISC-V instruction.
2026-08-13 17:41:16 +02:00
// Most instructions are 4 bytes; MOV with a large immediate and I-type
// arithmetic with a large immediate expand to several (possibly compressed)
// instructions.
func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
mnem := instr.Mnemonic.Text
ops := instr.Operands
mnem = riscvNormalisePseudo(mnem)
if mnem == "FUNCDATA" || mnem == "PCDATA" {
// The bookkeeping statements contribute no bytes.
return 0
}
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
if mnem == "RET" {
return len(riscvReturn(fi))
}
if strings.HasPrefix(mnem, "MOV") && len(ops) == 2 {
// MOV $sym(SB), rd → 8 bytes (AUIPC + ADDI).
if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && ops[0].Imm.Sym.Pseudo == "SB" {
return 8
}
// MOV sym(SB), rd → 8 bytes (AUIPC + LD).
if isMemOperand(ops[0]) && ops[0].Addr.Sym != nil && ops[0].Addr.Sym.Pseudo == "SB" {
return 8
}
// MOV rd, sym(SB) → 8 bytes (AUIPC + SD).
if isMemOperand(ops[1]) && ops[1].Addr.Sym != nil && ops[1].Addr.Sym.Pseudo == "SB" {
return 8
}
2026-08-13 17:41:16 +02:00
// MOV $imm, rd → size depends on the immediate and RVC compression.
if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil {
imm := riscvOperandImm64(ops[0])
if int64(int32(imm)) != imm {
return riscvMovImm64Size(regFromOperand(ops[1]), imm)
}
return riscvMovImmSize(regFromOperand(ops[1]), int32(imm))
}
// MOV $sym+off(FP|SP), rd → the frame-adjusted offset as an ADDI,
// compressed like riscvSPAddiBytes encodes it.
if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil &&
(ops[0].Imm.Sym.Pseudo == "FP" || ops[0].Imm.Sym.Pseudo == "SP") {
rd := regFromOperand(ops[1])
_, off := riscvResolvePseudo(ops[0].Imm.Sym, fi)
if rd > 0 && off == 0 {
return 2 // C.MV rd, SP
}
if isRVCIntReg(rd) && off > 0 && off < 1024 && off%4 == 0 {
return 2 // C.ADDI4SPN
}
return riscvItypeImmediateSize("ADDI", off)
}
// Frame-relative loads and stores: a frame offset beyond the signed
// 12-bit range materialises the address in X31 first.
if isMemOperand(ops[0]) && !isMemOperand(ops[1]) {
return riscvFrameMemSize(ops[0], fi)
}
if isMemOperand(ops[1]) && !isMemOperand(ops[0]) {
return riscvFrameMemSize(ops[1], fi)
}
}
// I-type arithmetic with a large immediate expands to several instructions.
if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) {
imm := immFromOperand(ops[0])
if immNeg {
imm = -imm
}
return riscvItypeImmediateSize(mnem, imm)
}
// BYTE lays down one raw byte per operand.
if mnem == "BYTE" {
return len(ops)
}
// The toolchain's synthesised instructions: some emit one word, others
// expand to a fixed sequence.
return riscvExtendedSize(mnem, ops)
}
// riscvExtendedSize returns the encoded size of the instructions the
// toolchain synthesises from other instructions (the ternary expansions and
// the vector slice); every caller keeps the layout in step with
// encodeRISCVExtended, which emits exactly these bytes.
func riscvExtendedSize(mnem string, ops []*ast.Operand) int {
switch mnem {
case "NOP":
// The toolchain drops a bare NOP entirely.
return 0
case "ANDN", "ORN":
return 8
case "MAX", "MAXU", "MIN", "MINU":
if riscvIdenticalMinMax(mnem, ops) {
rd := regFromOperand(ops[1])
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rd != 0 {
return 2 // C.MV, or C.LI when the sources are X0
}
return 4
}
return 20
case "ROR", "RORW":
if len(ops) >= 1 && isImmOperand(ops[0]) {
// SRL + [compressed] SLL of the reverse shift + OR.
return 4 + riscvRevShiftSize(mnem, ops) + 4
}
return 16 // SUB + shift + shift + OR
case "RORIW":
return 12
}
return 4
}
// riscvIdenticalMinMax reports whether a MIN/MAX sees two identical source
// registers (the toolchain folds that to ADDI $0).
func riscvIdenticalMinMax(mnem string, ops []*ast.Operand) bool {
if mnem != "MAX" && mnem != "MAXU" && mnem != "MIN" && mnem != "MINU" {
return false
}
if len(ops) != 2 && len(ops) != 3 {
return false
}
rs1 := regFromOperand(ops[1])
rs2 := regFromOperand(ops[0])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 == rd {
// The toolchain swaps the sources so the destination-identical one
// is processed first; identical sources stay identical.
rs1, rs2 = rs2, rs1
}
return rs1 >= 0 && rs1 == rs2
}
// riscvRevShiftSize returns the size of the reverse-shift instruction inside
// a ROR/RORW immediate expansion: the SLLI of the complementary amount, which
// compresses to C.SLLI only in the 64-bit form when rd == rs1, both non-zero,
// and the amount lands in 1-63. The W forms have no compressed shift.
func riscvRevShiftSize(mnem string, ops []*ast.Operand) int {
if mnem != "ROR" {
return 4 // SLLIW has no compressed form
}
imm := int(immFromOperand(ops[0]))
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
sll := (-imm) & 63
if rd == rs1 && rd != 0 && sll >= 1 && sll <= 63 {
return 2 // C.SLLI
}
return 4
}
// isBranchLike reports whether a mnemonic is a branch or jump that needs
// recalculated offsets after compression.
func isBranchLike(mnem string) bool {
switch mnem {
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "JMP", "JAL":
return true
}
return false
}
// riscvIsCondBranch reports whether m is a conditional branch, the only
// instruction class branch relaxation rewrites.
func riscvIsCondBranch(mnem string) bool {
switch mnem {
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU",
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
return true
}
return false
}
// riscvCSRNames maps the standard CSR mnemonics the assembler accepts onto
// their addresses.
var riscvCSRNames = map[string]int32{
"FFLAGS": 0x001,
"FRM": 0x002,
"FCSR": 0x003,
"VSTART": 0x008,
"VXSAT": 0x009,
"VXRM": 0x00A,
"VCSR": 0x00F,
"CYCLE": 0xC00,
"TIME": 0xC01,
"INSTRET": 0xC02,
"CYCLEH": 0xC80,
"TIMEH": 0xC81,
"INSTRETH": 0xC82,
"VL": 0xC20,
"VLENB": 0xC22,
}
// riscvCSRAddress resolves a CSR operand: an integer immediate or one of the
// standard CSR names.
func riscvCSRAddress(op *ast.Operand) (int32, bool) {
if isImmOperand(op) {
return immFromOperand(op), true
}
if op.Addr.Sym != nil {
if v, ok := riscvCSRNames[strings.ToUpper(op.Addr.Sym.Name)]; ok {
return v, true
}
}
return 0, false
}
// riscvPCRelOffset reports the N of a branch or jump operand spelled N(PC):
// the displacement counted in source instructions from the branch itself.
func riscvPCRelOffset(instr *ast.Instr) (int, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
switch mnem {
case "JMP":
if len(instr.Operands) != 1 {
return 0, false
}
case "JAL":
if len(instr.Operands) != 1 && len(instr.Operands) != 2 {
return 0, false
}
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU",
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
if len(instr.Operands) < 2 {
return 0, false
}
default:
return 0, false
}
op := instr.Operands[len(instr.Operands)-1]
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" {
return int(op.Addr.Offset), true
}
return 0, false
}
// riscvPCRelTargetOff resolves the target displacement of a branch whose last
// operand is N(PC): the toolchain's parser counts the source instructions at
// a uniform 4 bytes, so the target is the instruction N slots away, and the
// displacement tracks that instruction's final pc. A nil pcRelPcs (the
// layout passes) yields a placeholder range error; the caller tolerates it
// for branch-like instructions.
func riscvPCRelTargetOff(instr *ast.Instr, pc int, pcRelPcs map[*ast.Instr]int) (int, bool, error) {
off, ok := riscvPCRelOffset(instr)
if !ok {
return 0, false, nil
}
if pcRelPcs == nil {
return 0, true, &riscvRangeError{"pc-relative placeholder"}
}
targetPc, ok := pcRelPcs[instr]
if !ok {
return 0, true, fmt.Errorf("PC-relative target %d out of range", off)
}
return targetPc, true, nil
}
// riscvInvertedBranchEnc returns the encoding of mnem's inverted condition
// for the given operands: InvertBranch's table applied at the encoding level.
// The register operands are already in position for the inverted form.
func riscvInvertedBranchEnc(mnem string, ops []*ast.Operand) (riscvEnc, int, int, bool) {
reg := func(i int) int { return regFromOperand(ops[i]) }
switch mnem {
case "BEQ": // → BNE rs1, rs2
return riscvEnc{0x63, 0x1, 0x00}, reg(0), reg(1), true
case "BNE": // → BEQ rs1, rs2
return riscvEnc{0x63, 0x0, 0x00}, reg(0), reg(1), true
case "BLT": // → BGE rs1, rs2
return riscvEnc{0x63, 0x5, 0x00}, reg(0), reg(1), true
case "BGE": // → BLT rs1, rs2
return riscvEnc{0x63, 0x4, 0x00}, reg(0), reg(1), true
case "BLTU": // → BGEU rs1, rs2
return riscvEnc{0x63, 0x7, 0x00}, reg(0), reg(1), true
case "BGEU": // → BLTU rs1, rs2
return riscvEnc{0x63, 0x6, 0x00}, reg(0), reg(1), true
case "BEQZ": // → BNEZ rs, X0
return riscvEnc{0x63, 0x1, 0x00}, reg(0), 0, true
case "BNEZ": // → BEQZ rs, X0
return riscvEnc{0x63, 0x0, 0x00}, reg(0), 0, true
case "BLTZ": // → BGEZ rs, X0
return riscvEnc{0x63, 0x5, 0x00}, reg(0), 0, true
case "BGEZ": // → BLTZ rs, X0
return riscvEnc{0x63, 0x4, 0x00}, reg(0), 0, true
case "BLEZ": // → BGTZ: blt X0, rs
return riscvEnc{0x63, 0x4, 0x00}, 0, reg(0), true
case "BGTZ": // → BLEZ: bge X0, rs
return riscvEnc{0x63, 0x5, 0x00}, 0, reg(0), true
case "BGT": // → BLE: bge rs2, rs1
return riscvEnc{0x63, 0x5, 0x00}, reg(1), reg(0), true
case "BLE": // → BGT: blt rs2, rs1
return riscvEnc{0x63, 0x4, 0x00}, reg(1), reg(0), true
case "BGTU": // → BLEU: bgeu rs2, rs1
return riscvEnc{0x63, 0x7, 0x00}, reg(1), reg(0), true
case "BLEU": // → BGTU: bltu rs2, rs1
return riscvEnc{0x63, 0x6, 0x00}, reg(1), reg(0), true
}
return riscvEnc{}, 0, 0, false
}
// riscvRangeError reports a branch or jump displacement beyond its
// architecture limit. The layout passes tolerate it (the relaxation pass
// rewrites overlong conditional branches before the final encoding); a range
// error reaching the final pass is a real failure.
type riscvRangeError struct{ msg string }
func (e *riscvRangeError) Error() string { return e.msg }
// riscvIsRangeError reports whether err is a displacement-range rejection.
func riscvIsRangeError(err error) bool {
var re *riscvRangeError
return errors.As(err, &re)
}
// riscvRoundModes maps the rounding-mode suffixes onto their funct7 codes.
var riscvRoundModes = map[string]uint32{
"RNE": 0,
"RTZ": 1,
"RDN": 2,
"RUP": 3,
"RMM": 4,
}
// riscvCheckBranchOffset rejects a B-type displacement outside its signed
// 13-bit span [-4096, 4094]; the encoder masks to 13 bits, so an
// out-of-range offset would otherwise wrap to a wrong target.
func riscvCheckBranchOffset(target string, off int32) error {
if off < -4096 || off > 4094 {
return &riscvRangeError{fmt.Sprintf("branch to %q too far (13-bit range)", target)}
}
return nil
}
// riscvCheckJumpOffset rejects a J-type displacement outside its signed
// 21-bit span [-1048576, 1048574].
func riscvCheckJumpOffset(target string, off int32) error {
if off < -1048576 || off > 1048574 {
return &riscvRangeError{fmt.Sprintf("jump to %q too far (21-bit range)", target)}
}
return nil
}
// encodeRISCVInstr encodes a single RISC-V instruction.
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int, lits *riscvLiterals) ([]byte, error) {
mnem := instr.Mnemonic.Text
ops := instr.Operands
mnem = riscvNormalisePseudo(mnem)
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
var word uint32
// Handle pseudo-instructions and special cases first.
switch mnem {
case "RET":
2026-08-13 18:12:22 +02:00
// RET = epilogue (restore LR and close the frame when present) +
// uncompressed JALR X0, 0(X1) (the toolchain never compresses RET).
return riscvReturn(fi), nil
case "FUNCDATA":
// The assembler's bookkeeping statement, the expanded form of the
// GO_ARGS and NO_LOCAL_POINTERS macros: FUNCDATA $n, sym(SB)
// contributes no bytes, exactly as the toolchain's listing shows
// (the FUNCDATA entries and the instruction after them share a PC).
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
}
return nil, nil
case "PCDATA":
// The other bookkeeping statement, the expanded form of
// GO_RESULTS_INITIALIZED: PCDATA $n, $m contributes no bytes too.
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
return nil, fmt.Errorf("PCDATA expects $n, $m")
}
return nil, nil
case "WORD":
// WORD $w lays down a raw 32-bit little-endian word.
if len(ops) != 1 {
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
w := int64(immFromOperand(ops[0]))
if w < 0 || w > 0xFFFFFFFF {
return nil, fmt.Errorf("WORD: immediate %d does not fit a 32-bit word", w)
}
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}, nil
case "BYTE":
// BYTE $b lays down one raw byte per operand.
var out []byte
for _, op := range ops {
b := int64(immFromOperand(op))
if b < 0 || b > 0xFF {
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
}
out = append(out, byte(b))
}
return out, nil
case "CALL":
2026-08-13 18:12:22 +02:00
// CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL
// relocation. The Go assembler rejects CALL to a local branch label.
if len(ops) != 1 {
return nil, fmt.Errorf("CALL expects 1 operand, got %d", len(ops))
}
2026-08-13 18:12:22 +02:00
op := ops[0]
if op.Addr.Sym == nil || op.Addr.Sym.Pseudo != "SB" {
// CALL (X5): an indirect call, the toolchain's JALR X1, 0(X5).
if op.Addr.Sym == nil && op.Addr.Base != "" {
if op.Addr.Offset != 0 || op.Addr.Index != "" {
return nil, fmt.Errorf("CALL: invalid indirect operand %q", op.Raw)
}
rs1 := riscvRegNum(op.Addr.Base)
if rs1 < 0 {
return nil, fmt.Errorf("CALL: unknown branch register %q", op.Addr.Base)
}
word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 1, rs1, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
2026-08-13 18:12:22 +02:00
return nil, fmt.Errorf("CALL: local branch target is not supported (use CALL sym(SB))")
}
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelRISCVJal, Addend: op.Addr.Sym.Offset})
}
word = riscvJType(1, 0) // JAL X1, 0, the linker fills the offset
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JMP":
// JMP = JAL X0, target. The Go assembler never compresses this to
// C.J, so always emit the 32-bit JAL.
var target string
if len(ops) >= 1 {
// JMP sym(SB): a tail call, JAL X0 against a symbol relocation.
if ops[0].Addr.Sym != nil && ops[0].Addr.Sym.Pseudo == "SB" {
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: ops[0].Addr.Sym.Name, Kind: RelRISCVJal, Addend: ops[0].Addr.Sym.Offset})
}
word = riscvJType(0, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
target = labelFromOperand(ops[0])
// JMP N(PC): the PC-relative slot form, resolved like the
// branches (the toolchain counts source instructions at a
// uniform 4 bytes, so JMP 0(PC) is a self-loop and JMP -3(PC)
// reaches twelve bytes back). It must be recognised before the
// indirect-register form, whose operand it resembles.
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
offset := int32(off - pc)
if err := riscvCheckJumpOffset("", offset); err != nil {
return nil, err
}
word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// JMP (X5): an indirect branch, the toolchain's JALR X0, 0(X5).
if ops[0].Addr.Sym == nil && ops[0].Addr.Base != "" {
if ops[0].Addr.Offset != 0 || ops[0].Addr.Index != "" {
return nil, fmt.Errorf("JMP: invalid indirect operand %q", ops[0].Raw)
}
rs1 := riscvRegNum(ops[0].Addr.Base)
if rs1 < 0 {
return nil, fmt.Errorf("JMP: unknown branch register %q", ops[0].Addr.Base)
}
word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, rs1, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
}
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(target, offset); err != nil {
return nil, err
}
word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
case "JAL":
rd := 0
var target string
if len(ops) >= 2 {
rd = regFromOperand(ops[0])
target = labelFromOperand(ops[1])
} else if len(ops) == 1 {
target = labelFromOperand(ops[0])
}
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
targetOff := off
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(target, offset); err != nil {
return nil, err
}
word = riscvJType(rd, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(target, offset); err != nil {
return nil, err
}
word = riscvJType(rd, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
// MOV is a pseudo-instruction that the Go assembler uses for loads,
// stores, register moves and immediate loads. The width suffixes
// (MOVB/MOVH/MOVW and unsigned forms) select the access width, and
// MOVD/MOVF address the FP registers.
case "MOV", "MOVB", "MOVBU", "MOVH", "MOVHU", "MOVW", "MOVWU", "MOVF", "MOVD":
return encodeRISCVMov(instr, fi, relocs, lits)
// JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1).
case "JALR":
return encodeRISCVJALR(instr, fi)
// Branch-zero pseudos: BEQZ/BNEZ compare against X0, and BLTZ/BGEZ/
// BLEZ/BGTZ reorder the register operands of BLT/BGE accordingly.
case "BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rs := regFromOperand(ops[0])
if rs < 0 {
return nil, fmt.Errorf("%s: invalid register", mnem)
}
targetOff := 0
target := ""
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
targetOff = off
} else {
target = labelFromOperand(ops[1])
var ok bool
targetOff, ok = offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
}
var enc riscvEnc
rs1, rs2 := rs, 0
switch mnem {
case "BEQZ":
enc = riscvEnc{0x63, 0x0, 0x00} // beq rs, x0
case "BNEZ":
enc = riscvEnc{0x63, 0x1, 0x00} // bne rs, x0
case "BLTZ":
enc = riscvEnc{0x63, 0x4, 0x00} // blt rs, x0
case "BGEZ":
enc = riscvEnc{0x63, 0x5, 0x00} // bge rs, x0
case "BLEZ":
enc, rs1, rs2 = riscvEnc{0x63, 0x5, 0x00}, 0, rs // bge x0, rs
case "BGTZ":
enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, 0, rs // blt x0, rs
}
if err := riscvCheckBranchOffset(target, int32(targetOff-pc)); err != nil {
return nil, err
}
word = riscvBType(enc, rs1, rs2, int32(targetOff-pc))
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
// System instructions with no operands.
case "FENCE", "ECALL", "EBREAK", "FENCE.TSO", "PAUSE":
enc, ok := riscvInstrTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported system instruction %q", mnem)
}
// The bare FENCE expands to fence iorw, iorw: the predecessor and
// successor fields both carry 0xF in the I-type immediate
// (the toolchain's encodeFenceOperand TYPE_NONE default). FENCE.TSO
// carries the TSO fence mode with RW predecessor and successor.
imm := int32(0)
if mnem == "FENCE" {
imm = 0x0FF
}
if mnem == "FENCE.TSO" {
imm = 0x833
}
if mnem == "PAUSE" {
imm = 0x010
}
word = riscvIType(enc, 0, 0, imm)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// FP conversion / move instructions use a separate table (rs2 encodes
// the conversion type, not a register). Handle them before the main
// table lookup.
if cvtEnc, ok := riscvCvtTable[mnem]; ok {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rs1 := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
word := riscvCvtType(cvtEnc, rd, rs1)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// FP conversions with an explicit rounding mode: FCVTWS.RNE and friends
// suffix the base mnemonic with RNE/RTZ/RDN/RUP/RMM, which lands in the
// low three bits of the funct7 field.
if i := strings.IndexByte(mnem, '.'); i > 0 {
if base, ok := riscvCvtTable[mnem[:i]]; ok {
rm, ok := riscvRoundModes[mnem[i+1:]]
if !ok {
return nil, fmt.Errorf("unsupported rounding mode in %q", mnem)
}
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rs1 := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
base.funct7 = (base.funct7 &^ 7) | rm
word := riscvCvtType(base, rd, rs1)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
}
// R4-type fused multiply-add: INSTR rs1, rs2, rs3, rd (destination last).
if fmaEnc, ok := riscvFmaTable[mnem]; ok {
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
rs1 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1])
rs3 := regFromOperand(ops[2])
rd := regFromOperand(ops[3])
if rd < 0 || rs1 < 0 || rs2 < 0 || rs3 < 0 {
return nil, fmt.Errorf("invalid FP register in %s", mnem)
}
word := riscvFmaType(fmaEnc, rd, rs1, rs2, rs3)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// CSR instructions: INSTR csr, rs1|uimm, rd (destination last). The
// write-only pseudos (CSRS/CSRC/CSRW and their immediate forms) spell
// the source first, the CSR second, and read the destination as X0; the
// immediate or register variant follows the source operand's kind.
csrMnem := mnem
csrPseudo := false
csrRead := false
csrFix := int32(0)
switch mnem {
case "CSRS", "CSRW", "CSRC", "CSRSI", "CSRWI", "CSRCI":
csrMnem = map[string]string{
"CSRS": "CSRRS", "CSRW": "CSRRW", "CSRC": "CSRRC",
"CSRSI": "CSRRSI", "CSRWI": "CSRRWI", "CSRCI": "CSRRCI",
}[mnem]
csrPseudo = true
// CSRR csr, rd is CSRRS rd, csr, X0; the read pseudos RDCYCLE/RDTIME/
// RDINSTRET fix the CSR to cycle/time/instret.
case "CSRR":
csrMnem = "CSRRS"
csrPseudo = true
csrRead = true
case "RDCYCLE", "RDTIME", "RDINSTRET":
csrMnem = "CSRRS"
csrPseudo = true
csrRead = true
csrFix = map[string]int32{"RDCYCLE": 0xC00, "RDTIME": 0xC01, "RDINSTRET": 0xC02}[mnem]
}
if csrEnc, ok := riscvCsrTable[csrMnem]; ok {
if csrRead && len(ops) != 1 && len(ops) != 2 {
return nil, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops))
}
if csrPseudo && !csrRead && len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
if !csrPseudo && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
csrOp := ops[0]
srcOp := ops[0]
rdOp := ops[len(ops)-1]
switch {
case csrRead:
// CSRR csr, rd (or the fixed-CSR read pseudos with only rd).
case csrPseudo:
// src, csr.
if len(ops) > 1 {
csrOp, srcOp = ops[1], ops[0]
}
rdOp = nil
default:
// Either src, csr, rd or csr, src, rd: a CSR *name* in the
// second operand marks the toolchain's order.
srcOp = ops[1]
if op := ops[1]; op.Addr.Sym != nil {
if _, ok := riscvCSRNames[strings.ToUpper(op.Addr.Sym.Name)]; ok {
csrOp, srcOp = ops[1], ops[0]
}
}
}
csr, ok := riscvCSRAddress(csrOp)
if !ok && csrFix == 0 {
return nil, fmt.Errorf("%s: unknown CSR %q", mnem, csrOp.Raw)
}
if csrFix != 0 {
csr = csrFix
}
if csr < 0 || csr > 0xFFF {
return nil, fmt.Errorf("%s: CSR address %d out of range 0-0xFFF", mnem, csr)
}
rd := 0
if !csrPseudo {
rd = regFromOperand(rdOp) // destination register
if rd < 0 {
return nil, fmt.Errorf("invalid destination register in %s", mnem)
}
}
if csrRead {
rd = regFromOperand(rdOp)
if rd < 0 {
return nil, fmt.Errorf("invalid destination register in %s", mnem)
}
}
var src int
switch {
case csrRead:
// CSRR reads with rs1 = X0: src stays zero.
case isImmOperand(srcOp):
// Immediate variant: the source is a 5-bit unsigned immediate.
src = int(immFromOperand(srcOp))
if src < 0 || src > 31 {
return nil, fmt.Errorf("%s: uimm out of range 0-31", mnem)
}
case csrEnc.imm:
return nil, fmt.Errorf("%s expects an immediate source", mnem)
default:
// Register variant: the source is a register.
src = regFromOperand(srcOp)
if src < 0 {
return nil, fmt.Errorf("invalid source register in %s", mnem)
}
}
word := riscvCsrType(csrEnc, rd, src, csr)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// The toolchain's synthesised instructions and the RVV slice: expanded
// encodings the main table does not carry. FSGNJD is a plain table
// entry and stays with the FP arithmetic path.
if code, handled, err := encodeRISCVExtended(mnem, instr, pc, offsets); handled {
if err != nil {
return nil, err
}
return code, nil
}
enc, ok := riscvInstrTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported RISC-V instruction %q", mnem)
}
switch {
// R-type: Go reverses the ISA order, writing rs2, rs1, rd (destination
// last); the two-operand form INSTR rs2, rd uses rd as rs1.
case len(ops) == 3 && isRTypeInstr(mnem):
rs2 := regFromOperand(ops[0]) // first operand = rs2
rs1 := regFromOperand(ops[1]) // second operand = rs1
rd := regFromOperand(ops[2]) // destination (last operand)
if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvRType(enc, rd, rs1, rs2)
case len(ops) == 2 && isRTypeInstr(mnem):
rs2 := regFromOperand(ops[0]) // source (first operand)
rd := regFromOperand(ops[1]) // destination (second operand)
if rd < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvRType(enc, rd, rd, rs2)
// I-type shift (SLLI, SRLI, SRAI): INSTR $shamt, rs1, rd; the two-operand
// form INSTR $shamt, rd uses rd as the source.
case len(ops) == 3 && isShiftImmInstr(mnem):
shamt := int(immFromOperand(ops[0]))
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvRType(enc, rd, rs1, shamt)
case len(ops) == 2 && isShiftImmInstr(mnem):
shamt := int(immFromOperand(ops[0]))
rd := regFromOperand(ops[1])
if rd < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvRType(enc, rd, rd, shamt)
// AMO atomics: Plan 9 order is INSTR src, (addr), dst.
case len(ops) == 3 && isAMOInstr(mnem):
rs2 := regFromOperand(ops[0]) // source value
rs1, _ := memFromOperandWithFrame(ops[1], fi) // memory address
rd := regFromOperand(ops[2]) // destination (old value)
if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
word = riscvAMOType(enc, rd, rs1, rs2)
// FP arithmetic: Go reverses the ISA order, writing rs2, rs1, rd.
case len(ops) == 3 && isFPArithInstr(mnem):
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid FP register in %s", mnem)
}
word = riscvRType(enc, rd, rs1, rs2)
// FP arithmetic (2-operand): FSQRT src, dst.
case len(ops) == 2 && isFPArithInstr(mnem):
rs1 := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid FP register in %s", mnem)
}
word = riscvRType(enc, rd, rs1, 0)
// FP loads: INSTR addr, freg (Plan 9: source first).
case len(ops) == 2 && isFPLoadInstr(mnem):
rd := regFromOperand(ops[1])
rs1, imm := memFromOperandWithFrame(ops[0], fi)
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
word = riscvIType(enc, rd, rs1, imm)
// FP stores: INSTR freg, addr (Plan 9: source first).
case len(ops) == 2 && isFPStoreInstr(mnem):
rs2 := regFromOperand(ops[0])
rs1, imm := memFromOperandWithFrame(ops[1], fi)
if rs2 < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
word = riscvSType(enc, rs1, rs2, imm)
// LR (load-reserved): INSTR (addr), dst. The toolchain reads the
// operands positionally, so the base register comes from the first
// operand and the destination from the second whatever their parens.
case len(ops) == 2 && isLRInstr(mnem):
rs1 := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
word = riscvAMOType(enc, rd, rs1, 0) // rs2=0 for LR
// SC (store-conditional): INSTR src, (addr), dst, 3 operands.
case len(ops) == 3 && isSCInstr(mnem):
rs2 := regFromOperand(ops[0])
rs1, _ := memFromOperandWithFrame(ops[1], fi)
rd := regFromOperand(ops[2])
if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
word = riscvAMOType(enc, rd, rs1, rs2)
// FP compare: Go reverses the ISA order, writing rs2, rs1, rd.
case len(ops) == 3 && isFPCmpInstr(mnem):
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd < 0 || rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
word = riscvRType(enc, rd, rs1, rs2)
// I-type with immediate: Plan 9 order is INSTR $imm, rs1, rd; the
// two-operand form INSTR $imm, rd uses rd as the source.
case len(ops) == 3 && isITypeInstr(mnem):
imm, err := riscvImm32FromOperand(ops[0], immNeg) // immediate
if err != nil {
return nil, err
}
rs1 := regFromOperand(ops[1]) // source register
rd := regFromOperand(ops[2]) // destination
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
return encodeRISCVItypeImmediate(mnem, enc, rd, rs1, imm)
case len(ops) == 2 && isITypeInstr(mnem):
imm, err := riscvImm32FromOperand(ops[0], immNeg)
if err != nil {
return nil, err
}
rd := regFromOperand(ops[1])
if rd < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
return encodeRISCVItypeImmediate(mnem, enc, rd, rd, imm)
// Loads: rd, offset(rs1), Plan 9 order is LD src, dst.
case len(ops) == 2 && isLoadInstr(mnem):
rd := regFromOperand(ops[1]) // destination (last operand)
rs1, imm := memFromOperandWithFrame(ops[0], fi) // memory source (first operand)
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
word = riscvIType(enc, rd, rs1, imm)
// Stores: Plan 9 order is SD src, dst (src=register, dst=memory).
case len(ops) == 2 && isStoreInstr(mnem):
rs2 := regFromOperand(ops[0]) // source register (first operand)
rs1, imm := memFromOperandWithFrame(ops[1], fi) // memory dest (last operand)
if rs2 < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
word = riscvSType(enc, rs1, rs2, imm)
// Branches: rs1, rs2, label. BGT/BLE/BGTU/BLEU are the swapped-spelling
// forms of BLT/BGE/BLTU/BGEU (bgt rs1, rs2 is blt rs2, rs1).
case len(ops) == 3 && isBranchInstr(mnem):
rs1 := regFromOperand(ops[0])
rs2 := regFromOperand(ops[1])
target := labelFromOperand(ops[2])
switch mnem {
case "BGT":
enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, rs2, rs1 // blt rs2, rs1
case "BLE":
enc, rs1, rs2 = riscvEnc{0x63, 0x5, 0x00}, rs2, rs1 // bge rs2, rs1
case "BGTU":
enc, rs1, rs2 = riscvEnc{0x63, 0x6, 0x00}, rs2, rs1 // bltu rs2, rs1
case "BLEU":
enc, rs1, rs2 = riscvEnc{0x63, 0x7, 0x00}, rs2, rs1 // bgeu rs2, rs1
}
targetOff := 0
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
targetOff = off
} else {
var ok bool
targetOff, ok = offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
}
offset := int32(targetOff - pc)
if rs1 < 0 || rs2 < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
if err := riscvCheckBranchOffset(target, offset); err != nil {
return nil, err
}
// The Go assembler never compresses branches to C.BEQZ/C.BNEZ.
word = riscvBType(enc, rs1, rs2, offset)
// U-type: rd, imm (or the toolchain testdata's INSTR $imm, rd).
case len(ops) == 2 && isUTypeInstr(mnem):
var rd int
var imm int32
if isImmOperand(ops[0]) {
imm, rd = immFromOperand(ops[0]), regFromOperand(ops[1])
} else {
rd, imm = regFromOperand(ops[0]), immFromOperand(ops[1])
}
if rd < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem)
}
word = riscvUType(enc, rd, imm)
default:
return nil, fmt.Errorf("cannot encode %s with %d operands", mnem, len(ops))
}
// Emit as little-endian 32-bit word.
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// isMemOperand reports whether an operand is a memory reference
// (frame-relative such as name+off(FP) or register-relative such as (X10)).
func isMemOperand(op *ast.Operand) bool {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
return true // name+off(FP), name+off(SP)
}
if op.Addr.Base != "" && op.Addr.Sym == nil {
return true // (reg)
}
return false
}
// isImmOperand reports whether an operand is an immediate ($value).
func isImmOperand(op *ast.Operand) bool {
if op.Kind == ast.OpImmediate {
return true
}
if op.Imm.HasVal {
return true
}
return false
}
// encodeRISCVMov encodes the MOV pseudo-instruction.
//
// The Go RISC-V assembler uses MOV for:
// - MOV name+off(FP), Rd load from frame
// - MOV Rd, name+off(FP) store to frame
// - MOV (Rs), Rd register-relative load
// - MOV Rs, (Rd) register-relative store
// - MOV Rs, Rd register-to-register move (ADDI $0)
// - MOV $imm, Rd load immediate (ADDI or LUI+ADDIW)
func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc, lits *riscvLiterals) ([]byte, error) {
ops := instr.Operands
if len(ops) != 2 {
return nil, fmt.Errorf("MOV expects 2 operands, got %d", len(ops))
}
src := ops[0]
dst := ops[1]
// Immediate → register.
if isImmOperand(src) {
// MOV $sym(SB), rd, load address of a static symbol or external.
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
rd := regFromOperand(dst)
if rd < 0 {
return nil, fmt.Errorf("MOV $sym(SB): invalid destination register")
}
return encodeRISCVSBAddr(src.Imm.Sym, rd, relocs), nil
}
// MOV $sym+off(FP|SP), rd: the address of a frame slot as an
// immediate is the frame-adjusted offset against the hardware SP,
// the toolchain's ADDI $adj, SP, rd (argframe+0(FP) in the runtime's
// reflect trampolines is the spelling).
if src.Imm.Sym != nil && (src.Imm.Sym.Pseudo == "FP" || src.Imm.Sym.Pseudo == "SP") {
rd := regFromOperand(dst)
if rd < 0 {
return nil, fmt.Errorf("MOV $%s(%s): invalid destination register", src.Imm.Sym.Name, src.Imm.Sym.Pseudo)
}
_, off := riscvResolvePseudo(src.Imm.Sym, fi)
return riscvSPAddiBytes(rd, off), nil
}
// MOV $sym(FP/SP), rd, not supported: immediate symbol references
// other than the frame pseudos cannot be encoded as a simple
// immediate.
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" {
return nil, fmt.Errorf("MOV $%s(%s): unsupported immediate symbol reference (only SB is supported)", src.Imm.Sym.Name, src.Imm.Sym.Pseudo)
}
rd := regFromOperand(dst)
if rd < 0 {
return nil, fmt.Errorf("MOV $imm: invalid destination register")
}
imm := riscvOperandImm64(src)
if int64(int32(imm)) != imm {
// Beyond the signed 32-bit span the toolchain either builds the
// value from a shifted 32-bit part or loads it from the pooled
// $i64 constant it synthesises for the purpose.
return riscvLoadImm64(rd, imm, lits, relocs), nil
}
return encodeRISCVLoadImm(rd, int32(imm)), nil
}
// Memory → register (load).
if isMemOperand(src) && !isMemOperand(dst) {
rd := regFromOperand(dst)
// MOV sym(SB), rd, load from static data.
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" {
if rd < 0 {
return nil, fmt.Errorf("MOV sym(SB): invalid destination register")
}
return encodeRISCVSBLoad(src.Addr.Sym, rd, relocs), nil
}
rs1, off := memFromOperandWithFrame(src, fi)
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("MOV load: invalid operand")
}
return riscvFrameMemOp(riscvMovEnc(strings.ToUpper(instr.Mnemonic.Text), false), false, rd, rs1, off), nil
}
// Register → memory (store).
if !isMemOperand(src) && isMemOperand(dst) {
rs2 := regFromOperand(src)
// MOV rd, sym(SB), store to static data.
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" {
if rs2 < 0 {
return nil, fmt.Errorf("MOV rd, sym(SB): invalid source register")
}
return encodeRISCVSBStore(dst.Addr.Sym, rs2, relocs), nil
}
rs1, off := memFromOperandWithFrame(dst, fi)
if rs2 < 0 || rs1 < 0 {
return nil, fmt.Errorf("MOV store: invalid operand")
}
return riscvFrameMemOp(riscvMovEnc(strings.ToUpper(instr.Mnemonic.Text), true), true, rs2, rs1, off), nil
}
// Register → register: MOVD/MOVF are FP moves (fsgnj with rs2 = rs1),
// everything else is ADDI $0, src, dst.
{
rs1 := regFromOperand(src)
rd := regFromOperand(dst)
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("MOV: invalid register operand")
}
mnem := strings.ToUpper(instr.Mnemonic.Text)
if mnem == "MOVD" || mnem == "MOVF" {
op := uint32(0x20000053) // FSGNJ.S
if mnem == "MOVD" {
op = 0x22000053 // FSGNJ.D
}
return wordLE(op | uint32(rs1)<<15 | uint32(rs1)<<20 | uint32(rd)<<7), nil
}
word := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs1, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
}
// riscvMovEnc returns the load (store=false) or store (store=true) opcode for
// a MOV-family mnemonic: the suffix selects the access width, MOVD and MOVF
// select the FP load/store opcodes, and bare MOV is the 64-bit integer form.
func riscvMovEnc(mnem string, store bool) riscvEnc {
if store {
switch mnem {
case "MOVB":
return riscvEnc{0x23, 0x0, 0x00} // SB
case "MOVH":
return riscvEnc{0x23, 0x1, 0x00} // SH
case "MOVW":
return riscvEnc{0x23, 0x2, 0x00} // SW
case "MOVF":
return riscvEnc{0x27, 0x2, 0x00} // FSW
case "MOVD":
return riscvEnc{0x27, 0x3, 0x00} // FSD
}
return riscvEnc{0x23, 0x3, 0x00} // SD
}
switch mnem {
case "MOVB":
return riscvEnc{0x03, 0x0, 0x00} // LB
case "MOVBU":
return riscvEnc{0x03, 0x4, 0x00} // LBU
case "MOVH":
return riscvEnc{0x03, 0x1, 0x00} // LH
case "MOVHU":
return riscvEnc{0x03, 0x5, 0x00} // LHU
case "MOVW":
return riscvEnc{0x03, 0x2, 0x00} // LW
case "MOVWU":
return riscvEnc{0x03, 0x6, 0x00} // LWU
case "MOVF":
return riscvEnc{0x07, 0x2, 0x00} // FLW
case "MOVD":
return riscvEnc{0x07, 0x3, 0x00} // FLD
}
return riscvEnc{0x03, 0x3, 0x00} // LD
}
// riscvFrameMemOp encodes a register-relative load (store=false, I-type
// width 0x03) or store (store=true, S-type width 0x23) of the 64-bit width
// at off(rs1). Offsets beyond the signed 12-bit range materialise the
// address in X31 first: LUI hi (the rounding split), then ADD X31, rs1,
// matching the toolchain's large-frame addressing; the access uses the
// sign-extended low part, which always fits.
func riscvFrameMemOp(enc riscvEnc, store bool, reg, rs1 int, off int32) []byte {
if fits12(off) {
var word uint32
if store {
word = riscvSType(enc, rs1, reg, off)
} else {
word = riscvIType(enc, reg, rs1, off)
}
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
}
lo := off - (splitHi(off) << 12)
out := riscvAddressInX31WithBase(off, rs1)
var word uint32
if store {
word = riscvSType(enc, 31, reg, lo)
} else {
word = riscvIType(enc, reg, 31, lo)
}
return append(out, wordLE(word)...)
}
// riscvFrameMemSize returns the encoded size of a frame-relative MOV for the
// layout pass: 4 bytes when the offset fits, otherwise the X31
// materialisation plus the access.
func riscvFrameMemSize(op *ast.Operand, fi riscvFrameInfo) int {
rs1, off := memFromOperandWithFrame(op, fi)
if fits12(off) {
return 4
}
return len(riscvAddressInX31WithBase(off, rs1)) + 4
}
2026-08-13 17:41:16 +02:00
// encodeRISCVLoadImm encodes loading an immediate into a register (MOV $imm,
// rd), matching the toolchain's instructionsForMOVConst. For 12-bit
// immediates it emits ADDI $imm, ZERO, rd (compressed to C.LI when it fits
// six signed bits); for larger immediates it emits LUI + [ADDIW], with the LUI
// and ADDIW compressed to C.LUI / C.ADDIW when their immediate fits.
func encodeRISCVLoadImm(rd int, imm int32) []byte {
if imm >= -2048 && imm <= 2047 {
2026-08-13 17:41:16 +02:00
if rd != 0 && imm >= -32 && imm <= 31 {
return word16(rvcCI(0x2, uint32(rd), uint32(imm)&0x3F)) // C.LI
}
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, 0, imm))
}
2026-08-13 17:41:16 +02:00
low, high := splitRISCV32Imm(imm)
var out []byte
2026-08-13 17:41:16 +02:00
if rd != 0 && rd != 2 && high >= -32 && high <= 31 {
out = append(out, word16(rvcCI(0x3, uint32(rd), uint32(high)&0x3F))...) // C.LUI
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, rd, high<<12))...)
}
if low != 0 {
if low >= -32 && low <= 31 {
out = append(out, word16(rvcCI(0x1, uint32(rd), uint32(low)&0x3F))...) // C.ADDIW
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1B, 0x0, 0x00}, rd, rd, low))...)
}
}
return out
}
2026-08-13 17:41:16 +02:00
// riscvMovImmSize returns the encoded byte length of MOV $imm, rd, mirroring
// encodeRISCVLoadImm's expansion and compression.
func riscvMovImmSize(rd int, imm int32) int {
if imm >= -2048 && imm <= 2047 {
if rd != 0 && imm >= -32 && imm <= 31 {
return 2 // C.LI
}
return 4 // ADDI
}
low, high := splitRISCV32Imm(imm)
size := 0
if rd != 0 && rd != 2 && high >= -32 && high <= 31 {
size += 2 // C.LUI
} else {
size += 4 // LUI
}
if low != 0 {
if low >= -32 && low <= 31 {
size += 2 // C.ADDIW
} else {
size += 4 // ADDIW
}
}
return size
}
// splitRISCV32Imm splits a signed 32-bit immediate into a signed 12-bit low
// part and a signed 20-bit high part, mirroring cmd/internal/obj/riscv's
// Split32BitImmediate. The high part is returned unshifted; callers place it
// in the upper bits of LUI (or its compressed C.LUI form).
func splitRISCV32Imm(imm int32) (low, high int32) {
if imm >= -2048 && imm <= 2047 {
return imm, 0
}
h := int64(imm) >> 12
if imm&(1<<11) != 0 {
h++
}
low = int32((int64(imm) << 52) >> 52) // sign extend 12 bits
high = int32((h << 44) >> 44) // sign extend 20 bits
return low, high
}
// riscvNormalisePseudo rewrites the toolchain's UNDEF spelling onto EBREAK:
// the assembler accepts UNDEF where the hardware wants the trap instruction
// and emits ebreak (compressed to C.EBREAK under RVC), so every pass sees the
// canonical name.
func riscvNormalisePseudo(mnem string) string {
if strings.EqualFold(mnem, "UNDEF") {
return "EBREAK"
}
return mnem
}
// riscvOperandImm64 reads an immediate operand as a full signed 64-bit value,
// where immFromOperand would truncate to int32; the MOV immediate path uses
// it to classify the wide constants.
func riscvOperandImm64(op *ast.Operand) int64 {
if !op.Imm.HasVal {
return 0
}
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return v
}
// riscvSplitShiftConst mirrors cmd/internal/obj/riscv's splitShiftConst: it
// looks for the signed 32-bit integer a constant can be rebuilt from with a
// left shift, a left-and-right shift pair (a run of ones), or a zero-extended
// 32-bit pattern. A constant that fits none of the shapes is materialised
// from the pooled $i64 data symbol instead.
func riscvSplitShiftConst(v int64) (imm int64, lsh int, rsh int, ok bool) {
// Rebuild from a signed 32-bit integer shifted left.
lsh = bits.TrailingZeros64(uint64(v))
c := v >> lsh
if int64(int32(c)) == c {
return c, lsh, 0, true
}
// Rebuild from a small negative constant: shift left into place, then
// shift the sign-extended ones run right.
rsh = bits.LeadingZeros64(uint64(v))
ones := bits.OnesCount64((uint64(v) >> lsh) >> 11)
if rsh+ones+lsh+11 == 64 {
c = (1<<11 | ((v >> lsh) & 0x7ff)) << 52 >> 52 // sign extend 12 bits
if lsh > 0 || c != -1 {
lsh += rsh
}
return c, lsh, rsh, true
}
// Rebuild from a zero-extended signed 32-bit integer.
if int64(uint32(c)) == c {
c = int64(int32(c))
lsh, rsh = 32, 32-lsh
return c, lsh, rsh, true
}
return 0, 0, 0, false
}
// riscvSPAddiBytes encodes ADDI rd, SP, imm for the frame-address immediates
// (the MOV $sym+off(FP|SP) form), using the compressed forms the toolchain
// picks under RVC: C.ADDI4SPN for a positive 4-byte multiple that fits, C.MV
// for the zero offset, the plain ADDI otherwise.
func riscvSPAddiBytes(rd int, imm int32) []byte {
if rd != 0 && imm == 0 {
return word16(rvcCR(0x8, uint32(rd), 2)) // C.MV rd, SP
}
if isRVCIntReg(rd) && imm > 0 && imm < 1024 && imm%4 == 0 {
return word16(rvcCIW(0x0, rvcReg3(rd), uint32(imm)))
}
return wordLE(riscvIType(riscvInstrTable["ADDI"], rd, 2, imm))
}
// riscvMovImm64Size returns the encoded byte length of MOV $imm, rd when the
// immediate sits outside the signed 32-bit span: the shifted-part sequences
// of riscvLoadImm64, or the 8-byte AUIPC+LD pool load.
func riscvMovImm64Size(rd int, imm int64) int {
c, lsh, rsh, ok := riscvSplitShiftConst(imm)
if !ok {
return 8 // AUIPC + LD against the $i64 pool symbol
}
size := riscvMovImmSize(rd, int32(c))
if lsh > 0 {
size += riscvShiftImmSize(rd, true)
}
if rsh > 0 {
size += riscvShiftImmSize(rd, false)
}
return size
}
// riscvShiftImmSize returns the encoded size of one SLLI/SRLI expansion
// part: two bytes under RVC when the destination can carry a compressed
// shift (C.SLLI admits every register but X0, C.SRLI only X8 to X15), four
// otherwise.
func riscvShiftImmSize(rd int, left bool) int {
if rd != 0 && (left || isRVCIntReg(rd)) {
return 2
}
return 4
}
// riscvLoadImm64 encodes MOV $imm, rd for an immediate beyond the signed
// 32-bit span, mirroring the toolchain's instructionsForMOVConst: when a
// shifted 32-bit part rebuilds the value it emits that part (compressed like
// any written MOV) followed by the SLLI and SRLI shifts; otherwise it loads
// the constant from the pooled read-only $i64.<hex> symbol via AUIPC + LD
// and registers the literal so the data section carries its bytes.
func riscvLoadImm64(rd int, imm int64, lits *riscvLiterals, relocs *[]Reloc) []byte {
c, lsh, rsh, ok := riscvSplitShiftConst(imm)
if !ok {
name := fmt.Sprintf("$i64.%016x", uint64(imm))
if lits != nil {
lits.add(name, riscvLiteralBytes(imm))
}
return encodeRISCVSBLoad(&ast.Symbol{Name: name}, rd, relocs)
}
out := encodeRISCVLoadImm(rd, int32(c))
if lsh > 0 {
out = append(out, riscvShiftImmBytes(rd, lsh, true)...)
}
if rsh > 0 {
out = append(out, riscvShiftImmBytes(rd, rsh, false)...)
}
return out
}
// riscvShiftImmBytes encodes one SLLI (left) or SRLI expansion part, using
// the compressed form the toolchain picks under RVC: C.SLLI admits every
// register but X0, C.SRLI only X8 to X15.
func riscvShiftImmBytes(rd, shamt int, left bool) []byte {
if rd != 0 && shamt >= 1 && shamt <= 63 && (left || isRVCIntReg(rd)) {
if left {
return word16(rvcSLLI(uint32(rd), uint32(shamt)&0x3F))
}
return word16(rvcCBShift(0x0, rvcReg3(rd), uint32(shamt)&0x3F))
}
enc := riscvEnc{0x13, 0x1, 0x00} // SLLI
imm := int32(shamt)
if !left {
enc = riscvEnc{0x13, 0x5, 0x00} // SRLI: funct6 000000, funct3 101
}
return wordLE(riscvIType(enc, rd, rd, imm))
}
// riscvLiteralBytes renders a 64-bit constant as the little-endian bytes the
// $i64 pool symbol holds.
func riscvLiteralBytes(v int64) []byte {
return []byte{byte(v), byte(v >> 8), byte(v >> 16), byte(v >> 24),
byte(v >> 32), byte(v >> 40), byte(v >> 48), byte(v >> 56)}
}
// RiscvLiteral is one pooled 64-bit constant: a MOV whose immediate sits
// beyond both the 32-bit span and the shift sequences loads its bits from a
// read-only data symbol named like the toolchain's $i64 pool.
type RiscvLiteral struct {
Name string
Data []byte
}
// riscvLiterals collects the pooled constants the MOV expansions refer to,
// deduplicated by name, in first-use order.
type riscvLiterals struct {
order []RiscvLiteral
seen map[string]bool
}
func (l *riscvLiterals) add(name string, data []byte) {
if l.seen == nil {
l.seen = map[string]bool{}
}
if !l.seen[name] {
l.seen[name] = true
l.order = append(l.order, RiscvLiteral{Name: name, Data: data})
}
}
func (l *riscvLiterals) list() []RiscvLiteral { return l.order }
// encodeRISCVItypeImmediate encodes an I-type arithmetic instruction, expanding
// large immediates for ADDI/ANDI/ORI/XORI into LUI+ADDIW+op (or two ADDIs for
// ADDI), matching the Go assembler.
func encodeRISCVItypeImmediate(mnem string, enc riscvEnc, rd, rs1 int, imm int32) ([]byte, error) {
if imm >= -2048 && imm <= 2047 {
return wordLE(riscvIType(enc, rd, rs1, imm)), nil
}
var opMn string
switch mnem {
case "ADDI":
opMn = "ADD"
case "ANDI":
opMn = "AND"
case "ORI":
opMn = "OR"
case "XORI":
opMn = "XOR"
default:
return nil, fmt.Errorf("%s: immediate %d does not fit 12 bits", mnem, imm)
}
// ADDI with a small-ish immediate splits into two ADDIs.
if mnem == "ADDI" && imm >= -4096 && imm < 4095 {
imm0 := imm / 2
imm1 := imm - imm0
var out []byte
out = append(out, wordLE(riscvIType(enc, rd, rs1, imm0))...)
out = append(out, wordLE(riscvIType(enc, rd, rd, imm1))...)
return out, nil
}
// LUI $high, TMP; [ADDIW $low, TMP, TMP]; op TMP, rs1, rd. The LUI and
// ADDIW compress to their RVC forms (C.LUI / C.ADDIW) when the immediate
// fits 6 signed bits, matching the toolchain's compress pass.
low, high := splitRISCV32Imm(imm)
tmp := 31 // X31 = T6 = TMP
var out []byte
if high != 0 && high >= -32 && high <= 31 {
out = append(out, word16(rvcCI(0x3, uint32(tmp), uint32(high)&0x3F))...)
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, tmp, high<<12))...)
}
if low != 0 {
if low >= -32 && low <= 31 {
out = append(out, word16(rvcCI(0x1, uint32(tmp), uint32(low)&0x3F))...)
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1B, 0x0, 0x00}, tmp, tmp, low))...)
}
}
opEnc, ok := riscvInstrTable[opMn]
if !ok {
return nil, fmt.Errorf("%s: unsupported operation %q", mnem, opMn)
}
out = append(out, wordLE(riscvRType(opEnc, rd, rs1, tmp))...)
return out, nil
}
// riscvItypeImmediateSize returns the encoded byte length of an I-type
// immediate instruction, accounting for the large-immediate expansion.
func riscvItypeImmediateSize(mnem string, imm int32) int {
if imm >= -2048 && imm <= 2047 {
return 4
}
switch mnem {
case "ADDI", "ANDI", "ORI", "XORI":
default:
return 4
}
if mnem == "ADDI" && imm >= -4096 && imm < 4095 {
return 8
}
low, high := splitRISCV32Imm(imm)
size := 4 // the R-type op (TMP is X31, never compressed)
if high != 0 && high >= -32 && high <= 31 {
size += 2 // C.LUI
} else {
size += 4 // LUI
}
if low != 0 {
if low >= -32 && low <= 31 {
size += 2 // C.ADDIW
} else {
size += 4 // ADDIW
}
}
return size
}
// encodeRISCVSBAddr emits AUIPC + ADDI to load the address of a static
// symbol into rd, recording the single R_RISCV_PCREL_ITYPE relocation the Go
// toolchain uses for the pair (the object-file emitters expand or map it).
func encodeRISCVSBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
name := sym.Name
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType, Addend: sym.Offset})
}
auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0)
addi := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rd, 0)
return append(wordLE(auipc), wordLE(addi)...)
}
// encodeRISCVSBLoad emits AUIPC + LD to load from a static symbol into rd,
// recording the single R_RISCV_PCREL_ITYPE relocation for the pair.
func encodeRISCVSBLoad(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
name := sym.Name
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType, Addend: sym.Offset})
}
auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0)
ld := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rd, 0)
return append(wordLE(auipc), wordLE(ld)...)
}
// encodeRISCVSBStore emits AUIPC + SD to store a register into a static symbol,
// recording the single R_RISCV_PCREL_STYPE relocation for the pair.
func encodeRISCVSBStore(sym *ast.Symbol, rs2 int, relocs *[]Reloc) []byte {
tmp := 31 // X31 = T6
name := sym.Name
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELSType, Addend: sym.Offset})
}
auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, tmp, 0)
sd := riscvSType(riscvEnc{0x23, 0x3, 0x00}, tmp, rs2, 0)
var out []byte
out = append(out, wordLE(auipc)...)
out = append(out, wordLE(sd)...)
return out
}
// wordLE encodes a uint32 as 4 little-endian bytes.
func wordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// word16 encodes a uint16 as 2 little-endian bytes.
func word16(w uint16) []byte {
return []byte{byte(w), byte(w >> 8)}
}
// encodeRISCVJALR encodes the JALR indirect jump/call instruction.
// Plan 9: JALR rs1, rd (2 regs), JALR rd, offset(rs1) (the trampoline
// form), or JALR offset(rs1) (memory → rd=X1).
func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) {
ops := instr.Operands
// JALR rd, offset(rs1): the memory operand's base is the jump-target
// register, not the destination.
if len(ops) == 2 && isMemOperand(ops[1]) {
rd := regFromOperand(ops[0])
rs1, imm := memFromOperandWithFrame(ops[1], fi)
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("JALR: invalid register operand")
}
return wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, rd, rs1, imm)), nil
}
if len(ops) == 2 {
rs1 := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("JALR: invalid register operand")
}
return wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, rd, rs1, 0)), nil
}
if len(ops) == 1 {
rs1, imm := memFromOperandWithFrame(ops[0], fi)
if rs1 < 0 {
return nil, fmt.Errorf("JALR: invalid memory operand")
}
return wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 1, rs1, imm)), nil
}
return nil, fmt.Errorf("JALR expects 1 or 2 operands, got %d", len(ops))
}
// tryCompressRVC attempts to compress a RISC-V instruction to its 16-bit
// RVC form. It returns the compressed instruction word and true on success.
func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
mnem := riscvCompressMnem(instr)
mnem = riscvNormalisePseudo(mnem)
ops := instr.Operands
// The immediate aliases fold onto their I-type mnemonics before
// compression: the toolchain compresses ADD $imm, rd as c.addi, exactly
// as it compresses the spelling ADDI.
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
switch mnem {
case "LD", "MOV":
// LD rd, offset(SP) → C.LDSP when rd≠0 and uimm[8:3] fits.
// MOV name+off(FP), rd → load, same compression.
if mnem == "MOV" && len(ops) == 2 && isImmOperand(ops[0]) {
return 0, false
}
// MOV reg, reg → C.MV (CR-type: funct4=0x8).
if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && !isMemOperand(ops[1]) && !isImmOperand(ops[0]) {
rs1 := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rs1 != -1 && rd != -1 && rs1 != 0 && rd != 0 {
return rvcCR(0x8, uint32(rd), uint32(rs1)), true
}
}
rd, rs1, imm := extractLDParams(instr, fi)
if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcLSP(0x3, uint32(rd), uint32(imm)), true
}
// Register-relative C.LD: both in prime regs, 8-byte scaled offset.
if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCL(0x3, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true
}
// MOV reg, mem → store, try C.SDSP.
if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && isMemOperand(ops[1]) {
rs2, rs1, imm := extractSDParams(instr, fi)
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcSSP(0x7, uint32(rs2), uint32(imm)), true
}
}
case "SD":
// SD rs2, offset(SP) → C.SDSP when uimm[8:3] fits (CSS-type).
rs2, rs1, imm := extractSDParams(instr, fi)
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcSSP(0x7, uint32(rs2), uint32(imm)), true
}
// Register-relative C.SD: base and source in prime regs.
if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCS(0x7, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true
}
case "LW":
rd, rs1, imm := extractLDParams(instr, fi)
if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 256 && imm%4 == 0 {
return rvcLSP(0x2, uint32(rd), uint32(imm)), true
}
if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 128 && imm%4 == 0 {
return rvcCL(0x2, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true
}
case "SW":
rs2, rs1, imm := extractSDParams(instr, fi)
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 256 && imm%4 == 0 {
return rvcSSP(0x6, uint32(rs2), uint32(imm)), true
}
if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 128 && imm%4 == 0 {
return rvcCS(0x6, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true
}
case "ADDI":
rd, rs1, imm := extractITypeParams(instr)
if immNeg {
imm = -imm
}
if rd == -1 || rs1 == -1 {
return 0, false
}
if rd == 2 && rs1 == 2 && imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
// C.ADDI16SP: ADDI to SP by a nonzero 16-byte multiple.
return rvcADDI16SP(2, imm), true
}
if rd == rs1 && rd != 0 && imm != 0 && imm >= -32 && imm <= 31 {
// C.ADDI: funct3=0x0, rs1/rd, nzimm[5:0]
return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true
}
if isRVCIntReg(rd) && rs1 == 2 && imm != 0 && imm >= 0 && imm < 1024 && imm%4 == 0 {
// C.ADDI4SPN: ADDI $imm, SP, rd for a prime rd.
return rvcCIW(0x0, rvcReg3(rd), uint32(imm)), true
}
if rs1 == 0 && rd != 0 && imm >= -32 && imm <= 31 {
// C.LI: funct3=0x2, rd, imm[5:0]
return rvcCI(0x2, uint32(rd), uint32(imm)&0x3F), true
}
if rs1 != 0 && rd != 0 && imm == 0 {
// C.MV: funct4=0x8, rd, rs1 (CR-type)
return rvcCR(0x8, uint32(rd), uint32(rs1)), true
}
if rd == 0 && rs1 == 0 && imm == 0 {
// C.NOP
return 0x0001, true
}
case "JAL":
// JAL/JMP are never compressed to C.J by the Go assembler.
return 0, false
case "JMP":
// JAL/JMP are never compressed to C.J by the Go assembler.
return 0, false
case "BEQ":
// Branches are never compressed to C.BEQZ/C.BNEZ.
return 0, false
case "BNE":
// Branches are never compressed to C.BEQZ/C.BNEZ.
return 0, false
case "ADD":
// ADD rs2, rs1, rd → C.ADD (CR-type, funct4=0x9) when rd == rs1; ADD
// is commutative, so if rd == rs2, swap. ADD rs2, X0, rd is C.MV.
if len(ops) == 3 {
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 {
if rd == rs1 && rs2 != 0 {
return rvcCR(0x9, uint32(rd), uint32(rs2)), true
}
if rd == rs2 && rs1 != 0 {
return rvcCR(0x9, uint32(rd), uint32(rs1)), true
}
if rs1 == 0 && rs2 != 0 {
// ADD rs2, X0, rd → C.MV rd, rs2.
return rvcCR(0x8, uint32(rd), uint32(rs2)), true
}
}
}
case "SUB", "XOR", "OR", "AND":
// C.SUB (0x23,0), C.XOR (0x23,1), C.OR (0x23,2), C.AND (0x23,3)
if len(ops) == 3 {
var funct2 uint32
switch mnem {
case "SUB":
funct2 = 0x0
case "XOR":
funct2 = 0x1
case "OR":
funct2 = 0x2
case "AND":
funct2 = 0x3
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 {
if rd == rs1 && isRVCIntReg(rd) && isRVCIntReg(rs2) && rs2 != 0 {
return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs2)), true
}
// AND/OR/XOR are commutative; SUB is not.
if mnem != "SUB" && rd == rs2 && isRVCIntReg(rd) && isRVCIntReg(rs1) && rs1 != 0 {
return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs1)), true
}
}
}
case "ADDW", "SUBW":
// C.ADDW (0x27,1) / C.SUBW (0x27,0), CA-type, prime regs.
if len(ops) == 3 {
funct2 := uint32(0x0)
if mnem == "ADDW" {
funct2 = 0x1
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := regFromOperand(ops[2])
if rd != -1 && rs1 != -1 && rs2 != -1 && isRVCIntReg(rd) {
if rd == rs1 && isRVCIntReg(rs2) {
return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs2)), true
}
// ADDW is commutative; SUBW is not.
if mnem == "ADDW" && isRVCIntReg(rs1) && rd == rs2 {
return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs1)), true
}
}
}
case "FLD":
// FLD rd, imm(SP) → C.FLDSP (CI-type, funct3=0x1).
rd, rs1, imm := extractLDParams(instr, fi)
if rs1 == 2 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcLSP(0x1, uint32(rd), uint32(imm)), true
}
// Register-relative C.FLD: rd in F8-F15, base in X8-X15.
if rs1 != -1 && rd != -1 && rd >= 8 && rd <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCL(0x1, uint32(rd-8), rvcReg3(rs1), uint32(imm)), true
}
case "FSD":
// FSD rs2, imm(SP) → C.FSDSP (CSS-type, funct3=0x5).
rs2, rs1, imm := extractSDParams(instr, fi)
if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 {
return rvcSSP(0x5, uint32(rs2), uint32(imm)), true
}
// Register-relative C.FSD: source in F8-F15, base in X8-X15.
if rs1 != -1 && rs2 != -1 && rs2 >= 8 && rs2 <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 {
return rvcCS(0x5, uint32(rs2-8), rvcReg3(rs1), uint32(imm)), true
}
case "LUI":
// LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in six
// signed bits (matching the toolchain's compress pass).
if len(ops) == 2 {
rd := regFromOperand(ops[0])
imm := immFromOperand(ops[1])
if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= -32 && imm <= 31 {
return rvcCI(0x3, uint32(rd), uint32(imm)&0x3F), true
}
}
case "ADDIW":
rd, rs1, imm := extractITypeParams(instr)
if rd == rs1 && rd != 0 && imm >= -32 && imm <= 31 {
return rvcCI(0x1, uint32(rd), uint32(imm)&0x3F), true
}
case "SLLI", "SRLI", "SRAI":
rd, rs1, imm := extractITypeParams(instr)
if rd == rs1 && rd != 0 && imm != 0 && imm >= 1 && imm <= 63 {
if mnem == "SLLI" {
// C.SLLI: funct3=0, op=10 quadrant, shamt in bits [12|6:2].
return rvcSLLI(uint32(rd), uint32(imm)&0x3F), true
}
if isRVCIntReg(rd) {
funct2 := uint32(0x0)
if mnem == "SRAI" {
funct2 = 0x1
}
// C.SRLI/C.SRAI: CB-type, funct3=0x4.
return rvcCBShift(funct2, rvcReg3(rd), uint32(imm)&0x3F), true
}
}
case "ANDI":
rd, rs1, imm := extractITypeParams(instr)
if isRVCIntReg(rd) && rd == rs1 && imm >= -32 && imm <= 31 {
// C.ANDI: CB-type, funct3=0x4, funct2=0x2.
return rvcCBShift(0x2, rvcReg3(rd), uint32(imm)&0x3F), true
}
case "EBREAK":
// C.EBREAK: CR-type, funct4=0x9, rd=0, rs2=0.
return rvcCR(0x9, 0, 0), true
}
return 0, false
}
// riscvCompressMnem maps a MOV-family load or store onto the base mnemonic
// the toolchain lowers it to (MOVW 4(SP), X9 is LW under another name), so
// the width spellings compress exactly like their base forms. Register and
// immediate forms keep their own mnemonic: the C.MV path matches "MOV" and
// nothing else in the switch has a width case.
func riscvCompressMnem(instr *ast.Instr) string {
mnem := instr.Mnemonic.Text
ops := instr.Operands
if !strings.HasPrefix(mnem, "MOV") || len(ops) != 2 {
return mnem
}
load := isMemOperand(ops[0]) && !isMemOperand(ops[1])
store := !isMemOperand(ops[0]) && isMemOperand(ops[1])
if !load && !store {
return mnem
}
switch mnem {
case "MOVW":
if load {
return "LW"
}
return "SW"
case "MOVF":
if load {
return "FLW"
}
return "FSW"
case "MOVD":
if load {
return "FLD"
}
return "FSD"
case "MOV":
if load {
return "LD"
}
return "SD"
}
// MOVB/MOVBU/MOVH/MOVHU/MOVWU have no compressed form; their base
// mnemonics (LB/LBU/LH/LHU/LWU, SB/SH) match no case either.
return mnem
}
// extractLDParams extracts rd, rs1, and immediate offset for a load instruction.
func extractLDParams(instr *ast.Instr, fi riscvFrameInfo) (rd, rs1 int, imm int32) {
ops := instr.Operands
if len(ops) != 2 {
return -1, -1, 0
}
if instr.Mnemonic.Text == "MOV" {
if isMemOperand(ops[0]) {
rs1, imm = memFromOperandWithFrame(ops[0], fi)
rd = regFromOperand(ops[1])
} else {
return -1, -1, 0
}
} else {
rs1, imm = memFromOperandWithFrame(ops[0], fi)
rd = regFromOperand(ops[1])
}
return
}
// extractSDParams extracts rs2, rs1, and immediate offset for a store instruction.
func extractSDParams(instr *ast.Instr, fi riscvFrameInfo) (rs2, rs1 int, imm int32) {
ops := instr.Operands
if len(ops) != 2 {
return -1, -1, 0
}
rs2 = regFromOperand(ops[0])
rs1, imm = memFromOperandWithFrame(ops[1], fi)
return
}
// extractITypeParams extracts rd, rs1, and immediate for an I-type
// instruction. The Plan 9 order is INSTR $imm, rs1, rd (3 operands) or
// INSTR $imm, rd (2 operands, rd is also the source).
func extractITypeParams(instr *ast.Instr) (rd, rs1 int, imm int32) {
ops := instr.Operands
switch len(ops) {
case 3:
imm = immFromOperand(ops[0])
rs1 = regFromOperand(ops[1])
rd = regFromOperand(ops[2])
case 2:
imm = immFromOperand(ops[0])
rd = regFromOperand(ops[1])
rs1 = rd
default:
return -1, -1, 0
}
return
}
// ---- toolchain-synthesised instructions and the RVV slice ----
// encodeRISCVExtended encodes the instructions the Go toolchain synthesises
// from other instructions (ANDN/ORN, MIN/MAX, ROR and friends, the branch
// pseudos and FABSD), the CSR read RDTIME, and the RVV vector slice the
// compiler's kernels use. handled reports whether the mnemonic belongs to
// this group; err carries the diagnostic when it does but cannot be encoded.
// Each expansion reproduces the toolchain's instruction-for-instruction
// sequence, including its use of X31 (TMP) and its RVC compression.
func encodeRISCVExtended(mnem string, instr *ast.Instr, pc int, offsets map[string]int) ([]byte, bool, error) {
ops := instr.Operands
switch mnem {
case "NOP":
if len(ops) != 0 {
return nil, true, fmt.Errorf("NOP takes no operands")
}
// The toolchain drops a bare NOP: no bytes at all.
return nil, true, nil
case "RDTIME":
// RDTIME rd reads the time CSR through CSRRS with a zero source.
if len(ops) != 1 {
return nil, true, fmt.Errorf("RDTIME expects 1 operand, got %d", len(ops))
}
rd := regFromOperand(ops[0])
if rd < 0 {
return nil, true, fmt.Errorf("RDTIME: invalid register")
}
return wordLE(riscvIType(riscvEnc{0x73, 0x2, 0x00}, rd, 0, 0xC01)), true, nil
case "NEG", "NOT", "SEQZ":
if len(ops) != 1 && len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops))
}
rs := regFromOperand(ops[0])
rd := rs
if len(ops) == 2 {
rd = regFromOperand(ops[1])
}
if rs < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
var word uint32
switch mnem {
case "NEG":
word = riscvRType(riscvInstrTable["SUB"], rd, 0, rs)
case "NOT":
word = riscvIType(riscvInstrTable["XORI"], rd, rs, -1)
case "SEQZ":
word = riscvIType(riscvInstrTable["SLTIU"], rd, rs, 1)
}
return wordLE(word), true, nil
case "ANDN", "ORN":
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
rs2 := regFromOperand(ops[0]) // the operand to invert
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
notReg := rd
if rs1 == notReg {
notReg = 31 // TMP, when the destination would be clobbered
}
out := wordLE(riscvIType(riscvInstrTable["XORI"], notReg, rs2, -1))
op := riscvInstrTable["AND"]
if mnem == "ORN" {
op = riscvInstrTable["OR"]
}
return append(out, wordLE(riscvRType(op, rd, rs1, notReg))...), true, nil
case "MAX", "MAXU", "MIN", "MINU":
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
if rs1 == rd {
// Process the destination-identical source first, as the
// toolchain does, so the sequence stays in place.
rs1, rs2 = rs2, rs1
}
if rs1 == rs2 {
// Identical inputs fold to ADDI $0 (compressed to C.MV and
// friends by the toolchain's compressor).
return riscvFoldedMove(rd, rs1), true, nil
}
slt1, slt2 := rs2, rs1
cmp := riscvInstrTable["SLT"]
if mnem == "MAX" || mnem == "MAXU" {
slt1, slt2 = slt2, slt1
}
if mnem == "MAXU" || mnem == "MINU" {
cmp = riscvInstrTable["SLTU"]
}
var out []byte
out = append(out, wordLE(riscvRType(cmp, 31, slt1, slt2))...) // the compare into TMP
out = append(out, wordLE(riscvRType(riscvInstrTable["SUB"], 31, 0, 31))...) // NEG TMP
out = append(out, wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rs2))...)
out = append(out, wordLE(riscvRType(riscvInstrTable["AND"], rd, 31, rd))...)
out = append(out, wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rd))...)
return out, true, nil
case "ROR", "RORW", "RORIW":
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
if isImmOperand(ops[0]) {
// Immediate rotate: SRLI the amount, SLLI the complement, OR.
imm := int(immFromOperand(ops[0]))
shiftW := 63
srlEnc := riscvInstrTable["SRLI"]
sllEnc := riscvInstrTable["SLLI"]
if mnem != "ROR" {
shiftW = 31
srlEnc = riscvInstrTable["SRLIW"]
sllEnc = riscvInstrTable["SLLIW"]
}
if imm < 0 || imm > shiftW {
return nil, true, fmt.Errorf("%s: shift amount out of range [0, %d]", mnem, shiftW)
}
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
var out []byte
out = append(out, wordLE(riscvRType(srlEnc, 31, rs1, imm))...)
sll := (-imm) & shiftW
if mnem == "ROR" && rd == rs1 && rd != 0 && sll >= 1 && sll <= 63 {
out = append(out, word16(rvcSLLI(uint32(rd), uint32(sll)))...) // C.SLLI
} else {
out = append(out, wordLE(riscvRType(sllEnc, rd, rs1, sll))...)
}
return append(out, wordLE(riscvRType(riscvInstrTable["OR"], rd, 31, rd))...), true, nil
}
// Register rotate: OR of the two opposite shifts through TMP.
if mnem == "RORIW" {
return nil, true, fmt.Errorf("RORIW takes an immediate shift amount")
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
sllEnc := riscvInstrTable["SLL"]
srlEnc := riscvInstrTable["SRL"]
if mnem == "RORW" {
sllEnc = riscvInstrTable["SLLW"]
srlEnc = riscvInstrTable["SRLW"]
}
var out []byte
out = append(out, wordLE(riscvRType(riscvInstrTable["SUB"], 31, 0, rs2))...) // NEG
out = append(out, wordLE(riscvRType(sllEnc, 31, rs1, 31))...)
out = append(out, wordLE(riscvRType(srlEnc, rd, rs1, rs2))...)
out = append(out, wordLE(riscvRType(riscvInstrTable["OR"], rd, 31, rd))...)
return out, true, nil
case "BGT", "BGTU", "BLE", "BLEU":
// The reversed conditional branches: BGT a, b, label is BLT b, a.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
a := regFromOperand(ops[0])
b := regFromOperand(ops[1])
if a < 0 || b < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
target := labelFromOperand(ops[2])
targetOff, ok := offsets[target]
if !ok {
return nil, true, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
if err := riscvCheckBranchOffset(target, offset); err != nil {
return nil, true, err
}
var enc riscvEnc
switch mnem {
case "BGT":
enc = riscvEnc{0x63, 0x4, 0x00} // blt b, a
case "BGTU":
enc = riscvEnc{0x63, 0x6, 0x00} // bltu b, a
case "BLE":
enc = riscvEnc{0x63, 0x5, 0x00} // bge b, a
case "BLEU":
enc = riscvEnc{0x63, 0x7, 0x00} // bgeu b, a
}
return wordLE(riscvBType(enc, b, a, offset)), true, nil
case "FABSD":
// FABSD rs, rd is FSGNJX.D (sign XOR, funct3 2) with the source in
if len(ops) != 2 {
return nil, true, fmt.Errorf("FABSD expects 2 operands, got %d", len(ops))
}
rs := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rs < 0 || rd < 0 {
return nil, true, fmt.Errorf("FABSD: invalid register")
}
return wordLE(riscvRType(riscvEnc{0x53, 0x2, 0x11}, rd, rs, rs)), true, nil
default:
return encodeRISCVVector(mnem, ops)
}
}
// riscvFoldedMove emits the ADDI $0, rs, rd the toolchain folds identical
// MIN/MAX inputs into, with the same compression its compressor applies to
// the folded form.
func riscvFoldedMove(rd, rs int) []byte {
switch {
case rd != 0 && rs != 0:
return word16(rvcCR(0x8, uint32(rd), uint32(rs))) // C.MV
case rd == 0 && rs == 0:
return word16(0x0001) // C.NOP
case rs == 0:
return word16(rvcCI(0x2, uint32(rd), 0)) // C.LI rd, $0
default:
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs, 0))
}
}
// encodeRISCVVector encodes the RVV slice GOROOT's kernels use. Registers
// are accepted in either spelling: the vector V registers and the integer
// registers share their 5-bit numbers, and the superset keeps hand-written
// probes simple. handled is always true: every name reaching here is one of
// the vector mnemonics.
func encodeRISCVVector(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
reg := regFromOperand
switch mnem {
case "VSETVLI", "VSETIVLI":
// INSTR avl, vsew, vlmul, vta, vma, rd.
if len(ops) != 6 {
return nil, true, fmt.Errorf("%s expects 6 operands, got %d", mnem, len(ops))
}
avl := 0
if isImmOperand(ops[0]) {
avl = int(immFromOperand(ops[0]))
if avl < 0 || avl > 31 {
return nil, true, fmt.Errorf("%s: avl immediate out of range [0, 31]", mnem)
}
} else {
avl = reg(ops[0])
if avl < 0 {
return nil, true, fmt.Errorf("%s: invalid avl register", mnem)
}
}
if mnem == "VSETIVLI" && !isImmOperand(ops[0]) {
return nil, true, fmt.Errorf("VSETIVLI expects an immediate avl")
}
vsew, err := riscvVTypeToken(operandRegName(ops[1]), "E", map[string]int{"8": 0, "16": 1, "32": 2, "64": 3})
if err != nil {
return nil, true, fmt.Errorf("%s: %w", mnem, err)
}
vlmul, err := riscvVTypeToken(operandRegName(ops[2]), "M", map[string]int{"1": 0, "2": 1, "4": 2, "8": 3, "F8": 5, "F4": 6, "F2": 7})
if err != nil {
return nil, true, fmt.Errorf("%s: %w", mnem, err)
}
vta := 0
switch operandRegName(ops[3]) {
case "TA":
vta = 1
case "TU":
default:
return nil, true, fmt.Errorf("%s: invalid tail policy %q (want TA or TU)", mnem, operandRegName(ops[3]))
}
vma := 0
switch operandRegName(ops[4]) {
case "MA":
vma = 1
case "MU":
default:
return nil, true, fmt.Errorf("%s: invalid mask policy %q (want MA or MU)", mnem, operandRegName(ops[4]))
}
rd := reg(ops[5])
if rd < 0 {
return nil, true, fmt.Errorf("%s: invalid destination register", mnem)
}
// An immediate avl always encodes as vsetivli, even under the
// VSETVLI spelling: the toolchain canonicalises the pair, and
// `VSETVLI $15` and `VSETIVLI $15` come out byte-identical
// (0xcd07f657) from GOARCH=riscv64 go tool asm.
ivli := mnem == "VSETIVLI" || isImmOperand(ops[0])
return wordLE(riscvVSetEnc(ivli, avl, riscvVType(vsew, vlmul, vta, vma), rd)), true, nil
case "VLE8V":
// Unit-stride load: INSTR (base), vd.
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rs1, ok := riscvVecMem(ops[0])
if !ok {
return nil, true, fmt.Errorf("%s: invalid memory operand", mnem)
}
vd := reg(ops[1])
if vd < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
return wordLE(riscvVLSType(0x07, 0, 0, 0, 0, rs1, vd)), true, nil
case "VSE8V", "VSE32V":
// Unit-stride store: INSTR vs3, (base).
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vs3 := reg(ops[0])
rs1, ok := riscvVecMem(ops[1])
if !ok {
return nil, true, fmt.Errorf("%s: invalid memory operand", mnem)
}
if vs3 < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
width := 0
if mnem == "VSE32V" {
width = 6
}
return wordLE(riscvVLSType(0x27, 0, 0, width, 0, rs1, vs3)), true, nil
case "VLSSEG4E32V", "VLSSEG8E32V":
// Constant-stride segmented load: INSTR (base), stride, vd.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rs1, ok := riscvVecMem(ops[0])
if !ok {
return nil, true, fmt.Errorf("%s: invalid memory operand", mnem)
}
rs2 := reg(ops[1])
vd := reg(ops[2])
if rs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid register operand", mnem)
}
nf := 3 // 4 fields
if mnem == "VLSSEG8E32V" {
nf = 7 // 8 fields
}
return wordLE(riscvVLSType(0x07, nf, 2, 6, int32(rs2), rs1, vd)), true, nil
case "VADDVV", "VXORVV", "VMSNEVV":
// Vector-vector: INSTR vs1, vs2, vd.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
vs1, vs2, vd := reg(ops[0]), reg(ops[1]), reg(ops[2])
if vs1 < 0 || vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
funct6 := map[string]int{"VADDVV": 0x00, "VXORVV": 0x0B, "VMSNEVV": 0x19}[mnem]
return wordLE(riscvVVInstr(funct6, riscvVf3VV, int32(vs1), vs2, vd)), true, nil
case "VADDVX", "VMSEQVX":
// Vector-scalar: INSTR rs1, vs2, vd (the scalar in the rs1 field).
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rs1, vs2, vd := reg(ops[0]), reg(ops[1]), reg(ops[2])
if rs1 < 0 || vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid register operand", mnem)
}
funct6 := 0x00
if mnem == "VMSEQVX" {
funct6 = 0x18
}
return wordLE(riscvVVInstr(funct6, riscvVf3VX, int32(rs1), vs2, vd)), true, nil
case "VSLLVI", "VSRLVI":
// Vector-immediate shift: INSTR $uimm, vs2, vd.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
imm := int(immFromOperand(ops[0]))
if imm < 0 || imm > 31 {
return nil, true, fmt.Errorf("%s: immediate out of range [0, 31]", mnem)
}
vs2, vd := reg(ops[1]), reg(ops[2])
if vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
funct6 := 0x25 // vsll.vi
if mnem == "VSRLVI" {
funct6 = 0x28 // vsrl.vi
}
return wordLE(riscvVVInstr(funct6, riscvVf3VI, int32(imm), vs2, vd)), true, nil
case "VFIRSTM":
// vmfirst.m rd, vs2: the unmasked form carries 0x11 in the rs1 field
// and sets the mask bit (funct7 = 0x20 | 1).
if len(ops) != 2 {
return nil, true, fmt.Errorf("VFIRSTM expects 2 operands, got %d", len(ops))
}
vs2, rd := reg(ops[0]), reg(ops[1])
if vs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("VFIRSTM: invalid register operand")
}
return wordLE(riscvVUnaryInstr(0x10, riscvVf3MV, 0x11, vs2, rd)), true, nil
case "VIDV":
// vid.v vd (vs2 must be v0; the unmasked form sets the mask bit).
if len(ops) != 1 {
return nil, true, fmt.Errorf("VIDV expects 1 operand, got %d", len(ops))
}
vd := reg(ops[0])
if vd < 0 {
return nil, true, fmt.Errorf("VIDV: invalid vector register")
}
return wordLE(riscvVUnaryInstr(0x14, riscvVf3MV, 0x11, 0, vd)), true, nil
case "VMV4RV":
// vmv4r.v vd, vs2: whole-register group move.
if len(ops) != 2 {
return nil, true, fmt.Errorf("VMV4RV expects 2 operands, got %d", len(ops))
}
vs2, vd := reg(ops[0]), reg(ops[1])
if vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("VMV4RV: invalid vector register")
}
return wordLE(riscvVUnaryInstr(0x27, 0x3, 0x3, vs2, vd)), true, nil
}
return nil, false, nil
}
// riscvVTypeToken parses a vsetvli configuration token (E8, M8, MF2 and
// friends): the letter prefix selects the field and the suffix its value
// through the given table.
func riscvVTypeToken(name, prefix string, codes map[string]int) (int, error) {
if len(name) <= len(prefix) || name[:len(prefix)] != prefix {
return 0, fmt.Errorf("invalid vtype token %q (want %s<width>)", name, prefix)
}
code, ok := codes[name[len(prefix):]]
if !ok {
return 0, fmt.Errorf("invalid vtype token %q", name)
}
return code, nil
}
// riscvVecMem reads a vector memory operand: a bare base register, the only
// addressing form the vector loads and stores carry. Frame-pseudo bases are
// rejected: the toolchain resolves no frame reference on the vector forms.
func riscvVecMem(op *ast.Operand) (rs1 int, ok bool) {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
return -1, false
}
if op.Addr.Base == "" || op.Addr.Offset != 0 {
return -1, false
}
rs1 = riscvRegNum(op.Addr.Base)
return rs1, rs1 >= 0
}
// Instruction type classifiers.
func isRTypeInstr(m string) bool {
switch m {
case "ADD", "SUB", "SLL", "SLT", "SLTU", "XOR", "SRL", "SRA", "OR", "AND",
"ADDW", "SUBW", "SLLW", "SRLW", "SRAW",
"MUL", "MULH", "MULHSU", "MULHU", "DIV", "DIVU", "REM", "REMU",
"MULW", "DIVW", "DIVUW", "REMW", "REMUW",
"CZEROEQZ", "CZERONEZ":
return true
}
return false
}
func isShiftImmInstr(m string) bool {
switch m {
case "SLLI", "SRLI", "SRAI", "SLLIW", "SRLIW", "SRAIW":
return true
}
return false
}
func isITypeInstr(m string) bool {
switch m {
case "ADDI", "ADDIW", "SLTI", "SLTIU", "XORI", "ORI", "ANDI", "JALR":
return true
}
return false
}
func isLoadInstr(m string) bool {
switch m {
case "LB", "LH", "LW", "LD", "LBU", "LHU", "LWU":
return true
}
return false
}
func isStoreInstr(m string) bool {
switch m {
case "SB", "SH", "SW", "SD":
return true
}
return false
}
func isBranchInstr(m string) bool {
switch m {
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU":
return true
}
return false
}
func isUTypeInstr(m string) bool {
return m == "LUI" || m == "AUIPC"
}
func isAMOInstr(m string) bool {
switch m {
case "AMOSWAPW", "AMOSWAPD", "AMOADDW", "AMOADDD",
"AMOANDW", "AMOANDD", "AMOORW", "AMOORD",
"AMOXORW", "AMOXORD", "AMOMAXW", "AMOMAXD",
"AMOMINW", "AMOMIND", "AMOMAXUW", "AMOMAXUD",
"AMOMINUW", "AMOMINUD":
return true
}
return false
}
func isFPArithInstr(m string) bool {
switch m {
case "FADDS", "FSUBS", "FMULS", "FDIVS",
"FADDD", "FSUBD", "FMULD", "FDIVD",
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD",
"FSGNJS", "FSGNJX", "FSGNJXD", "FSGNJXS", "FSGNJND", "FSGNJNS", "FSGNJNX":
return true
}
return false
}
func isFPLoadInstr(m string) bool {
return m == "FLW" || m == "FLD"
}
func isFPStoreInstr(m string) bool {
return m == "FSW" || m == "FSD"
}
func isLRInstr(m string) bool {
return m == "LRW" || m == "LRD"
}
func isSCInstr(m string) bool {
return m == "SCW" || m == "SCD"
}
func isFPCmpInstr(m string) bool {
switch m {
case "FEQS", "FLTS", "FLES", "FEQD", "FLTD", "FLED":
return true
}
return false
}
// Operand helpers.
func regFromOperand(op *ast.Operand) int {
// Register is in Addr.Base (from (base) syntax) or Addr.Sym.Name (bare ident).
if op.Addr.Base != "" {
return riscvRegNum(op.Addr.Base)
}
if op.Addr.Sym != nil && op.Addr.Sym.Name != "" {
return riscvRegNum(op.Addr.Sym.Name)
}
return -1
}
func immFromOperand(op *ast.Operand) int32 {
if op.Imm.HasVal {
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return int32(v)
}
return 0
}
// riscvImm32FromOperand reads an immediate for the MOV/I-type paths as a
// signed 32-bit value. The toolchain materialises wider constants through
// its SLLI expansion, which this assembler does not implement, so values
// outside the int32 span are diagnosed instead of silently truncated (MOV
// $0x123456789 must not assemble as $0x3456789). The neg flag carries the
// SUB $imm alias, whose negated value may fit when the written one does not.
func riscvImm32FromOperand(op *ast.Operand, neg bool) (int32, error) {
var v int64
if op.Imm.HasVal {
v = op.Imm.Val
if op.Imm.Neg {
v = -v
}
}
if neg {
v = -v
}
if int64(int32(v)) != v {
return 0, fmt.Errorf("immediate %d out of range; 64-bit materialisation not supported", v)
}
return int32(v), nil
}
func memFromOperand(op *ast.Operand) (rs1 int, imm int32) {
rs1 = riscvRegNum(op.Addr.Base)
imm = int32(op.Addr.Offset)
return
}
// memFromOperandWithFrame resolves a memory operand, handling FP/SP
// pseudo-registers via the frame mapping.
func memFromOperandWithFrame(op *ast.Operand, fi riscvFrameInfo) (rs1 int, imm int32) {
// Check for a pseudo-register reference (name+offset(FP) or name+offset(SP)).
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
return riscvResolvePseudo(op.Addr.Sym, fi)
}
// Plain register+offset memory reference.
return memFromOperand(op)
}
func labelFromOperand(op *ast.Operand) string {
if op.Addr.Sym != nil {
return op.Addr.Sym.Name
}
return op.Raw
}
// suggestLabel returns a "did you mean" suggestion for an undefined label.
func suggestLabel(target string, offsets map[string]int) string {
if len(offsets) == 0 {
return ""
}
// Find the closest matching label using Levenshtein distance.
bestDist := len(target) + 1
var best string
for name := range offsets {
dist := levenshtein(target, name)
if dist < bestDist {
bestDist = dist
best = name
}
}
// Only suggest if the distance is small enough.
if bestDist <= 3 && bestDist < len(target)/2+1 {
return fmt.Sprintf("; did you mean %q?", best)
}
return ""
}
// levenshtein computes the Levenshtein distance between two strings.
func levenshtein(a, b string) int {
la, lb := len(a), len(b)
if la == 0 {
return lb
}
if lb == 0 {
return la
}
// Create a matrix of distances.
prev := make([]int, lb+1)
curr := make([]int, lb+1)
for j := 0; j <= lb; j++ {
prev[j] = j
}
for i := 1; i <= la; i++ {
curr[0] = i
for j := 1; j <= lb; j++ {
cost := 1
if a[i-1] == b[j-1] {
cost = 0
}
curr[j] = min3(curr[j-1]+1, prev[j]+1, prev[j-1]+cost)
}
prev, curr = curr, prev
}
return prev[lb]
}
func min3(a, b, c int) int {
if a < b {
if a < c {
return a
}
return c
}
if b < c {
return b
}
return c
}