// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "fmt" "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" ) // assembleRISCV assembles a RISC-V TEXT function body into machine code. // It handles the full RV64IMAFDC instruction set including RVC compression. func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { fi := riscvComputeFrame(t) prologue := riscvPrologue(fi) var relocs []Reloc var spadj []SpadjStep // The prologue raises the SP delta by autosize; the boundary is reported // at the pc just past its ADDI, exactly as the toolchain's pctospadj does. if fi.autosize != 0 { spadj = append(spadj, SpadjStep{PC: riscvPrologueSpadjPC(fi), Value: fi.autosize}) } // Pass 1: collect instructions and compute label offsets assuming 4 bytes // per instruction (or 8 for MOV $large-imm). No encoding yet. type instrRec struct { instr *ast.Instr compressed bool code []byte } var recs []instrRec offsets := map[string]int{} pos := len(prologue) for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: offsets[s.Name.Text] = pos case *ast.Instr: recs = append(recs, instrRec{instr: s}) pos += riscvInstrSize(s, fi) } } // Pass 2: encode each instruction using Pass-1 offsets. pc := len(prologue) for i := range recs { code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil) // no relocs in Pass 2 if err != nil { return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err) } recs[i].code = code pc += len(code) } // Pass 3: try RVC compression. for i := range recs { if c16, ok := tryCompressRVC(recs[i].instr, fi); ok { recs[i].compressed = true recs[i].code = []byte{byte(c16), byte(c16 >> 8)} } } // Pass 4: recompute offsets with actual sizes. offsets = map[string]int{} pos = len(prologue) for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: offsets[s.Name.Text] = pos case *ast.Instr: for _, r := range recs { if r.instr == s { pos += len(r.code) break } } } } // Pass 5: re-encode branches with corrected offsets. Record relocations // during this final pass (relocation offsets are relative to instruction start). out := append([]byte(nil), prologue...) pc = len(prologue) preCount := len(relocs) var lines []LineEntry for _, r := range recs { lines = append(lines, LineEntry{Offset: pc, Line: r.instr.Pos().Line}) if r.compressed && !isBranchLike(r.instr.Mnemonic.Text) { out = append(out, r.code...) pc += len(r.code) } else { code, err := encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs) if err != nil { return nil, nil, nil, nil, nil, err } if c16, ok := tryCompressRVC(r.instr, fi); ok { code = []byte{byte(c16), byte(c16 >> 8)} } // Make newly added relocation offsets absolute (subtract prologue to make // them function-relative, then the caller adds fn.Offset). After // points just past the AUIPC+second-instruction pair, which is // always 8 bytes wide for these static-symbol references. for j := preCount; j < len(relocs); j++ { relocs[j].Off += pc - len(prologue) relocs[j].After = relocs[j].Off + 8 } preCount = len(relocs) // The RET's epilogue closes the frame: the SP delta returns to zero // after its ADDI (restore LR + ADDI). if strings.ToUpper(r.instr.Mnemonic.Text) == "RET" && fi.autosize != 0 { spadj = append(spadj, SpadjStep{PC: pc + riscvReturnEpilogueLen(fi), Value: 0}) } out = append(out, code...) pc += len(code) } } return out, offsets, relocs, lines, spadj, nil } // riscvInstrSize returns the encoded size in bytes of a RISC-V instruction. // Most instructions are 4 bytes; MOV with a large immediate is 8 (LUI+ADDIW). func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int { mnem := instr.Mnemonic.Text ops := instr.Operands if mnem == "RET" { return len(riscvReturn(fi)) } if mnem == "MOV" && len(ops) == 2 { // MOV $sym(SB), rd → 8 bytes (AUIPC + ADDI). if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && ops[0].Imm.Sym.Pseudo == "SB" { return 8 } // MOV sym(SB), rd → 8 bytes (AUIPC + LD). if isMemOperand(ops[0]) && ops[0].Addr.Sym != nil && ops[0].Addr.Sym.Pseudo == "SB" { return 8 } // MOV rd, sym(SB) → 8 bytes (AUIPC + SD). if isMemOperand(ops[1]) && ops[1].Addr.Sym != nil && ops[1].Addr.Sym.Pseudo == "SB" { return 8 } // MOV $imm, rd → large immediate needs LUI+ADDIW. if isImmOperand(ops[0]) { imm := immFromOperand(ops[0]) if imm < -2048 || imm > 2047 { return 8 } } } return 4 } // isBranchLike reports whether a mnemonic is a branch or jump that needs // recalculated offsets after compression. func isBranchLike(mnem string) bool { switch mnem { case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "JMP", "JAL": return true } return false } // encodeRISCVInstr encodes a single RISC-V instruction. func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) { mnem := instr.Mnemonic.Text ops := instr.Operands var word uint32 // Handle pseudo-instructions and special cases first. switch mnem { case "RET": // RET = epilogue (restore LR and close the frame when present) + C.JR ra. return riscvReturn(fi), nil case "CALL": // CALL target → AUIPC X1, %pcrel_hi + JALR X1, %pcrel_lo(X1). // For now, emit AUIPC X1, 0 + JALR X1, 0(X1) with zero offsets. // The relocation system will fill the actual offsets. if len(ops) >= 1 { target := labelFromOperand(ops[0]) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } offset := int32(targetOff - pc) // AUIPC X1, upper 20 bits hi := (offset + 0x800) >> 12 word1 := riscvUType(riscvEnc{0x17, 0x0, 0x00}, 1, hi<<12) // JALR X1, lower 12 bits(X1) lo := offset - (hi << 12) word2 := riscvIType(riscvEnc{0x67, 0x0, 0x00}, 1, 1, lo) var out []byte out = append(out, byte(word1), byte(word1>>8), byte(word1>>16), byte(word1>>24)) out = append(out, byte(word2), byte(word2>>8), byte(word2>>16), byte(word2>>24)) return out, nil } // CALL with no target: encode as NOP (unsupported). word = riscvIType(riscvEnc{0x13, 0x0, 0x00}, 0, 0, 0) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil case "JMP": // JMP = JAL X0, target. Try C.J compression. var target string if len(ops) >= 1 { target = labelFromOperand(ops[0]) } targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } offset := int32(targetOff - pc) // C.J: funct3=0x5, offset in ±2 KB, bit 0 must be 0. if offset >= -2048 && offset <= 2046 && offset%2 == 0 { c16 := rvcCJ(0x5, offset) return []byte{byte(c16), byte(c16 >> 8)}, nil } word = riscvJType(0, offset) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil case "JAL": rd := 0 var target string if len(ops) >= 2 { rd = regFromOperand(ops[0]) target = labelFromOperand(ops[1]) } else if len(ops) == 1 { target = labelFromOperand(ops[0]) } targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } offset := int32(targetOff - pc) // JAL X0, target → C.J when offset fits. if rd == 0 && offset >= -2048 && offset <= 2046 && offset%2 == 0 { c16 := rvcCJ(0x5, offset) return []byte{byte(c16), byte(c16 >> 8)}, nil } word = riscvJType(rd, offset) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil // MOV is a pseudo-instruction that the Go assembler uses for loads, // stores, register moves and immediate loads. case "MOV": return encodeRISCVMov(instr, offsets, fi, relocs) // JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1). case "JALR": return encodeRISCVJALR(instr, fi) // System instructions with no operands. case "FENCE", "ECALL", "EBREAK": enc, ok := riscvInstrTable[mnem] if !ok { return nil, fmt.Errorf("unsupported system instruction %q", mnem) } word = riscvIType(enc, 0, 0, 0) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // FP conversion / move instructions use a separate table (rs2 encodes // the conversion type, not a register). Handle them before the main // table lookup. if cvtEnc, ok := riscvCvtTable[mnem]; ok { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rs1 := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } word := riscvCvtType(cvtEnc, rd, rs1) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // R4-type fused multiply-add: INSTR rs1, rs2, rs3, rd (destination last). if fmaEnc, ok := riscvFmaTable[mnem]; ok { if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } rs1 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[1]) rs3 := regFromOperand(ops[2]) rd := regFromOperand(ops[3]) if rd < 0 || rs1 < 0 || rs2 < 0 || rs3 < 0 { return nil, fmt.Errorf("invalid FP register in %s", mnem) } word := riscvFmaType(fmaEnc, rd, rs1, rs2, rs3) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // CSR instructions: INSTR csr, rs1|uimm, rd (destination last). if csrEnc, ok := riscvCsrTable[mnem]; ok { if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } csr := immFromOperand(ops[0]) // CSR address (12-bit) rd := regFromOperand(ops[2]) // destination register if rd < 0 { return nil, fmt.Errorf("invalid destination register in %s", mnem) } var src int if csrEnc.imm { // Immediate variant: ops[1] is a 5-bit unsigned immediate. src = int(immFromOperand(ops[1])) if src < 0 || src > 31 { return nil, fmt.Errorf("%s: uimm out of range 0-31", mnem) } } else { // Register variant: ops[1] is a register. src = regFromOperand(ops[1]) if src < 0 { return nil, fmt.Errorf("invalid source register in %s", mnem) } } word := riscvCsrType(csrEnc, rd, src, csr) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } enc, ok := riscvInstrTable[mnem] if !ok { return nil, fmt.Errorf("unsupported RISC-V instruction %q", mnem) } switch { // R-type: Plan 9 order is INSTR src1, src2, dst (destination last). case len(ops) == 3 && isRTypeInstr(mnem): rs1 := regFromOperand(ops[0]) // source 1 (first operand) rs2 := regFromOperand(ops[1]) // source 2 (second operand) rd := regFromOperand(ops[2]) // destination (last operand) if rd < 0 || rs1 < 0 || rs2 < 0 { return nil, fmt.Errorf("invalid register in %s", mnem) } word = riscvRType(enc, rd, rs1, rs2) // I-type shift (SLLI, SRLI, SRAI): INSTR rs, $shamt, rd. case len(ops) == 3 && isShiftImmInstr(mnem): rs1 := regFromOperand(ops[0]) shamt := int(immFromOperand(ops[1])) rd := regFromOperand(ops[2]) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid register in %s", mnem) } word = riscvRType(enc, rd, rs1, shamt) // AMO atomics: Plan 9 order is INSTR src, (addr), dst. case len(ops) == 3 && isAMOInstr(mnem): rs2 := regFromOperand(ops[0]) // source value rs1, _ := memFromOperandWithFrame(ops[1], fi) // memory address rd := regFromOperand(ops[2]) // destination (old value) if rd < 0 || rs1 < 0 || rs2 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } word = riscvAMOType(enc, rd, rs1, rs2) // FP arithmetic: Plan 9 order is INSTR src1, src2, dst. case len(ops) == 3 && isFPArithInstr(mnem): rs1 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[1]) rd := regFromOperand(ops[2]) if rd < 0 || rs1 < 0 || rs2 < 0 { return nil, fmt.Errorf("invalid FP register in %s", mnem) } word = riscvRType(enc, rd, rs1, rs2) // FP arithmetic (2-operand): FSQRT src, dst. case len(ops) == 2 && isFPArithInstr(mnem): rs1 := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid FP register in %s", mnem) } word = riscvRType(enc, rd, rs1, 0) // FP loads: INSTR addr, freg (Plan 9: source first). case len(ops) == 2 && isFPLoadInstr(mnem): rd := regFromOperand(ops[1]) rs1, imm := memFromOperandWithFrame(ops[0], fi) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } word = riscvIType(enc, rd, rs1, imm) // FP stores: INSTR freg, addr (Plan 9: source first). case len(ops) == 2 && isFPStoreInstr(mnem): rs2 := regFromOperand(ops[0]) rs1, imm := memFromOperandWithFrame(ops[1], fi) if rs2 < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } word = riscvSType(enc, rs1, rs2, imm) // LR (load-reserved): INSTR (addr), dst — 2 operands. case len(ops) == 2 && isLRInstr(mnem): rs1, _ := memFromOperandWithFrame(ops[0], fi) rd := regFromOperand(ops[1]) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } word = riscvAMOType(enc, rd, rs1, 0) // rs2=0 for LR // SC (store-conditional): INSTR src, (addr), dst — 3 operands. case len(ops) == 3 && isSCInstr(mnem): rs2 := regFromOperand(ops[0]) rs1, _ := memFromOperandWithFrame(ops[1], fi) rd := regFromOperand(ops[2]) if rd < 0 || rs1 < 0 || rs2 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } word = riscvAMOType(enc, rd, rs1, rs2) // FP compare: INSTR src1, src2, dst(int) — result in integer register. case len(ops) == 3 && isFPCmpInstr(mnem): rs1 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[1]) rd := regFromOperand(ops[2]) if rd < 0 || rs1 < 0 || rs2 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } word = riscvRType(enc, rd, rs1, rs2) // I-type with immediate: Plan 9 order is INSTR src, imm, dst. case len(ops) == 3 && isITypeInstr(mnem): rs1 := regFromOperand(ops[0]) // source register imm := immFromOperand(ops[1]) // immediate rd := regFromOperand(ops[2]) // destination if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid register in %s", mnem) } word = riscvIType(enc, rd, rs1, imm) // Loads: rd, offset(rs1) — Plan 9 order is LD src, dst. case len(ops) == 2 && isLoadInstr(mnem): rd := regFromOperand(ops[1]) // destination (last operand) rs1, imm := memFromOperandWithFrame(ops[0], fi) // memory source (first operand) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } word = riscvIType(enc, rd, rs1, imm) // Stores: Plan 9 order is SD src, dst (src=register, dst=memory). case len(ops) == 2 && isStoreInstr(mnem): rs2 := regFromOperand(ops[0]) // source register (first operand) rs1, imm := memFromOperandWithFrame(ops[1], fi) // memory dest (last operand) if rs2 < 0 || rs1 < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } word = riscvSType(enc, rs1, rs2, imm) // Branches: rs1, rs2, label. case len(ops) == 3 && isBranchInstr(mnem): rs1 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[1]) target := labelFromOperand(ops[2]) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } offset := int32(targetOff - pc) if rs1 < 0 || rs2 < 0 { return nil, fmt.Errorf("invalid register in %s", mnem) } // Try C.BEQZ / C.BNEZ compression. if (mnem == "BEQ" || mnem == "BNE") && rs2 == 0 && isRVCIntReg(rs1) { if cOff := offset; cOff >= -256 && cOff <= 254 && cOff%2 == 0 { funct3 := uint32(0x6) // C.BEQZ if mnem == "BNE" { funct3 = 0x7 // C.BNEZ } c16 := rvcCB(funct3, rvcReg3(rs1), offset) return []byte{byte(c16), byte(c16 >> 8)}, nil } } word = riscvBType(enc, rs1, rs2, offset) // U-type: rd, imm. case len(ops) == 2 && isUTypeInstr(mnem): rd := regFromOperand(ops[0]) imm := immFromOperand(ops[1]) if rd < 0 { return nil, fmt.Errorf("invalid register in %s", mnem) } word = riscvUType(enc, rd, imm) default: return nil, fmt.Errorf("cannot encode %s with %d operands", mnem, len(ops)) } // Emit as little-endian 32-bit word. return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // isMemOperand reports whether an operand is a memory reference // (frame-relative such as name+off(FP) or register-relative such as (X10)). func isMemOperand(op *ast.Operand) bool { if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { return true // name+off(FP), name+off(SP) } if op.Addr.Base != "" && op.Addr.Sym == nil { return true // (reg) } return false } // isImmOperand reports whether an operand is an immediate ($value). func isImmOperand(op *ast.Operand) bool { if op.Kind == ast.OpImmediate { return true } if op.Imm.HasVal { return true } return false } // encodeRISCVMov encodes the MOV pseudo-instruction. // // The Go RISC-V assembler uses MOV for: // - MOV name+off(FP), Rd load from frame // - MOV Rd, name+off(FP) store to frame // - MOV (Rs), Rd register-relative load // - MOV Rs, (Rd) register-relative store // - MOV Rs, Rd register-to-register move (ADDI $0) // - MOV $imm, Rd load immediate (ADDI or LUI+ADDIW) func encodeRISCVMov(instr *ast.Instr, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) { ops := instr.Operands if len(ops) != 2 { return nil, fmt.Errorf("MOV expects 2 operands, got %d", len(ops)) } src := ops[0] dst := ops[1] // Immediate → register. if isImmOperand(src) { // MOV $sym(SB), rd — load address of a static symbol or external. if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { rd := regFromOperand(dst) if rd < 0 { return nil, fmt.Errorf("MOV $sym(SB): invalid destination register") } return encodeRISCVSBAddr(src.Imm.Sym, rd, relocs), nil } // MOV $sym(FP/SP), rd — not supported: immediate symbol references // other than SB cannot be encoded as a simple immediate. if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" { return nil, fmt.Errorf("MOV $%s(%s): unsupported immediate symbol reference (only SB is supported)", src.Imm.Sym.Name, src.Imm.Sym.Pseudo) } rd := regFromOperand(dst) if rd < 0 { return nil, fmt.Errorf("MOV $imm: invalid destination register") } imm := immFromOperand(src) return encodeRISCVLoadImm(rd, imm), nil } // Memory → register (load). if isMemOperand(src) && !isMemOperand(dst) { rd := regFromOperand(dst) // MOV sym(SB), rd — load from static data. if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" { if rd < 0 { return nil, fmt.Errorf("MOV sym(SB): invalid destination register") } return encodeRISCVSBLoad(src.Addr.Sym, rd, relocs), nil } rs1, off := memFromOperandWithFrame(src, fi) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("MOV load: invalid operand") } word := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rs1, off) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // Register → memory (store). if !isMemOperand(src) && isMemOperand(dst) { rs2 := regFromOperand(src) // MOV rd, sym(SB) — store to static data. if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" { if rs2 < 0 { return nil, fmt.Errorf("MOV rd, sym(SB): invalid source register") } return encodeRISCVSBStore(dst.Addr.Sym, rs2, relocs), nil } rs1, off := memFromOperandWithFrame(dst, fi) if rs2 < 0 || rs1 < 0 { return nil, fmt.Errorf("MOV store: invalid operand") } word := riscvSType(riscvEnc{0x23, 0x3, 0x00}, rs1, rs2, off) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // Register → register (ADDI $0, src, dst). { rs1 := regFromOperand(src) rd := regFromOperand(dst) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("MOV: invalid register operand") } word := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs1, 0) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } } // encodeRISCVLoadImm encodes loading an immediate into a register. // For 12-bit immediates: ADDI $imm, ZERO, rd. // For larger: LUI $hi, rd + ADDIW $lo, rd, rd. func encodeRISCVLoadImm(rd int, imm int32) []byte { if imm >= -2048 && imm <= 2047 { word := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, 0, imm) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)} } // LUI + ADDIW for larger constants. var out []byte hi := int32((uint32(imm)+0x800)>>12) << 12 // LUI loads upper 20 bits lo := imm - hi wordLUI := riscvUType(riscvEnc{0x37, 0x0, 0x00}, rd, hi) out = append(out, byte(wordLUI), byte(wordLUI>>8), byte(wordLUI>>16), byte(wordLUI>>24)) if lo != 0 { wordADDIW := riscvIType(riscvEnc{0x1B, 0x0, 0x00}, rd, rd, lo) out = append(out, byte(wordADDIW), byte(wordADDIW>>8), byte(wordADDIW>>16), byte(wordADDIW>>24)) } return out } // encodeRISCVSBAddr emits AUIPC + ADDI to load the address of a static // symbol into rd, recording the single R_RISCV_PCREL_ITYPE relocation the Go // toolchain uses for the pair (the object-file emitters expand or map it). func encodeRISCVSBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { name := sym.Name if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType, Addend: sym.Offset}) } auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0) addi := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rd, 0) return append(wordLE(auipc), wordLE(addi)...) } // encodeRISCVSBLoad emits AUIPC + LD to load from a static symbol into rd, // recording the single R_RISCV_PCREL_ITYPE relocation for the pair. func encodeRISCVSBLoad(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { name := sym.Name if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType, Addend: sym.Offset}) } auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0) ld := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rd, 0) return append(wordLE(auipc), wordLE(ld)...) } // encodeRISCVSBStore emits AUIPC + SD to store a register into a static symbol, // recording the single R_RISCV_PCREL_STYPE relocation for the pair. func encodeRISCVSBStore(sym *ast.Symbol, rs2 int, relocs *[]Reloc) []byte { tmp := 31 // X31 = T6 name := sym.Name if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELSType, Addend: sym.Offset}) } auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, tmp, 0) sd := riscvSType(riscvEnc{0x23, 0x3, 0x00}, tmp, rs2, 0) var out []byte out = append(out, wordLE(auipc)...) out = append(out, wordLE(sd)...) return out } // wordLE encodes a uint32 as 4 little-endian bytes. func wordLE(w uint32) []byte { return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)} } // encodeRISCVJALR encodes the JALR indirect jump/call instruction. // Plan 9: JALR rs1, rd (2 regs) or JALR offset(rs1) (memory → rd=X1). func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) { ops := instr.Operands if len(ops) == 2 { rs1 := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("JALR: invalid register operand") } word := riscvIType(riscvEnc{0x67, 0x0, 0x00}, rd, rs1, 0) return wordLE(word), nil } if len(ops) == 1 { rs1, imm := memFromOperandWithFrame(ops[0], fi) if rs1 < 0 { return nil, fmt.Errorf("JALR: invalid memory operand") } word := riscvIType(riscvEnc{0x67, 0x0, 0x00}, 1, rs1, imm) return wordLE(word), nil } return nil, fmt.Errorf("JALR expects 1 or 2 operands, got %d", len(ops)) } // tryCompressRVC attempts to compress a RISC-V instruction to its 16-bit // RVC form. It returns the compressed instruction word and true on success. func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) { mnem := instr.Mnemonic.Text ops := instr.Operands switch mnem { case "LD", "MOV": // LD rd, offset(SP) → C.LDSP when rd≠0 and uimm[8:3] fits. // MOV name+off(FP), rd → load, same compression. if mnem == "MOV" && len(ops) == 2 && isImmOperand(ops[0]) { return 0, false } // MOV reg, reg → C.MV (CR-type: funct4=0x8). if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && !isMemOperand(ops[1]) && !isImmOperand(ops[0]) { rs1 := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rs1 != -1 && rd != -1 && rs1 != 0 && rd != 0 { return rvcCR(0x8, uint32(rd), uint32(rs1)), true } } rd, rs1, imm := extractLDParams(instr, fi) if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcLSP(0x3, uint32(rd), uint32(imm)), true } // MOV reg, mem → store, try C.SDSP. if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && isMemOperand(ops[1]) { rs2, rs1, imm := extractSDParams(instr, fi) if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcSSP(0x7, uint32(rs2), uint32(imm)), true } } case "SD": // SD rs2, offset(SP) → C.SDSP when uimm[8:3] fits (CSS-type). rs2, rs1, imm := extractSDParams(instr, fi) if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcSSP(0x7, uint32(rs2), uint32(imm)), true } case "ADDI": rd, rs1, imm := extractITypeParams(instr, fi) if rd == -1 || rs1 == -1 { return 0, false } if rd == rs1 && rd != 0 && imm != 0 && imm >= -32 && imm <= 31 { // C.ADDI: funct3=0x0, rs1/rd, nzimm[5:0] return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true } if rs1 == 0 && rd != 0 && imm >= -32 && imm <= 31 { // C.LI: funct3=0x2, rd, imm[5:0] return rvcCI(0x2, uint32(rd), uint32(imm)&0x3F), true } if rs1 != 0 && rd != 0 && imm == 0 { // C.MV: funct4=0x8, rd, rs1 (CR-type) return rvcCR(0x8, uint32(rd), uint32(rs1)), true } case "JAL": // JAL X0, target → C.J when offset fits in ±2KB. if len(ops) >= 1 { // For JAL with implicit rd=0 (JMP alias), check target. // C.J: funct3=0x5 // Offset is computed at encode time — we can't check it here. return 0, false } case "JMP": // C.J — handled in encodeRISCVInstr with actual offset. return 0, false case "BEQ": // C.BEQZ — handled in encodeRISCVInstr with actual offset. return 0, false case "BNE": // C.BNEZ — handled in encodeRISCVInstr with actual offset. return 0, false case "ADD": // ADD rd, rs2 → C.ADD (CR-type, funct4=0x9) when rd == rs1; ADD is // commutative, so if rd == rs2, swap. if len(ops) == 3 { rs1 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[1]) rd := regFromOperand(ops[2]) if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 { if rd == rs1 && rs2 != 0 { return rvcCR(0x9, uint32(rd), uint32(rs2)), true } if rd == rs2 && rs1 != 0 { return rvcCR(0x9, uint32(rd), uint32(rs1)), true } } } case "SUB", "XOR", "OR", "AND": // C.SUB (0x23,0), C.XOR (0x23,1), C.OR (0x23,2), C.AND (0x23,3) if len(ops) == 3 { var funct2 uint32 switch mnem { case "SUB": funct2 = 0x0 case "XOR": funct2 = 0x1 case "OR": funct2 = 0x2 case "AND": funct2 = 0x3 } rs1 := regFromOperand(ops[0]) rs2 := regFromOperand(ops[1]) rd := regFromOperand(ops[2]) if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 { if rd == rs1 && isRVCIntReg(rd) && isRVCIntReg(rs2) && rs2 != 0 { return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs2)), true } } } case "FLD": // FLD rd, imm(SP) → C.FLDSP (CI-type, funct3=0x1). rd, rs1, imm := extractLDParams(instr, fi) if rs1 == 2 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcLSP(0x1, uint32(rd), uint32(imm)), true } case "FSD": // FSD rs2, imm(SP) → C.FSDSP (CSS-type, funct3=0x5). rs2, rs1, imm := extractSDParams(instr, fi) if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcSSP(0x5, uint32(rs2), uint32(imm)), true } case "LUI": // LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in 6 bits. if len(ops) == 2 { rd := regFromOperand(ops[0]) imm := immFromOperand(ops[1]) if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= 1 && imm <= 63 { return rvcCI(0x3, uint32(rd), uint32(imm)&0x3F), true } } case "ADDIW": rd, rs1, imm := extractITypeParams(instr, fi) if rd == rs1 && rd != 0 && imm >= -32 && imm <= 31 { return rvcCI(0x1, uint32(rd), uint32(imm)&0x3F), true } case "SLLI", "SRLI", "SRAI": // C.SLLI (funct3=0x0), C.SRLI (funct3=0x4, funct2=0), C.SRAI (funct3=0x4, funct2=1). rd, rs1, imm := extractITypeParams(instr, fi) if rd == rs1 && rd != 0 && imm != 0 && imm >= 1 && imm <= 63 { if mnem == "SLLI" { // C.SLLI: funct3=0, CI-type with shamt in bits [12|6:2]. // For simplicity, use the standard CI format — the shamt is in imm[5:0]. return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true } if isRVCIntReg(rd) { funct2 := uint32(0x0) if mnem == "SRAI" { funct2 = 0x1 } // CB-format shift: funct3=0x4, shamt in bits [12|6:2]. // Use simplified encoding for now. _ = funct2 return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true } } case "ANDI": rd, rs1, imm := extractITypeParams(instr, fi) if isRVCIntReg(rd) && rd == rs1 && imm >= -32 && imm <= 31 { // C.ANDI: funct3=0x4, funct2=0x2 (CB-type). // Simplified encoding for now. return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true } } return 0, false } // extractLDParams extracts rd, rs1, and immediate offset for a load instruction. func extractLDParams(instr *ast.Instr, fi riscvFrameInfo) (rd, rs1 int, imm int32) { ops := instr.Operands if len(ops) != 2 { return -1, -1, 0 } if instr.Mnemonic.Text == "MOV" { if isMemOperand(ops[0]) { rs1, imm = memFromOperandWithFrame(ops[0], fi) rd = regFromOperand(ops[1]) } else { return -1, -1, 0 } } else { rs1, imm = memFromOperandWithFrame(ops[0], fi) rd = regFromOperand(ops[1]) } return } // extractSDParams extracts rs2, rs1, and immediate offset for a store instruction. func extractSDParams(instr *ast.Instr, fi riscvFrameInfo) (rs2, rs1 int, imm int32) { ops := instr.Operands if len(ops) != 2 { return -1, -1, 0 } rs2 = regFromOperand(ops[0]) rs1, imm = memFromOperandWithFrame(ops[1], fi) return } // extractITypeParams extracts rd, rs1, and immediate for an I-type instruction. func extractITypeParams(instr *ast.Instr, fi riscvFrameInfo) (rd, rs1 int, imm int32) { ops := instr.Operands if len(ops) != 3 { return -1, -1, 0 } rs1 = regFromOperand(ops[0]) imm = immFromOperand(ops[1]) rd = regFromOperand(ops[2]) return } // Instruction type classifiers. func isRTypeInstr(m string) bool { switch m { case "ADD", "SUB", "SLL", "SLT", "SLTU", "XOR", "SRL", "SRA", "OR", "AND", "ADDW", "SUBW", "SLLW", "SRLW", "SRAW", "MUL", "MULH", "MULHSU", "MULHU", "DIV", "DIVU", "REM", "REMU", "MULW", "DIVW", "DIVUW", "REMW", "REMUW": return true } return false } func isShiftImmInstr(m string) bool { switch m { case "SLLI", "SRLI", "SRAI", "SLLIW", "SRLIW", "SRAIW": return true } return false } func isITypeInstr(m string) bool { switch m { case "ADDI", "ADDIW", "SLTI", "SLTIU", "XORI", "ORI", "ANDI", "JALR": return true } return false } func isLoadInstr(m string) bool { switch m { case "LB", "LH", "LW", "LD", "LBU", "LHU", "LWU": return true } return false } func isStoreInstr(m string) bool { switch m { case "SB", "SH", "SW", "SD": return true } return false } func isBranchInstr(m string) bool { switch m { case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU": return true } return false } func isUTypeInstr(m string) bool { return m == "LUI" || m == "AUIPC" } func isAMOInstr(m string) bool { switch m { case "AMOSWAPW", "AMOSWAPD", "AMOADDW", "AMOADDD", "AMOANDW", "AMOANDD", "AMOORW", "AMOORD", "AMOXORW", "AMOXORD", "AMOMAXW", "AMOMAXD", "AMOMINW", "AMOMIND", "AMOMAXUW", "AMOMAXUD", "AMOMINUW", "AMOMINUD": return true } return false } func isFPArithInstr(m string) bool { switch m { case "FADDS", "FSUBS", "FMULS", "FDIVS", "FADDD", "FSUBD", "FMULD", "FDIVD", "FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD": return true } return false } func isFPLoadInstr(m string) bool { return m == "FLW" || m == "FLD" } func isFPStoreInstr(m string) bool { return m == "FSW" || m == "FSD" } func isLRInstr(m string) bool { return m == "LRW" || m == "LRD" } func isSCInstr(m string) bool { return m == "SCW" || m == "SCD" } func isFPCmpInstr(m string) bool { switch m { case "FEQS", "FLTS", "FLES", "FEQD", "FLTD", "FLED": return true } return false } func isFPCvtInstr(m string) bool { _, ok := riscvCvtTable[m] return ok } // Operand helpers. func regFromOperand(op *ast.Operand) int { // Register is in Addr.Base (from (base) syntax) or Addr.Sym.Name (bare ident). if op.Addr.Base != "" { return riscvRegNum(op.Addr.Base) } if op.Addr.Sym != nil && op.Addr.Sym.Name != "" { return riscvRegNum(op.Addr.Sym.Name) } return -1 } func immFromOperand(op *ast.Operand) int32 { if op.Imm.HasVal { v := op.Imm.Val if op.Imm.Neg { v = -v } return int32(v) } return 0 } func memFromOperand(op *ast.Operand) (rs1 int, imm int32) { rs1 = riscvRegNum(op.Addr.Base) imm = int32(op.Addr.Offset) return } // memFromOperandWithFrame resolves a memory operand, handling FP/SP // pseudo-registers via the frame mapping. func memFromOperandWithFrame(op *ast.Operand, fi riscvFrameInfo) (rs1 int, imm int32) { // Check for a pseudo-register reference (name+offset(FP) or name+offset(SP)). if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { return riscvResolvePseudo(op.Addr.Sym, fi) } // Plain register+offset memory reference. return memFromOperand(op) } func labelFromOperand(op *ast.Operand) string { if op.Addr.Sym != nil { return op.Addr.Sym.Name } return op.Raw } // suggestLabel returns a "did you mean" suggestion for an undefined label. func suggestLabel(target string, offsets map[string]int) string { if len(offsets) == 0 { return "" } // Find the closest matching label using Levenshtein distance. bestDist := len(target) + 1 var best string for name := range offsets { dist := levenshtein(target, name) if dist < bestDist { bestDist = dist best = name } } // Only suggest if the distance is small enough. if bestDist <= 3 && bestDist < len(target)/2+1 { return fmt.Sprintf(" — did you mean %q?", best) } return "" } // levenshtein computes the Levenshtein distance between two strings. func levenshtein(a, b string) int { la, lb := len(a), len(b) if la == 0 { return lb } if lb == 0 { return la } // Create a matrix of distances. prev := make([]int, lb+1) curr := make([]int, lb+1) for j := 0; j <= lb; j++ { prev[j] = j } for i := 1; i <= la; i++ { curr[0] = i for j := 1; j <= lb; j++ { cost := 1 if a[i-1] == b[j-1] { cost = 0 } curr[j] = min3(curr[j-1]+1, prev[j]+1, prev[j-1]+cost) } prev, curr = curr, prev } return prev[lb] } func min3(a, b, c int) int { if a < b { if a < c { return a } return c } if b < c { return b } return c }