// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "errors" "fmt" "math" "math/bits" "slices" "strconv" "strings" "sourcedock.dev/petrbalvin/gasm-sdk/ast" ) // assembleRISCV assembles a RISC-V TEXT function body into machine code. // It handles the full RV64IMAFDC instruction set including RVC compression. func assembleRISCV(t *ast.Text, tlsSyms map[string]bool) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, []RiscvLiteral, error) { fi := riscvComputeFrame(t) prologue := riscvPrologue(fi) guardLen, err := riscvGuardLen(fi) if err != nil { return nil, nil, nil, nil, nil, nil, err } lits := &riscvLiterals{} var relocs []Reloc var spadj []SpadjStep // The prologue raises the SP delta by autosize; the boundary is reported // at the pc just past its ADDI, exactly as the toolchain's pctospadj does. // The guard prefix shifts its PC. if fi.autosize != 0 { spadj = append(spadj, SpadjStep{PC: guardLen + riscvPrologueSpadjPC(fi), Value: fi.autosize}) } // Pass 1: collect instructions and compute label offsets assuming 4 bytes // per instruction (or 8 for MOV $large-imm). No encoding yet. PCALIGN // contributes only its padding, which is attached to the following // instruction and emitted ahead of it. A relaxed branch carries the // inverted condition and is followed by an inserted JMP rec (jmpTo set) // that carries the original target. type instrRec struct { instr *ast.Instr compressed bool code []byte pad int relaxed bool jmpTo string } var recs []instrRec offsets := map[string]int{} pos := guardLen + len(prologue) pendingPad := 0 for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: offsets[s.Name.Text] = pos case *ast.Instr: if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" { pendingPad += riscvPCAlignPad(pos, s) pos += riscvPCAlignPad(pos, s) continue } recs = append(recs, instrRec{instr: s, pad: pendingPad}) pendingPad = 0 pos += riscvInstrSize(s, fi, tlsSyms) } } // Pass 2: encode each instruction using Pass-1 offsets. A branch or // jump the offsets prove overlong encodes to a placeholder of the // instruction's own size: the relaxation pass rewrites it before the // final encoding. pcRelPcs is unavailable this early, so the N(PC) // forms take the same placeholder path. pc := len(prologue) for i := range recs { branchLike := isBranchLike(recs[i].instr.Mnemonic.Text) || riscvIsCondBranch(recs[i].instr.Mnemonic.Text) code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil, nil, lits, tlsSyms) // no relocs in Pass 2 if err != nil && !(branchLike && riscvIsRangeError(err)) { return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err) } if err != nil { code = make([]byte, riscvInstrSize(recs[i].instr, fi, tlsSyms)) } recs[i].code = code pc += len(code) } // Pass 3: try RVC compression. for i := range recs { if c16, ok := tryCompressRVC(recs[i].instr, fi); ok { recs[i].compressed = true recs[i].code = []byte{byte(c16), byte(c16 >> 8)} } } // Pass 4: recompute offsets with actual sizes. recs holds the // instructions in emission order, so an index into it walks t.Body in // lockstep (the same single pass Pass 1 uses) instead of rescanning the // whole slice per statement. PCALIGN padding is recomputed here, since // compression has shifted instruction sizes since Pass 1. offsets = map[string]int{} pos = guardLen + len(prologue) ri := 0 pendingPad = 0 for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: offsets[s.Name.Text] = pos case *ast.Instr: if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" { pad := riscvPCAlignPad(pos, s) pendingPad += pad pos += pad continue } recs[ri].pad = pendingPad pendingPad = 0 pos += len(recs[ri].code) ri++ } } // Pass 4b: relax overlong conditional branches exactly as the toolchain // does: invert the branch condition, point it at the instruction after an // inserted JMP, let the JMP carry the original target, and re-layout until // a pass inserts nothing. Inserted JMP recs share their branch's source // line and trail it in emission order, so the body walk flushes them // before every statement and at the end. var pcRelPcs map[*ast.Instr]int for { offsets = map[string]int{} pos = guardLen + len(prologue) ri := 0 pcs := make([]int, len(recs)) flushJmps := func() { for ri < len(recs) && recs[ri].jmpTo != "" { pcs[ri] = pos pos += 4 ri++ } } for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: flushJmps() offsets[s.Name.Text] = pos case *ast.Instr: flushJmps() if ri >= len(recs) { continue } pcs[ri] = pos + recs[ri].pad pos += recs[ri].pad + len(recs[ri].code) ri++ } } flushJmps() changed := false for i := range recs { r := &recs[i] if r.relaxed || r.jmpTo != "" { continue } mnem := strings.ToUpper(r.instr.Mnemonic.Text) if !riscvIsCondBranch(mnem) || len(r.instr.Operands) == 0 { continue } target := labelFromOperand(r.instr.Operands[len(r.instr.Operands)-1]) targetOff, ok := offsets[target] if !ok { continue } if delta := int32(targetOff - pcs[i]); delta < -4096 || delta >= 4096 { r.relaxed = true recs = slices.Insert(recs, i+1, instrRec{instr: r.instr, jmpTo: target}) changed = true } } if !changed { // Capture the final pcs for the N(PC) branch and jump forms: the // target is the instruction N source slots away (N=0 the branch // itself, N negative backwards), resolved by index against the // final layout. pcRelPcs = map[*ast.Instr]int{} for i := range recs { n, ok := riscvPCRelOffset(recs[i].instr) if !ok { continue } if i+n < 0 || i+n >= len(recs) { continue } pcRelPcs[recs[i].instr] = pcs[i+n] } break } } // Pass 5: re-encode branches with corrected offsets. Record relocations // during this final pass (relocation offsets are relative to instruction // start). The guard prefix precedes the prologue; its branches target // the morestack block at the end of the function, which the previous // passes have sized. var out []byte guardBytes, guardReloc, err := riscvGuard(fi) if err != nil { return nil, nil, nil, nil, nil, nil, err } if fi.needSplit { out = append(out, guardBytes...) } out = append(out, prologue...) pc = guardLen + len(prologue) preCount := len(relocs) var lines []LineEntry for _, r := range recs { // PCALIGN padding precedes the instruction it was attached to. if r.pad > 0 { out = append(out, riscvPadBytes(r.pad)...) pc += r.pad } lines = append(lines, LineEntry{Offset: pc, Line: r.instr.Pos().Line}) var code []byte switch { case r.jmpTo != "": // The JMP a relaxation inserted: JAL X0 to the original target. targetOff, ok := offsets[r.jmpTo] if !ok { return nil, nil, nil, nil, nil, nil, fmt.Errorf("undefined label %q", r.jmpTo) } offset := int32(targetOff - pc) if err := riscvCheckJumpOffset(r.jmpTo, offset); err != nil { return nil, nil, nil, nil, nil, nil, err } word := riscvJType(0, offset) code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)} case r.relaxed: // The inverted half of a relaxed branch: it targets the inserted // JMP, always the very next instruction (offset 4). enc, rs1, rs2, ok := riscvInvertedBranchEnc(strings.ToUpper(r.instr.Mnemonic.Text), r.instr.Operands) if !ok { return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: cannot relax branch", r.instr.Mnemonic.Text) } word := riscvBType(enc, rs1, rs2, 4) code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)} case r.compressed && !isBranchLike(r.instr.Mnemonic.Text): code = r.code default: var err error code, err = encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs, pcRelPcs, lits, tlsSyms) if err != nil { return nil, nil, nil, nil, nil, nil, err } if c16, ok := tryCompressRVC(r.instr, fi); ok { code = []byte{byte(c16), byte(c16 >> 8)} } // Make newly added relocation offsets function-relative. Each // instruction records its reloc offset relative to its own start; // the current pc is that instruction's offset from the function // start (which includes the prologue). After is the address just // past the relocated field, shifted by the same amount. for j := preCount; j < len(relocs); j++ { relocs[j].Off += pc relocs[j].After += pc } preCount = len(relocs) // The RET's epilogue closes the frame: the SP delta returns to zero // after its ADDI (restore LR + ADDI). if strings.ToUpper(r.instr.Mnemonic.Text) == "RET" && fi.autosize != 0 { spadj = append(spadj, SpadjStep{PC: pc + riscvReturnEpilogueLen(fi), Value: 0}) } } out = append(out, code...) pc += len(code) } if fi.needSplit { relocs = append(relocs, guardReloc) } return out, offsets, relocs, lines, spadj, lits.list(), nil } // riscvImmAlias maps the R-type ALU mnemonics onto their I-type immediate // forms: the toolchain accepts ADD $imm, rj, rd and emits addi. Applied // whenever the first operand is an immediate. var riscvImmAlias = map[string]string{ "ADD": "ADDI", "ADDW": "ADDIW", "AND": "ANDI", "OR": "ORI", "XOR": "XORI", "SLT": "SLTI", "SLTU": "SLTIU", "SLL": "SLLI", "SRL": "SRLI", "SRA": "SRAI", "SLLW": "SLLIW", "SRLW": "SRLIW", "SRAW": "SRAIW", } // riscvNormaliseImmAlias rewrites the mnemonic to its immediate form when the // first operand is an immediate: the toolchain accepts ADD $imm, rj, rd and // emits addi, and SUB $imm becomes addi with the negated immediate. The // second result reports that negation; the operand itself is left untouched // because several passes normalise the same instruction. func riscvNormaliseImmAlias(mnem string, ops []*ast.Operand) (string, bool) { if len(ops) >= 2 && isImmOperand(ops[0]) { switch strings.ToUpper(mnem) { case "SUB": return "ADDI", true case "SUBW": return "ADDIW", true } if alias, ok := riscvImmAlias[strings.ToUpper(mnem)]; ok { return alias, false } } return mnem, false } // riscvPCAlignPad returns the padding PCALIGN inserts before the next // instruction so that it starts at the requested boundary relative to the // function start. The boundary must be a power of two between 8 and 2048, as // the toolchain requires; anything else pads nothing. func riscvPCAlignPad(pos int, instr *ast.Instr) int { if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) { return 0 } align := int(immFromOperand(instr.Operands[0])) if align < 8 || align > 2048 || align&(align-1) != 0 { return 0 } return (align - pos%align) % align } // riscvPadBytes renders PCALIGN padding: 4-byte NOPs (addi $0, X0, X0) with a // trailing 2-byte compressed NOP when the pad is 2 mod 4, exactly as the // toolchain lays the bytes down. func riscvPadBytes(pad int) []byte { out := make([]byte, 0, pad) for ; pad >= 4; pad -= 4 { out = append(out, 0x13, 0x00, 0x00, 0x00) } if pad == 2 { out = append(out, 0x01, 0x00) } return out } // riscvFenceFlags carries the toolchain's sixteen FENCE flag spellings (its // specialOperands table between SPOP_FENCE_BEGIN and SPOP_FENCE_END), each // with the 4-bit encoding the predecessor and successor fields pack. var riscvFenceFlags = map[string]uint32{ "W": 1, "R": 2, "RW": 3, "O": 4, "OW": 5, "OR": 6, "ORW": 7, "I": 8, "IW": 9, "IR": 10, "IRW": 11, "IO": 12, "IOW": 13, "IOR": 14, "IORW": 15, } // riscvFenceFlag resolves one FENCE flag operand (a bare name such as W or // IORW) to its 4-bit encoding. func riscvFenceFlag(op *ast.Operand) (uint32, bool) { name := "" if op.Addr.Sym != nil { name = op.Addr.Sym.Name } v, ok := riscvFenceFlags[strings.ToUpper(name)] return v, ok } // riscvInstrSize returns the encoded size in bytes of a RISC-V instruction. // Most instructions are 4 bytes; MOV with a large immediate and I-type // arithmetic with a large immediate expand to several (possibly compressed) // instructions. func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo, tlsSyms map[string]bool) int { mnem := instr.Mnemonic.Text ops := instr.Operands mnem = riscvNormalisePseudo(mnem) if mnem == "FUNCDATA" || mnem == "PCDATA" || mnem == "END" { // The bookkeeping statements and the function-end marker contribute // no bytes. return 0 } if mnem == "GETCALLERPC" { return riscvGetCallerPCSize(ops, fi) } var immNeg bool mnem, immNeg = riscvNormaliseImmAlias(mnem, ops) if mnem == "RET" { return len(riscvReturn(fi)) } if strings.HasPrefix(mnem, "MOV") && len(ops) == 2 { // FP constant: FMV from X0 for a zero bit pattern, otherwise the // 8-byte AUIPC + FLW/FLD pool load. if (mnem == "MOVF" || mnem == "MOVD") && isImmOperand(ops[0]) && !ops[0].Imm.HasVal && ops[0].Imm.Sym == nil && ops[0].Imm.Str == "" { if pattern, _, err := riscvFPConstBits(mnem, ops[0]); err == nil { if pattern == 0 { return 4 } return 8 } } // A TLSBSS symbol's memory reference takes the 16-byte local-exec // sequence (LUI + ADDIW + ADD of TP + the access). tlsRef := func(op *ast.Operand) bool { return op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" && tlsSyms[op.Addr.Sym.Name] } if tlsRef(ops[0]) || tlsRef(ops[1]) { return 16 } // MOV $sym(SB), rd → 8 bytes (AUIPC + ADDI). if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && ops[0].Imm.Sym.Pseudo == "SB" { return 8 } // MOV sym(SB), rd → 8 bytes (AUIPC + LD). if isMemOperand(ops[0]) && ops[0].Addr.Sym != nil && ops[0].Addr.Sym.Pseudo == "SB" { return 8 } // MOV rd, sym(SB) → 8 bytes (AUIPC + SD). if isMemOperand(ops[1]) && ops[1].Addr.Sym != nil && ops[1].Addr.Sym.Pseudo == "SB" { return 8 } // MOV $imm, rd → size depends on the immediate and RVC compression. if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil { imm := riscvOperandImm64(ops[0]) if int64(int32(imm)) != imm { return riscvMovImm64Size(regFromOperand(ops[1]), imm) } return riscvMovImmSize(regFromOperand(ops[1]), int32(imm)) } // MOV $sym+off(FP|SP), rd → the frame-adjusted offset as an ADDI, // compressed like riscvSPAddiBytes encodes it. if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && (ops[0].Imm.Sym.Pseudo == "FP" || ops[0].Imm.Sym.Pseudo == "SP") { rd := regFromOperand(ops[1]) _, off := riscvResolvePseudo(ops[0].Imm.Sym, fi) if rd > 0 && off == 0 { return 2 // C.MV rd, SP } if isRVCIntReg(rd) && off > 0 && off < 1024 && off%4 == 0 { return 2 // C.ADDI4SPN } return riscvItypeImmediateSize("ADDI", rd, 2, off) } // Frame-relative loads and stores: a frame offset beyond the signed // 12-bit range materialises the address in X31 first. if isMemOperand(ops[0]) && !isMemOperand(ops[1]) { return riscvFrameMemSize(ops[0], fi) } if isMemOperand(ops[1]) && !isMemOperand(ops[0]) { return riscvFrameMemSize(ops[1], fi) } // Register-to-register width moves: the SLLI + SRAI/SRLI extension // pairs carry their per-half compression; the single-word forms are // four bytes before the RVC pass (bare MOV compresses to C.MV or // C.LI, which the post-encoding pass resolves from real bytes). if !isMemOperand(ops[0]) && !isMemOperand(ops[1]) { rd := regFromOperand(ops[1]) rs1 := regFromOperand(ops[0]) switch mnem { case "MOVB", "MOVH": shamt := 56 if mnem == "MOVH" { shamt = 48 } return len(riscvExtendBytes(rd, rs1, shamt, true)) case "MOVHU", "MOVWU": shamt := 48 if mnem == "MOVWU" { shamt = 32 } return len(riscvExtendBytes(rd, rs1, shamt, false)) } return 4 } } // Plain loads and stores whose offset leaves the signed 12-bit span // expand to the X31 materialisation plus the access word. if n, ok := riscvMemInstrSize(mnem, ops, fi); ok { return n } // I-type arithmetic with a large immediate expands to several instructions. if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) { imm := immFromOperand(ops[0]) if immNeg { imm = -imm } rd, rs1 := -1, -1 switch len(ops) { case 3: rs1 = regFromOperand(ops[1]) rd = regFromOperand(ops[2]) case 2: rd = regFromOperand(ops[1]) rs1 = rd } return riscvItypeImmediateSize(mnem, rd, rs1, imm) } // BYTE lays down one raw byte per operand. if mnem == "BYTE" { return len(ops) } // The toolchain's synthesised instructions: some emit one word, others // expand to a fixed sequence. return riscvExtendedSize(mnem, ops) } // riscvExtendedSize returns the encoded size of the instructions the // toolchain synthesises from other instructions (the ternary expansions and // the vector slice); every caller keeps the layout in step with // encodeRISCVExtended, which emits exactly these bytes. func riscvExtendedSize(mnem string, ops []*ast.Operand) int { switch mnem { case "NOP": // The toolchain drops a bare NOP entirely. return 0 case "ANDN", "ORN", "XNOR", "FNES", "FNED": return 8 case "MAX", "MAXU", "MIN", "MINU": if riscvIdenticalMinMax(mnem, ops) { rd := regFromOperand(ops[1]) if len(ops) == 3 { rd = regFromOperand(ops[2]) } if rd != 0 { return 2 // C.MV, or C.LI when the sources are X0 } return 4 } return 20 case "ROL", "ROLW", "ROR", "RORI", "RORW": if len(ops) >= 1 && isImmOperand(ops[0]) { // SRL + [compressed] SLL of the reverse shift + OR. return 4 + riscvRevShiftSize(mnem, ops) + 4 } return 16 // SUB + shift + shift + OR case "RORIW": return 12 } if isRVCInstr(mnem) { return 2 } return 4 } // riscvIdenticalMinMax reports whether a MIN/MAX sees two identical source // registers (the toolchain folds that to ADDI $0). func riscvIdenticalMinMax(mnem string, ops []*ast.Operand) bool { if mnem != "MAX" && mnem != "MAXU" && mnem != "MIN" && mnem != "MINU" { return false } if len(ops) != 2 && len(ops) != 3 { return false } rs1 := regFromOperand(ops[1]) rs2 := regFromOperand(ops[0]) rd := rs1 if len(ops) == 3 { rd = regFromOperand(ops[2]) } if rs1 == rd { // The toolchain swaps the sources so the destination-identical one // is processed first; identical sources stay identical. rs1, rs2 = rs2, rs1 } return rs1 >= 0 && rs1 == rs2 } // riscvRevShiftSize returns the size of the reverse-shift instruction inside // a ROR/RORI immediate expansion: the SLLI of the complementary amount, which // compresses to C.SLLI only in the 64-bit form when rd == rs1, both non-zero, // and the amount lands in 1-63. The W forms have no compressed shift. func riscvRevShiftSize(mnem string, ops []*ast.Operand) int { if mnem != "ROR" && mnem != "RORI" { return 4 // SLLIW has no compressed form } if len(ops) < 2 { // A malformed one-operand form: encoding rejects it with a // diagnostic, and the layout pass only needs a word count. return 4 } imm := int(immFromOperand(ops[0])) rs1 := regFromOperand(ops[1]) rd := rs1 if len(ops) == 3 { rd = regFromOperand(ops[2]) } sll := (-imm) & 63 if rd == rs1 && rd != 0 && sll >= 1 && sll <= 63 { return 2 // C.SLLI } return 4 } // isBranchLike reports whether a mnemonic is a branch or jump that needs // recalculated offsets after compression. func isBranchLike(mnem string) bool { switch mnem { case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "JMP", "JAL", "CJ", "CBEQZ", "CBNEZ": return true } return false } // riscvIsCondBranch reports whether m is a conditional branch, the only // instruction class branch relaxation rewrites. func riscvIsCondBranch(mnem string) bool { switch mnem { case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU", "BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ": return true } return false } // riscvCSRNames maps every CSR mnemonic the assembler accepts onto its // address: the RISC-V privileged specification's register set as the Go // toolchain spells it, so a name `go tool asm` reads resolves here too. var riscvCSRNames = map[string]int32{ "FFLAGS": 0x001, "FRM": 0x002, "FCSR": 0x003, "UTVT": 0x007, "VSTART": 0x008, "VXSAT": 0x009, "VXRM": 0x00A, "VCSR": 0x00F, "SSP": 0x011, "SEED": 0x015, "JVT": 0x017, "UNXTI": 0x045, "UINTSTATUS": 0x046, "USCRATCHCSW": 0x048, "USCRATCHCSWL": 0x049, "SSTATUS": 0x100, "SIE": 0x104, "STVEC": 0x105, "SCOUNTEREN": 0x106, "STVT": 0x107, "SENVCFG": 0x10A, "SSTATEEN0": 0x10C, "SSTATEEN1": 0x10D, "SSTATEEN2": 0x10E, "SSTATEEN3": 0x10F, "SCOUNTINHIBIT": 0x120, "SSCRATCH": 0x140, "SEPC": 0x141, "SCAUSE": 0x142, "STVAL": 0x143, "SIP": 0x144, "SNXTI": 0x145, "SINTSTATUS": 0x146, "SSCRATCHCSW": 0x148, "SSCRATCHCSWL": 0x149, "STIMECMP": 0x14D, "SCTRCTL": 0x14E, "SCTRSTATUS": 0x14F, "SISELECT": 0x150, "SIREG": 0x151, "SIREG2": 0x152, "SIREG3": 0x153, "SIREG4": 0x155, "SIREG5": 0x156, "SIREG6": 0x157, "STOPEI": 0x15C, "SCTRDEPTH": 0x15F, "SATP": 0x180, "SRMCFG": 0x181, "SPMPEN": 0x183, "VSSTATUS": 0x200, "VSIE": 0x204, "VSTVEC": 0x205, "VSSCRATCH": 0x240, "VSEPC": 0x241, "VSCAUSE": 0x242, "VSTVAL": 0x243, "VSIP": 0x244, "VSTIMECMP": 0x24D, "VSCTRCTL": 0x24E, "VSISELECT": 0x250, "VSIREG": 0x251, "VSIREG2": 0x252, "VSIREG3": 0x253, "VSIREG4": 0x255, "VSIREG5": 0x256, "VSIREG6": 0x257, "VSTOPEI": 0x25C, "VSATP": 0x280, "MSTATUS": 0x300, "MISA": 0x301, "MEDELEG": 0x302, "MIDELEG": 0x303, "MIE": 0x304, "MTVEC": 0x305, "MCOUNTEREN": 0x306, "MTVT": 0x307, "MVIEN": 0x308, "MVIP": 0x309, "MENVCFG": 0x30A, "MSTATEEN0": 0x30C, "MSTATEEN1": 0x30D, "MSTATEEN2": 0x30E, "MSTATEEN3": 0x30F, "MPMPDELEG": 0x316, "MCOUNTINHIBIT": 0x320, "MCYCLECFG": 0x321, "MINSTRETCFG": 0x322, "MHPMEVENT3": 0x323, "MHPMEVENT4": 0x324, "MHPMEVENT5": 0x325, "MHPMEVENT6": 0x326, "MHPMEVENT7": 0x327, "MHPMEVENT8": 0x328, "MHPMEVENT9": 0x329, "MHPMEVENT10": 0x32A, "MHPMEVENT11": 0x32B, "MHPMEVENT12": 0x32C, "MHPMEVENT13": 0x32D, "MHPMEVENT14": 0x32E, "MHPMEVENT15": 0x32F, "MHPMEVENT16": 0x330, "MHPMEVENT17": 0x331, "MHPMEVENT18": 0x332, "MHPMEVENT19": 0x333, "MHPMEVENT20": 0x334, "MHPMEVENT21": 0x335, "MHPMEVENT22": 0x336, "MHPMEVENT23": 0x337, "MHPMEVENT24": 0x338, "MHPMEVENT25": 0x339, "MHPMEVENT26": 0x33A, "MHPMEVENT27": 0x33B, "MHPMEVENT28": 0x33C, "MHPMEVENT29": 0x33D, "MHPMEVENT30": 0x33E, "MHPMEVENT31": 0x33F, "MSCRATCH": 0x340, "MEPC": 0x341, "MCAUSE": 0x342, "MTVAL": 0x343, "MIP": 0x344, "MNXTI": 0x345, "MINTSTATUS": 0x346, "MSCRATCHCSW": 0x348, "MSCRATCHCSWL": 0x349, "MTINST": 0x34A, "MTVAL2": 0x34B, "MCTRCTL": 0x34E, "MISELECT": 0x350, "MIREG": 0x351, "MIREG2": 0x352, "MIREG3": 0x353, "MIREG4": 0x355, "MIREG5": 0x356, "MIREG6": 0x357, "MTOPEI": 0x35C, "PMPCFG0": 0x3A0, "PMPCFG1": 0x3A1, "PMPCFG2": 0x3A2, "PMPCFG3": 0x3A3, "PMPCFG4": 0x3A4, "PMPCFG5": 0x3A5, "PMPCFG6": 0x3A6, "PMPCFG7": 0x3A7, "PMPCFG8": 0x3A8, "PMPCFG9": 0x3A9, "PMPCFG10": 0x3AA, "PMPCFG11": 0x3AB, "PMPCFG12": 0x3AC, "PMPCFG13": 0x3AD, "PMPCFG14": 0x3AE, "PMPCFG15": 0x3AF, "PMPADDR0": 0x3B0, "PMPADDR1": 0x3B1, "PMPADDR2": 0x3B2, "PMPADDR3": 0x3B3, "PMPADDR4": 0x3B4, "PMPADDR5": 0x3B5, "PMPADDR6": 0x3B6, "PMPADDR7": 0x3B7, "PMPADDR8": 0x3B8, "PMPADDR9": 0x3B9, "PMPADDR10": 0x3BA, "PMPADDR11": 0x3BB, "PMPADDR12": 0x3BC, "PMPADDR13": 0x3BD, "PMPADDR14": 0x3BE, "PMPADDR15": 0x3BF, "PMPADDR16": 0x3C0, "PMPADDR17": 0x3C1, "PMPADDR18": 0x3C2, "PMPADDR19": 0x3C3, "PMPADDR20": 0x3C4, "PMPADDR21": 0x3C5, "PMPADDR22": 0x3C6, "PMPADDR23": 0x3C7, "PMPADDR24": 0x3C8, "PMPADDR25": 0x3C9, "PMPADDR26": 0x3CA, "PMPADDR27": 0x3CB, "PMPADDR28": 0x3CC, "PMPADDR29": 0x3CD, "PMPADDR30": 0x3CE, "PMPADDR31": 0x3CF, "PMPADDR32": 0x3D0, "PMPADDR33": 0x3D1, "PMPADDR34": 0x3D2, "PMPADDR35": 0x3D3, "PMPADDR36": 0x3D4, "PMPADDR37": 0x3D5, "PMPADDR38": 0x3D6, "PMPADDR39": 0x3D7, "PMPADDR40": 0x3D8, "PMPADDR41": 0x3D9, "PMPADDR42": 0x3DA, "PMPADDR43": 0x3DB, "PMPADDR44": 0x3DC, "PMPADDR45": 0x3DD, "PMPADDR46": 0x3DE, "PMPADDR47": 0x3DF, "PMPADDR48": 0x3E0, "PMPADDR49": 0x3E1, "PMPADDR50": 0x3E2, "PMPADDR51": 0x3E3, "PMPADDR52": 0x3E4, "PMPADDR53": 0x3E5, "PMPADDR54": 0x3E6, "PMPADDR55": 0x3E7, "PMPADDR56": 0x3E8, "PMPADDR57": 0x3E9, "PMPADDR58": 0x3EA, "PMPADDR59": 0x3EB, "PMPADDR60": 0x3EC, "PMPADDR61": 0x3ED, "PMPADDR62": 0x3EE, "PMPADDR63": 0x3EF, "SCONTEXT": 0x5A8, "HSTATUS": 0x600, "HEDELEG": 0x602, "HIDELEG": 0x603, "HIE": 0x604, "HTIMEDELTA": 0x605, "HCOUNTEREN": 0x606, "HGEIE": 0x607, "HVIEN": 0x608, "HVICTL": 0x609, "HENVCFG": 0x60A, "HSTATEEN0": 0x60C, "HSTATEEN1": 0x60D, "HSTATEEN2": 0x60E, "HSTATEEN3": 0x60F, "HTVAL": 0x643, "HIP": 0x644, "HVIP": 0x645, "HVIPRIO1": 0x646, "HVIPRIO2": 0x647, "HTINST": 0x64A, "HGATP": 0x680, "HCONTEXT": 0x6A8, "MSECCFG": 0x747, "TSELECT": 0x7A0, "TDATA1": 0x7A1, "TDATA2": 0x7A2, "TDATA3": 0x7A3, "TINFO": 0x7A4, "TCONTROL": 0x7A5, "MCONTEXT": 0x7A8, "MSCONTEXT": 0x7AA, "DCSR": 0x7B0, "DPC": 0x7B1, "DSCRATCH0": 0x7B2, "DSCRATCH1": 0x7B3, "MCYCLE": 0xB00, "MINSTRET": 0xB02, "MHPMCOUNTER3": 0xB03, "MHPMCOUNTER4": 0xB04, "MHPMCOUNTER5": 0xB05, "MHPMCOUNTER6": 0xB06, "MHPMCOUNTER7": 0xB07, "MHPMCOUNTER8": 0xB08, "MHPMCOUNTER9": 0xB09, "MHPMCOUNTER10": 0xB0A, "MHPMCOUNTER11": 0xB0B, "MHPMCOUNTER12": 0xB0C, "MHPMCOUNTER13": 0xB0D, "MHPMCOUNTER14": 0xB0E, "MHPMCOUNTER15": 0xB0F, "MHPMCOUNTER16": 0xB10, "MHPMCOUNTER17": 0xB11, "MHPMCOUNTER18": 0xB12, "MHPMCOUNTER19": 0xB13, "MHPMCOUNTER20": 0xB14, "MHPMCOUNTER21": 0xB15, "MHPMCOUNTER22": 0xB16, "MHPMCOUNTER23": 0xB17, "MHPMCOUNTER24": 0xB18, "MHPMCOUNTER25": 0xB19, "MHPMCOUNTER26": 0xB1A, "MHPMCOUNTER27": 0xB1B, "MHPMCOUNTER28": 0xB1C, "MHPMCOUNTER29": 0xB1D, "MHPMCOUNTER30": 0xB1E, "MHPMCOUNTER31": 0xB1F, "CYCLE": 0xC00, "TIME": 0xC01, "INSTRET": 0xC02, "HPMCOUNTER3": 0xC03, "HPMCOUNTER4": 0xC04, "HPMCOUNTER5": 0xC05, "HPMCOUNTER6": 0xC06, "HPMCOUNTER7": 0xC07, "HPMCOUNTER8": 0xC08, "HPMCOUNTER9": 0xC09, "HPMCOUNTER10": 0xC0A, "HPMCOUNTER11": 0xC0B, "HPMCOUNTER12": 0xC0C, "HPMCOUNTER13": 0xC0D, "HPMCOUNTER14": 0xC0E, "HPMCOUNTER15": 0xC0F, "HPMCOUNTER16": 0xC10, "HPMCOUNTER17": 0xC11, "HPMCOUNTER18": 0xC12, "HPMCOUNTER19": 0xC13, "HPMCOUNTER20": 0xC14, "HPMCOUNTER21": 0xC15, "HPMCOUNTER22": 0xC16, "HPMCOUNTER23": 0xC17, "HPMCOUNTER24": 0xC18, "HPMCOUNTER25": 0xC19, "HPMCOUNTER26": 0xC1A, "HPMCOUNTER27": 0xC1B, "HPMCOUNTER28": 0xC1C, "HPMCOUNTER29": 0xC1D, "HPMCOUNTER30": 0xC1E, "HPMCOUNTER31": 0xC1F, "VL": 0xC20, "VTYPE": 0xC21, "VLENB": 0xC22, "SCOUNTOVF": 0xDA0, "STOPI": 0xDB0, "HGEIP": 0xE12, "VSTOPI": 0xEB0, "MVENDORID": 0xF11, "MARCHID": 0xF12, "MIMPID": 0xF13, "MHARTID": 0xF14, "MCONFIGPTR": 0xF15, "MTOPI": 0xFB0, } // riscvCSRAddress resolves a CSR operand: an integer immediate or one of the // standard CSR names. func riscvCSRAddress(op *ast.Operand) (int32, bool) { if isImmOperand(op) { return immFromOperand(op), true } if op.Addr.Sym != nil { if v, ok := riscvCSRNames[strings.ToUpper(op.Addr.Sym.Name)]; ok { return v, true } } return 0, false } // riscvPCRelOffset reports the N of a branch or jump operand spelled N(PC): // the displacement counted in source instructions from the branch itself. func riscvPCRelOffset(instr *ast.Instr) (int, bool) { mnem := strings.ToUpper(instr.Mnemonic.Text) switch mnem { case "JMP": if len(instr.Operands) != 1 { return 0, false } case "JAL": if len(instr.Operands) != 1 && len(instr.Operands) != 2 { return 0, false } case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU", "BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ": if len(instr.Operands) < 2 { return 0, false } case "CJ": if len(instr.Operands) != 1 { return 0, false } case "CBEQZ", "CBNEZ": if len(instr.Operands) != 2 { return 0, false } default: return 0, false } op := instr.Operands[len(instr.Operands)-1] if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" { return int(op.Addr.Offset), true } return 0, false } // riscvPCRelTargetOff resolves the target displacement of a branch whose last // operand is N(PC): the toolchain's parser counts the source instructions at // a uniform 4 bytes, so the target is the instruction N slots away, and the // displacement tracks that instruction's final pc. A nil pcRelPcs (the // layout passes) yields a placeholder range error; the caller tolerates it // for branch-like instructions. func riscvPCRelTargetOff(instr *ast.Instr, pc int, pcRelPcs map[*ast.Instr]int) (int, bool, error) { off, ok := riscvPCRelOffset(instr) if !ok { return 0, false, nil } if pcRelPcs == nil { return 0, true, &riscvRangeError{"pc-relative placeholder"} } targetPc, ok := pcRelPcs[instr] if !ok { return 0, true, fmt.Errorf("PC-relative target %d out of range", off) } return targetPc, true, nil } // riscvInvertedBranchEnc returns the encoding of mnem's inverted condition // for the given operands: InvertBranch's table applied at the encoding level. // The register operands are already in position for the inverted form. func riscvInvertedBranchEnc(mnem string, ops []*ast.Operand) (riscvEnc, int, int, bool) { reg := func(i int) int { return regFromOperand(ops[i]) } switch mnem { case "BEQ": // → BNE rs1, rs2 return riscvEnc{0x63, 0x1, 0x00}, reg(0), reg(1), true case "BNE": // → BEQ rs1, rs2 return riscvEnc{0x63, 0x0, 0x00}, reg(0), reg(1), true case "BLT": // → BGE rs1, rs2 return riscvEnc{0x63, 0x5, 0x00}, reg(0), reg(1), true case "BGE": // → BLT rs1, rs2 return riscvEnc{0x63, 0x4, 0x00}, reg(0), reg(1), true case "BLTU": // → BGEU rs1, rs2 return riscvEnc{0x63, 0x7, 0x00}, reg(0), reg(1), true case "BGEU": // → BLTU rs1, rs2 return riscvEnc{0x63, 0x6, 0x00}, reg(0), reg(1), true case "BEQZ": // → BNEZ rs, X0 return riscvEnc{0x63, 0x1, 0x00}, reg(0), 0, true case "BNEZ": // → BEQZ rs, X0 return riscvEnc{0x63, 0x0, 0x00}, reg(0), 0, true case "BLTZ": // → BGEZ rs, X0 return riscvEnc{0x63, 0x5, 0x00}, reg(0), 0, true case "BGEZ": // → BLTZ rs, X0 return riscvEnc{0x63, 0x4, 0x00}, reg(0), 0, true case "BLEZ": // → BGTZ: blt X0, rs return riscvEnc{0x63, 0x4, 0x00}, 0, reg(0), true case "BGTZ": // → BLEZ: bge X0, rs return riscvEnc{0x63, 0x5, 0x00}, 0, reg(0), true case "BGT": // → BLE: bge rs2, rs1 return riscvEnc{0x63, 0x5, 0x00}, reg(1), reg(0), true case "BLE": // → BGT: blt rs2, rs1 return riscvEnc{0x63, 0x4, 0x00}, reg(1), reg(0), true case "BGTU": // → BLEU: bgeu rs2, rs1 return riscvEnc{0x63, 0x7, 0x00}, reg(1), reg(0), true case "BLEU": // → BGTU: bltu rs2, rs1 return riscvEnc{0x63, 0x6, 0x00}, reg(1), reg(0), true } return riscvEnc{}, 0, 0, false } // riscvRangeError reports a branch or jump displacement beyond its // architecture limit. The layout passes tolerate it (the relaxation pass // rewrites overlong conditional branches before the final encoding); a range // error reaching the final pass is a real failure. type riscvRangeError struct{ msg string } func (e *riscvRangeError) Error() string { return e.msg } // riscvIsRangeError reports whether err is a displacement-range rejection. func riscvIsRangeError(err error) bool { var re *riscvRangeError return errors.As(err, &re) } // riscvRoundModes maps the rounding-mode suffixes onto their funct7 codes. var riscvRoundModes = map[string]uint32{ "RNE": 0, "RTZ": 1, "RDN": 2, "RUP": 3, "RMM": 4, } // riscvCheckBranchOffset rejects a B-type displacement outside its signed // 13-bit span [-4096, 4094]; the encoder masks to 13 bits, so an // out-of-range offset would otherwise wrap to a wrong target. func riscvCheckBranchOffset(target string, off int32) error { if off < -4096 || off > 4094 { return &riscvRangeError{fmt.Sprintf("branch to %q too far (13-bit range)", target)} } return nil } // riscvCheckJumpOffset rejects a J-type displacement outside its signed // 21-bit span [-1048576, 1048574]. func riscvCheckJumpOffset(target string, off int32) error { if off < -1048576 || off > 1048574 { return &riscvRangeError{fmt.Sprintf("jump to %q too far (21-bit range)", target)} } return nil } // encodeRISCVInstr encodes a single RISC-V instruction. func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int, lits *riscvLiterals, tlsSyms map[string]bool) ([]byte, error) { mnem := instr.Mnemonic.Text ops := instr.Operands mnem = riscvNormalisePseudo(mnem) var immNeg bool mnem, immNeg = riscvNormaliseImmAlias(mnem, ops) var word uint32 // Handle pseudo-instructions and special cases first. switch mnem { case "RET": // RET = epilogue (restore LR and close the frame when present) + // uncompressed JALR X0, 0(X1) (the toolchain never compresses RET). return riscvReturn(fi), nil case "FUNCDATA": // The assembler's bookkeeping statement, the expanded form of the // GO_ARGS and NO_LOCAL_POINTERS macros: FUNCDATA $n, sym(SB) // contributes no bytes, exactly as the toolchain's listing shows // (the FUNCDATA entries and the instruction after them share a PC). if len(ops) != 2 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)") } return nil, nil case "PCDATA": // The other bookkeeping statement, the expanded form of // GO_RESULTS_INITIALIZED: PCDATA $n, $m contributes no bytes too. if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) { return nil, fmt.Errorf("PCDATA expects $n, $m") } return nil, nil case "END": // The function-end marker: the toolchain accepts it anywhere in a // body, ignores whatever operands follow it, and emits nothing. return nil, nil case "GETCALLERPC": return encodeRISCVGetCallerPC(ops, fi, relocs) case "WORD": // WORD $w lays down a raw 32-bit little-endian word, in the range // [0, 0xffffffff] exactly as the toolchain's validation bounds it. if len(ops) != 1 { return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) } w, ok := riscvRawImm(ops[0]) if !ok { return nil, fmt.Errorf("WORD expects an immediate") } if w < 0 || w > 0xFFFFFFFF { return nil, fmt.Errorf("WORD: immediate %d must be in range [0x0, 0xffffffff]", w) } return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}, nil case "BYTE": // BYTE $b lays down one raw byte per operand. var out []byte for _, op := range ops { b, ok := riscvRawImm(op) if !ok { return nil, fmt.Errorf("BYTE expects immediates") } if b < 0 || b > 0xFF { return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b) } out = append(out, byte(b)) } return out, nil case "CALL": // CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL // relocation. The Go assembler rejects CALL to a local branch label. if len(ops) != 1 { return nil, fmt.Errorf("CALL expects 1 operand, got %d", len(ops)) } op := ops[0] if op.Addr.Sym == nil || op.Addr.Sym.Pseudo != "SB" { // CALL (X5): an indirect call, the toolchain's JALR X1, 0(X5). if op.Addr.Sym == nil && op.Addr.Base != "" { if op.Addr.Offset != 0 || op.Addr.Index != "" { return nil, fmt.Errorf("CALL: invalid indirect operand %q", op.Raw) } rs1 := riscvRegNum(op.Addr.Base) if rs1 < 0 { return nil, fmt.Errorf("CALL: unknown branch register %q", op.Addr.Base) } word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 1, rs1, 0) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } return nil, fmt.Errorf("CALL: local branch target is not supported (use CALL sym(SB))") } if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelRISCVJal, Addend: op.Addr.Sym.Offset}) } word = riscvJType(1, 0) // JAL X1, 0, the linker fills the offset return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil case "JMP": // JMP = JAL X0, target. The Go assembler never compresses this to // C.J, so always emit the 32-bit JAL. var target string if len(ops) >= 1 { // JMP sym(SB): a tail call, JAL X0 against a symbol relocation. if ops[0].Addr.Sym != nil && ops[0].Addr.Sym.Pseudo == "SB" { if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: ops[0].Addr.Sym.Name, Kind: RelRISCVJal, Addend: ops[0].Addr.Sym.Offset}) } word = riscvJType(0, 0) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } target = labelFromOperand(ops[0]) // JMP N(PC): the PC-relative slot form, resolved like the // branches (the toolchain counts source instructions at a // uniform 4 bytes, so JMP 0(PC) is a self-loop and JMP -3(PC) // reaches twelve bytes back). It must be recognised before the // indirect-register form, whose operand it resembles. if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel { if err != nil { return nil, err } offset := int32(off - pc) if err := riscvCheckJumpOffset("", offset); err != nil { return nil, err } word = riscvJType(0, offset) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // JMP (X5) and JMP 4(X5): an indirect branch, the toolchain's // JALR X0, imm(X5) with the offset carried in the I-type // immediate (JMP 4(X5) is 0x67804200). if ops[0].Addr.Sym == nil && ops[0].Addr.Base != "" { if ops[0].Addr.Index != "" { return nil, fmt.Errorf("JMP: invalid indirect operand %q", ops[0].Raw) } rs1 := riscvRegNum(ops[0].Addr.Base) if rs1 < 0 { return nil, fmt.Errorf("JMP: unknown branch register %q", ops[0].Addr.Base) } imm := int32(ops[0].Addr.Offset) if imm < -2048 || imm > 2047 { return nil, fmt.Errorf("JMP: displacement %d does not fit in 12 bits", imm) } word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, rs1, imm) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } } targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } offset := int32(targetOff - pc) if err := riscvCheckJumpOffset(target, offset); err != nil { return nil, err } word = riscvJType(0, offset) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil case "JAL": rd := 0 var target string if len(ops) >= 2 { var err error if rd, err = riscvWantIntReg(mnem, "rd", ops[0]); err != nil { return nil, err } target = labelFromOperand(ops[1]) } else if len(ops) == 1 { target = labelFromOperand(ops[0]) } if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel { if err != nil { return nil, err } targetOff := off offset := int32(targetOff - pc) if err := riscvCheckJumpOffset(target, offset); err != nil { return nil, err } word = riscvJType(rd, offset) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } offset := int32(targetOff - pc) if err := riscvCheckJumpOffset(target, offset); err != nil { return nil, err } word = riscvJType(rd, offset) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil // MOV is a pseudo-instruction that the Go assembler uses for loads, // stores, register moves and immediate loads. The width suffixes // (MOVB/MOVH/MOVW and unsigned forms) select the access width, and // MOVD/MOVF address the FP registers. case "MOV", "MOVB", "MOVBU", "MOVH", "MOVHU", "MOVW", "MOVWU", "MOVF", "MOVD": return encodeRISCVMov(instr, fi, relocs, lits, tlsSyms) // JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1). case "JALR": return encodeRISCVJALR(instr, fi) // Branch-zero pseudos: BEQZ/BNEZ compare against X0, and BLTZ/BGEZ/ // BLEZ/BGTZ reorder the register operands of BLT/BGE accordingly. case "BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ": if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rs, err := riscvWantIntReg(mnem, "rs1", ops[0]) if err != nil { return nil, err } targetOff := 0 target := "" if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel { if err != nil { return nil, err } targetOff = off } else { target = labelFromOperand(ops[1]) var ok bool targetOff, ok = offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } } var enc riscvEnc rs1, rs2 := rs, 0 switch mnem { case "BEQZ": enc = riscvEnc{0x63, 0x0, 0x00} // beq rs, x0 case "BNEZ": enc = riscvEnc{0x63, 0x1, 0x00} // bne rs, x0 case "BLTZ": enc = riscvEnc{0x63, 0x4, 0x00} // blt rs, x0 case "BGEZ": enc = riscvEnc{0x63, 0x5, 0x00} // bge rs, x0 case "BLEZ": enc, rs1, rs2 = riscvEnc{0x63, 0x5, 0x00}, 0, rs // bge x0, rs case "BGTZ": enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, 0, rs // blt x0, rs } if err := riscvCheckBranchOffset(target, int32(targetOff-pc)); err != nil { return nil, err } word = riscvBType(enc, rs1, rs2, int32(targetOff-pc)) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil // System instructions with no operands. case "FENCE", "ECALL", "EBREAK", "FENCE.TSO", "PAUSE": enc, ok := riscvInstrTable[mnem] if !ok { return nil, fmt.Errorf("unsupported system instruction %q", mnem) } // The bare FENCE expands to fence iorw, iorw: the predecessor and // successor fields both carry 0xF in the I-type immediate // (the toolchain's encodeFenceOperand TYPE_NONE default). FENCE.TSO // carries the TSO fence mode with RW predecessor and successor. imm := int32(0) if mnem == "FENCE" { imm = 0x0FF } if mnem == "FENCE.TSO" { imm = 0x833 } if mnem == "PAUSE" { imm = 0x010 } if mnem == "FENCE.TSO" && len(ops) != 0 { return nil, fmt.Errorf("FENCE.TSO must not have operands") } // FENCE pred, succ spells both flags with the toolchain's sixteen // IORW combinations, packed as pred<<4 | succ in the immediate. if mnem == "FENCE" && len(ops) > 0 { if len(ops) != 2 { return nil, fmt.Errorf("FENCE expects 0 or 2 operands, got %d", len(ops)) } pred, ok := riscvFenceFlag(ops[0]) if !ok { return nil, fmt.Errorf("invalid FENCE predecessor operand %q", ops[0].Raw) } succ, ok := riscvFenceFlag(ops[1]) if !ok { return nil, fmt.Errorf("invalid FENCE successor operand %q", ops[1].Raw) } imm = int32(pred)<<4 | int32(succ) } word = riscvIType(enc, 0, 0, imm) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil case "SRET", "MRET", "WFI", "DRET": // The privileged traps and the wait instruction: fixed funct7 and // rs2 fields packed into the I-type immediate. The toolchain's // object table carries the first three (its assembler accepts no // mnemonic for them); DRET is the debug specification's own, so the // golden vector pins it: SRET 0x10200073, MRET 0x30200073, // WFI 0x10500073, DRET 0x7b200073. imm := map[string]int32{"SRET": 0x102, "MRET": 0x302, "WFI": 0x105, "DRET": 0x7B2}[mnem] word = riscvIType(riscvEnc{0x73, 0x0, 0x00}, 0, 0, imm) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil case "SFENCEVMA": // INSTR rs1, rs2: the memory-management fence, funct7 0x09 and an // all-zero rd. The toolchain's object table carries the encoding // (ASFENCEVMA, funct7 9) but its assembler accepts no mnemonic for // it, so the privileged specification's form pins it: // SFENCEVMA X10, X11 is 0x12b50073. if len(ops) != 2 { return nil, fmt.Errorf("SFENCEVMA expects 2 operands, got %d", len(ops)) } rs1, err := riscvWantIntReg(mnem, "rs1", ops[0]) if err != nil { return nil, err } rs2, err := riscvWantIntReg(mnem, "rs2", ops[1]) if err != nil { return nil, err } word = riscvRType(riscvEnc{0x73, 0x0, 0x09}, 0, rs1, rs2) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil case "NEG", "NEGW": // INSTR rs [, rd]: SUB/SUBW with X0 in the rs1 field, the // one-operand form negating in place (ANEG: NEG rs, rd -> SUB rs, // X0, rd). The toolchain pins the bytes: NEG X5 is 0x405002b3. if len(ops) != 1 && len(ops) != 2 { return nil, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops)) } rs, err := riscvWantIntReg(mnem, "rs1", ops[0]) if err != nil { return nil, err } rd := rs if len(ops) == 2 { if rd, err = riscvWantIntReg(mnem, "rd", ops[1]); err != nil { return nil, err } } enc := riscvEnc{0x33, 0x0, 0x20} // sub if mnem == "NEGW" { enc = riscvEnc{0x3B, 0x0, 0x20} // subw } word = riscvRType(enc, rd, 0, rs) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil case "SEQZ", "SNEZ": // INSTR rs, rd: the set-equal and set-not-equal pseudos read as // SLTIU $1 and SLTU against X0 (ASEQZ/ASNEZ). The toolchain pins // the bytes: SEQZ X14, X15 is 0x00173793. if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rs, err := riscvWantIntReg(mnem, "rs1", ops[0]) if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } if mnem == "SEQZ" { word = riscvIType(riscvEnc{0x13, 0x3, 0x00}, rd, rs, 1) // sltiu $1 } else { word = riscvRType(riscvEnc{0x33, 0x3, 0x00}, rd, 0, rs) // sltu rd, x0, rs } return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // FP conversion / move instructions use a separate table (rs2 encodes // the conversion type, not a register). Handle them before the main // table lookup. if cvtEnc, ok := riscvCvtTable[mnem]; ok { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rdBank, rs1Bank := riscvCvtBanks(mnem) rs1, err := riscvWantRegDescr(mnem, "rs1", rs1Bank.String(), ops[0], rs1Bank, 0, 31) if err != nil { return nil, err } rd, err := riscvWantRegDescr(mnem, "rd", rdBank.String(), ops[1], rdBank, 0, 31) if err != nil { return nil, err } word := riscvCvtType(cvtEnc, rd, rs1) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // FP conversions with an explicit rounding mode: FCVTWS.RNE and friends // suffix the base mnemonic with RNE/RTZ/RDN/RUP/RMM, which lands in the // funct3 field, replacing the bare form's default. if i := strings.IndexByte(mnem, '.'); i > 0 { if base, ok := riscvCvtTable[mnem[:i]]; ok { rm, ok := riscvRoundModes[mnem[i+1:]] if !ok { return nil, fmt.Errorf("unsupported rounding mode in %q", mnem) } if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rdBank, rs1Bank := riscvCvtBanks(mnem[:i]) rs1, err := riscvWantRegDescr(mnem, "rs1", rs1Bank.String(), ops[0], rs1Bank, 0, 31) if err != nil { return nil, err } rd, err := riscvWantRegDescr(mnem, "rd", rdBank.String(), ops[1], rdBank, 0, 31) if err != nil { return nil, err } base.funct3 = uint32(rm) word := riscvCvtType(base, rd, rs1) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } } // R4-type fused multiply-add: INSTR rs1, rs2, rs3, rd (destination last). if fmaEnc, ok := riscvFmaTable[mnem]; ok { if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } rs1, err := riscvWantFloatReg(mnem, "rs1", ops[0]) if err != nil { return nil, err } rs2, err := riscvWantFloatReg(mnem, "rs2", ops[1]) if err != nil { return nil, err } rs3, err := riscvWantFloatReg(mnem, "rs3", ops[2]) if err != nil { return nil, err } rd, err := riscvWantFloatReg(mnem, "rd", ops[3]) if err != nil { return nil, err } word := riscvFmaType(fmaEnc, rd, rs1, rs2, rs3) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // CSR instructions: INSTR csr, rs1|uimm, rd (destination last). The // write-only pseudos (CSRS/CSRC/CSRW and their immediate forms) spell // the source first, the CSR second, and read the destination as X0; the // immediate or register variant follows the source operand's kind. csrMnem := mnem csrPseudo := false csrRead := false csrFix := int32(0) switch mnem { case "CSRS", "CSRW", "CSRC", "CSRSI", "CSRWI", "CSRCI": csrMnem = map[string]string{ "CSRS": "CSRRS", "CSRW": "CSRRW", "CSRC": "CSRRC", "CSRSI": "CSRRSI", "CSRWI": "CSRRWI", "CSRCI": "CSRRCI", }[mnem] csrPseudo = true // CSRR csr, rd is CSRRS rd, csr, X0; the read pseudos RDCYCLE/RDTIME/ // RDINSTRET fix the CSR to cycle/time/instret. case "CSRR": csrMnem = "CSRRS" csrPseudo = true csrRead = true case "RDCYCLE", "RDTIME", "RDINSTRET": csrMnem = "CSRRS" csrPseudo = true csrRead = true csrFix = map[string]int32{"RDCYCLE": 0xC00, "RDTIME": 0xC01, "RDINSTRET": 0xC02}[mnem] } if csrEnc, ok := riscvCsrTable[csrMnem]; ok { if csrRead && len(ops) != 1 && len(ops) != 2 { return nil, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops)) } if csrPseudo && !csrRead && len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } if !csrPseudo && len(ops) != 3 && len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } csrOp := ops[0] srcOp := ops[0] rdOp := ops[len(ops)-1] switch { case csrRead: // CSRR csr, rd (or the fixed-CSR read pseudos with only rd). case csrPseudo: // src, csr. if len(ops) > 1 { csrOp, srcOp = ops[1], ops[0] } rdOp = nil // The toolchain picks the opcode form from the source's kind: // CSRW $2, csr assembles as CSRRWI exactly as CSRWI does, and // the register spellings stay on CSRRW. if isImmOperand(srcOp) { csrMnem = map[string]string{ "CSRS": "CSRRSI", "CSRW": "CSRRWI", "CSRC": "CSRRCI", "CSRSI": "CSRRSI", "CSRWI": "CSRRWI", "CSRCI": "CSRRCI", }[mnem] } else { csrMnem = map[string]string{ "CSRS": "CSRRS", "CSRW": "CSRRW", "CSRC": "CSRRC", "CSRSI": "CSRRSI", "CSRWI": "CSRRWI", "CSRCI": "CSRRCI", }[mnem] } csrEnc = riscvCsrTable[csrMnem] case len(ops) == 2: // The read form of a full CSR name: CSRRW csr, rd. One operand // must name a CSR; neither does is the toolchain's "missing CSR // name". if _, ok := riscvCSRAddress(ops[0]); ok { rdOp = ops[1] } else if _, ok := riscvCSRAddress(ops[1]); ok { csrOp, rdOp = ops[1], ops[0] } else { return nil, fmt.Errorf("%s: missing CSR name", mnem) } csrRead = true default: // Either src, csr, rd or csr, src, rd: a CSR *name* in the // second operand marks the toolchain's order. srcOp = ops[1] if op := ops[1]; op.Addr.Sym != nil { if _, ok := riscvCSRNames[strings.ToUpper(op.Addr.Sym.Name)]; ok { csrOp, srcOp = ops[1], ops[0] } } // An immediate source selects the immediate opcode (CSRRW $2, // c, rd encodes CSRRWI, byte-identical to the explicit form), // mirroring the toolchain's constant rewrite. if isImmOperand(srcOp) { csrMnem = map[string]string{ "CSRRW": "CSRRWI", "CSRRS": "CSRRSI", "CSRRC": "CSRRCI", }[csrMnem] if csrMnem != "" { csrEnc = riscvCsrTable[csrMnem] } } } // The source takes an integer register or an immediate: a memory // operand is the toolchain's first-operand rejection. if !csrRead && len(ops) > 0 && isMemOperand(srcOp) { return nil, fmt.Errorf("%s: integer register or immediate expected for 1st operand", mnem) } csr, ok := riscvCSRAddress(csrOp) if !ok && csrFix == 0 { return nil, fmt.Errorf("%s: unknown CSR %q", mnem, csrOp.Raw) } if csrFix != 0 { csr = csrFix } if csr < 0 || csr > 0xFFF { return nil, fmt.Errorf("%s: CSR address %d out of range 0-0xFFF", mnem, csr) } rd := 0 if !csrPseudo || csrRead { // The destination takes an integer register: a memory operand // is the toolchain's output rejection. if rdOp != nil && isMemOperand(rdOp) { return nil, fmt.Errorf("%s: needs an integer register output", mnem) } var err error if rd, err = riscvWantIntReg(mnem, "rd", rdOp); err != nil { return nil, err } } var src int switch { case csrRead: // CSRR reads with rs1 = X0: src stays zero. case isImmOperand(srcOp): // Immediate variant: the source is a 5-bit unsigned immediate. src = int(immFromOperand(srcOp)) if src < 0 || src > 31 { return nil, fmt.Errorf("%s: immediate %d out of range 0 to 31", mnem, src) } case csrEnc.imm: return nil, fmt.Errorf("%s expects an immediate source", mnem) default: // Register variant: the source is an integer register. var err error if src, err = riscvWantIntReg(mnem, "rs1", srcOp); err != nil { return nil, err } } word := riscvCsrType(csrEnc, rd, src, csr) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // The toolchain's synthesised instructions and the RVV slice: expanded // encodings the main table does not carry. FSGNJD is a plain table // entry and stays with the FP arithmetic path. if code, handled, err := encodeRISCVExtended(mnem, instr, pc, offsets, pcRelPcs); handled { if err != nil { return nil, err } return code, nil } enc, ok := riscvInstrTable[mnem] if !ok { return nil, fmt.Errorf("unsupported RISC-V instruction %q", mnem) } switch { // R-type: Go reverses the ISA order, writing rs2, rs1, rd (destination // last); the two-operand form INSTR rs2, rd uses rd as rs1. case len(ops) == 3 && isRTypeInstr(mnem): rs2, err := riscvWantIntReg(mnem, "rs2", ops[0]) // first operand = rs2 if err != nil { return nil, err } rs1, err := riscvWantIntReg(mnem, "rs1", ops[1]) // second operand = rs1 if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[2]) // destination (last operand) if err != nil { return nil, err } word = riscvRType(enc, rd, rs1, rs2) case len(ops) == 2 && isRTypeInstr(mnem): rs2, err := riscvWantIntReg(mnem, "rs2", ops[0]) // source (first operand) if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) // destination (second operand) if err != nil { return nil, err } word = riscvRType(enc, rd, rd, rs2) // I-type shift (SLLI, SRLI, SRAI): INSTR $shamt, rs1, rd; the two-operand // form INSTR $shamt, rd uses rd as the source. The shift amount is // bounded at the instruction width, as the toolchain validates it: 0-63 // for the doubleword forms, 0-31 for the word forms. case len(ops) == 3 && isShiftImmInstr(mnem): shamt, ok := riscvRawImm(ops[0]) if !ok { return nil, fmt.Errorf("%s expects an immediate shift amount", mnem) } if hi := riscvShiftMax(mnem); shamt < 0 || shamt > hi { return nil, fmt.Errorf("%s: immediate %d out of range 0 to %d", mnem, shamt, hi) } rs1, err := riscvWantIntReg(mnem, "rs1", ops[1]) if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[2]) if err != nil { return nil, err } word = riscvRType(enc, rd, rs1, int(shamt)) case len(ops) == 2 && isShiftImmInstr(mnem): shamt, ok := riscvRawImm(ops[0]) if !ok { return nil, fmt.Errorf("%s expects an immediate shift amount", mnem) } if hi := riscvShiftMax(mnem); shamt < 0 || shamt > hi { return nil, fmt.Errorf("%s: immediate %d out of range 0 to %d", mnem, shamt, hi) } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } word = riscvRType(enc, rd, rd, int(shamt)) // AMO atomics: Plan 9 order is INSTR src, (addr), dst. case len(ops) == 3 && isAMOInstr(mnem): rs2, err := riscvWantIntReg(mnem, "rs2", ops[0]) // source value if err != nil { return nil, err } rs1, _ := memFromOperandWithFrame(ops[1], fi) // memory address rd, err := riscvWantIntReg(mnem, "rd", ops[2]) if err != nil { return nil, err } if rs1 < 0 { return nil, fmt.Errorf("%s: expected integer register in rs1 position", mnem) } word = riscvAMOType(enc, rd, rs1, rs2) // Zbb unary bit operations: INSTR rs, rd, exactly two operands as the // toolchain spells them. The rs2 field is fixed, not zero: the table // below carries the constant each operation reads (CLZ counts leading // zeros with an empty field, REV8 works on bytes at position 24). case len(ops) == 2 && isZbUnaryInstr(mnem): rs1, err := riscvWantIntReg(mnem, "rs1", ops[0]) if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } word = riscvRType(enc, rd, rs1, riscvZbUnaryRS2[mnem]) // FP arithmetic: Go reverses the ISA order, writing rs2, rs1, rd. case len(ops) == 3 && isFPArithInstr(mnem): rs2, err := riscvWantFloatReg(mnem, "rs2", ops[0]) if err != nil { return nil, err } rs1, err := riscvWantFloatReg(mnem, "rs1", ops[1]) if err != nil { return nil, err } rd, err := riscvWantFloatReg(mnem, "rd", ops[2]) if err != nil { return nil, err } word = riscvRType(enc, rd, rs1, rs2) // FP arithmetic (2-operand): FSQRT src, dst. case len(ops) == 2 && isFPArithInstr(mnem): rs1, err := riscvWantFloatReg(mnem, "rs1", ops[0]) if err != nil { return nil, err } rd, err := riscvWantFloatReg(mnem, "rd", ops[1]) if err != nil { return nil, err } word = riscvRType(enc, rd, rs1, 0) // FP loads: INSTR addr, freg (Plan 9: source first). case len(ops) == 2 && isFPLoadInstr(mnem): rd, err := riscvWantFloatReg(mnem, "rd", ops[1]) if err != nil { return nil, err } rs1, imm, err := riscvAccessMem(mnem, ops[0], fi) if err != nil { return nil, err } return riscvFrameMemOp(enc, false, rd, rs1, imm), nil // FP stores: INSTR freg, addr (Plan 9: source first). case len(ops) == 2 && isFPStoreInstr(mnem): rs2, err := riscvWantFloatReg(mnem, "rs2", ops[0]) if err != nil { return nil, err } rs1, imm, err := riscvAccessMem(mnem, ops[1], fi) if err != nil { return nil, err } return riscvFrameMemOp(enc, true, rs2, rs1, imm), nil // LR (load-reserved): INSTR (addr), dst. The toolchain reads the // operands positionally, so the base register comes from the first // operand and the destination from the second whatever their parens. case len(ops) == 2 && isLRInstr(mnem): rs1 := regFromOperand(ops[0]) rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } if rs1 < 0 { return nil, fmt.Errorf("%s: expected integer register in rs1 position", mnem) } word = riscvAMOType(enc, rd, rs1, 0) // rs2=0 for LR // SC (store-conditional): INSTR src, (addr), dst, 3 operands. case len(ops) == 3 && isSCInstr(mnem): rs2, err := riscvWantIntReg(mnem, "rs2", ops[0]) if err != nil { return nil, err } rs1, _ := memFromOperandWithFrame(ops[1], fi) rd, err := riscvWantIntReg(mnem, "rd", ops[2]) if err != nil { return nil, err } if rs1 < 0 { return nil, fmt.Errorf("%s: expected integer register in rs1 position", mnem) } word = riscvAMOType(enc, rd, rs1, rs2) // FP compare: Go reverses the ISA order, writing rs2, rs1, rd. case len(ops) == 3 && isFPCmpInstr(mnem): rs2, err := riscvWantFloatReg(mnem, "rs2", ops[0]) if err != nil { return nil, err } rs1, err := riscvWantFloatReg(mnem, "rs1", ops[1]) if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[2]) if err != nil { return nil, err } word = riscvRType(enc, rd, rs1, rs2) // I-type with immediate: Plan 9 order is INSTR $imm, rs1, rd; the // two-operand form INSTR $imm, rd uses rd as the source. case len(ops) == 3 && isITypeInstr(mnem): imm, err := riscvImm32FromOperand(ops[0], immNeg) // immediate if err != nil { return nil, err } rs1, err := riscvWantIntReg(mnem, "rs1", ops[1]) // source register if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[2]) // destination if err != nil { return nil, err } return encodeRISCVItypeImmediate(mnem, enc, rd, rs1, imm) case len(ops) == 2 && isITypeInstr(mnem): imm, err := riscvImm32FromOperand(ops[0], immNeg) if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } return encodeRISCVItypeImmediate(mnem, enc, rd, rd, imm) // Loads: rd, offset(rs1), Plan 9 order is LD src, dst. case len(ops) == 2 && isLoadInstr(mnem): rd, err := riscvWantIntReg(mnem, "rd", ops[1]) // destination (last operand) if err != nil { return nil, err } rs1, imm, err := riscvAccessMem(mnem, ops[0], fi) // memory source (first operand) if err != nil { return nil, err } return riscvFrameMemOp(enc, false, rd, rs1, imm), nil // Stores: Plan 9 order is SD src, dst (src=register, dst=memory). case len(ops) == 2 && isStoreInstr(mnem): rs2, err := riscvWantIntReg(mnem, "rs2", ops[0]) // source register (first operand) if err != nil { return nil, err } rs1, imm, err := riscvAccessMem(mnem, ops[1], fi) // memory dest (last operand) if err != nil { return nil, err } return riscvFrameMemOp(enc, true, rs2, rs1, imm), nil // Branches: rs1, rs2, label. BGT/BLE/BGTU/BLEU are the swapped-spelling // forms of BLT/BGE/BLTU/BGEU (bgt rs1, rs2 is blt rs2, rs1). case len(ops) == 3 && isBranchInstr(mnem): rs1, err := riscvWantIntReg(mnem, "rs1", ops[0]) if err != nil { return nil, err } rs2, err := riscvWantIntReg(mnem, "rs2", ops[1]) if err != nil { return nil, err } // The toolchain rejects a branch whose third operand does not name // a destination at all: a constant or a register-indirect address // is not one (the N(PC) relative form is). if ops[2].Imm.HasVal || (ops[2].Addr.Base != "" && ops[2].Addr.Base != "PC" && ops[2].Addr.Sym == nil) { return nil, fmt.Errorf("%s: instruction with branch-like opcode lacks destination", mnem) } target := labelFromOperand(ops[2]) switch mnem { case "BGT": enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, rs2, rs1 // blt rs2, rs1 case "BLE": enc, rs1, rs2 = riscvEnc{0x63, 0x5, 0x00}, rs2, rs1 // bge rs2, rs1 case "BGTU": enc, rs1, rs2 = riscvEnc{0x63, 0x6, 0x00}, rs2, rs1 // bltu rs2, rs1 case "BLEU": enc, rs1, rs2 = riscvEnc{0x63, 0x7, 0x00}, rs2, rs1 // bgeu rs2, rs1 } targetOff := 0 if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel { if err != nil { return nil, err } targetOff = off } else { var ok bool targetOff, ok = offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } } offset := int32(targetOff - pc) if err := riscvCheckBranchOffset(target, offset); err != nil { return nil, err } // The Go assembler never compresses branches to C.BEQZ/C.BNEZ. word = riscvBType(enc, rs1, rs2, offset) // U-type: rd, imm (or the toolchain testdata's INSTR $imm, rd). The // immediate rides the field raw (riscv64.s: AUIPC $524287, X10 encodes // 7ffff517), so it shifts into imm[31:12] here, and the span is the // signed 20-bit range the toolchain checks. case len(ops) == 2 && isUTypeInstr(mnem): var rd int var imm int32 var err error if isImmOperand(ops[0]) { imm = immFromOperand(ops[0]) rd, err = riscvWantIntReg(mnem, "rd", ops[1]) } else { rd, err = riscvWantIntReg(mnem, "rd", ops[0]) imm = immFromOperand(ops[1]) } if err != nil { return nil, err } if imm < -(1<<19) || imm > (1<<19)-1 { return nil, fmt.Errorf("%s: signed immediate 0x%x must be in range [-0x80000, 0x7ffff] (20 bits)", mnem, imm) } word = riscvUType(enc, rd, imm<<12) default: return nil, fmt.Errorf("cannot encode %s with %d operands", mnem, len(ops)) } // Emit as little-endian 32-bit word. return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } // isMemOperand reports whether an operand is a memory reference // (frame-relative such as name+off(FP) or register-relative such as (X10)). func isMemOperand(op *ast.Operand) bool { if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { return true // name+off(FP), name+off(SP) } if op.Addr.Base != "" && op.Addr.Sym == nil { return true // (reg) } return false } // isImmOperand reports whether an operand is an immediate ($value). func isImmOperand(op *ast.Operand) bool { if op.Kind == ast.OpImmediate { return true } if op.Imm.HasVal { return true } return false } // encodeRISCVMov encodes the MOV pseudo-instruction. // // The Go RISC-V assembler uses MOV for: // - MOV name+off(FP), Rd load from frame // - MOV Rd, name+off(FP) store to frame // - MOV (Rs), Rd register-relative load // - MOV Rs, (Rd) register-relative store // - MOV Rs, Rd register-to-register move (ADDI $0) // - MOV $imm, Rd load immediate (ADDI or LUI+ADDIW) func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc, lits *riscvLiterals, tlsSyms map[string]bool) ([]byte, error) { ops := instr.Operands if len(ops) != 2 { return nil, fmt.Errorf("illegal MOV instruction") } src := ops[0] dst := ops[1] mnem := strings.ToUpper(instr.Mnemonic.Text) // Immediate → register. if isImmOperand(src) { // An address constant names a pseudo register, either as a symbol // ($sym(SB), $sym+4(FP)) or as a bare offset whose raw spelling // carries it ($8(SP)). The toolchain wants a register target for // an address and supports the width-less MOV alone. addrPseudo := "" if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" { addrPseudo = src.Imm.Sym.Pseudo } else if src.Imm.HasVal { // The raw spelling carries the pseudo the number rides ($8(SP)); // the spaces the parser records between the tokens come out. raw := strings.Join(strings.Fields(src.Raw), "") for _, p := range []string{"(FP)", "(SP)", "(SB)", "(PC)"} { if strings.HasSuffix(raw, p) { addrPseudo = strings.Trim(p, "()") break } } } if addrPseudo != "" { if isMemOperand(dst) { return nil, fmt.Errorf("%s: address load must target register", mnem) } if mnem != "MOV" { return nil, fmt.Errorf("%s: unsupported address load", mnem) } } else { // A constant load: the toolchain wants a register target and // supports only the width-less MOV spelling, plus the // FP-constant forms of MOVF and MOVD. fpConst := (mnem == "MOVF" || mnem == "MOVD") && !src.Imm.HasVal && src.Imm.Sym == nil && src.Imm.Str == "" if isMemOperand(dst) { return nil, fmt.Errorf("%s: constant load must target register", mnem) } if mnem != "MOV" && !fpConst { return nil, fmt.Errorf("%s: unsupported constant load", mnem) } } // FP constant → FP register: a zero bit pattern moves through FMV // from X0, anything else loads from the pooled $f32/$f64 constant // symbol the toolchain synthesises (AUIPC + FLW/FLD through TMP). // An integer constant is not an FP load source, exactly as the // toolchain rejects the non-FCONST forms. if (mnem == "MOVF" || mnem == "MOVD") && !src.Imm.HasVal && src.Imm.Sym == nil && src.Imm.Str == "" { rd := regFromOperand(dst) if rd < 0 || !riscvIsFloatRegOperand(dst) { return nil, fmt.Errorf("%s $float: invalid destination register", mnem) } pattern, _, err := riscvFPConstBits(mnem, src) if err != nil { return nil, fmt.Errorf("%s: invalid floating-point constant %q", mnem, src.Imm.Float) } if pattern == 0 { op := uint32(0x78) << 25 // FMV.W.X if mnem == "MOVD" { op = uint32(0x79) << 25 // FMV.D.X } return wordLE(op | uint32(rd)<<7 | 0x53), nil } var name string var data []byte double := mnem == "MOVD" if double { name = fmt.Sprintf("$f64.%016x", pattern) data = riscvLiteralBytes(int64(pattern)) } else { name = fmt.Sprintf("$f32.%08x", uint32(pattern)) data = []byte{byte(pattern), byte(pattern >> 8), byte(pattern >> 16), byte(pattern >> 24)} } if lits != nil { lits.add(name, data) } return encodeRISCVSBFPLoad(name, rd, double, relocs), nil } // MOV $sym(SB), rd, load address of a static symbol or external. if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { rd, err := riscvWantMovReg(mnem, "rd", dst) if err != nil { return nil, err } return encodeRISCVSBAddr(src.Imm.Sym, rd, relocs), nil } // MOV $sym+off(FP|SP), rd: the address of a frame slot as an // immediate is the frame-adjusted offset against the hardware SP, // the toolchain's ADDI $adj, SP, rd (argframe+0(FP) in the runtime's // reflect trampolines is the spelling). A bare offset with the // pseudo in its raw spelling ($8(SP)) carries no frame adjustment. if addrPseudo == "FP" || addrPseudo == "SP" { rd, err := riscvWantMovReg(mnem, "rd", dst) if err != nil { return nil, err } off := int32(0) if src.Imm.Sym != nil { _, off = riscvResolvePseudo(src.Imm.Sym, fi) } else { v := src.Imm.Val if src.Imm.Neg { v = -v } off = int32(v) if addrPseudo == "FP" { off += int32(fi.autosize) + 8 } else { off += int32(fi.autosize) } } return riscvSPAddiBytes(rd, off), nil } // MOV $sym(FP/SP), rd, not supported: immediate symbol references // other than the frame pseudos cannot be encoded as a simple // immediate. if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" { return nil, fmt.Errorf("MOV $%s(%s): unsupported immediate symbol reference (only SB is supported)", src.Imm.Sym.Name, src.Imm.Sym.Pseudo) } rd, err := riscvWantMovReg(mnem, "rd", dst) if err != nil { return nil, err } imm := riscvOperandImm64(src) if int64(int32(imm)) != imm { // Beyond the signed 32-bit span the toolchain either builds the // value from a shifted 32-bit part or loads it from the pooled // $i64 constant it synthesises for the purpose. return riscvLoadImm64(rd, imm, lits, relocs), nil } return encodeRISCVLoadImm(rd, int32(imm)), nil } // Memory → register (load). if isMemOperand(src) && !isMemOperand(dst) { rd, err := riscvWantMovReg(mnem, "rd", dst) if err != nil { return nil, err } // MOV sym(SB), rd, load from static data. A TLSBSS symbol takes // the local-exec sequence: LUI + ADDIW carry the offset against TP, // the ADD folds the thread pointer in, the access reads through TMP. if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" { if tlsSyms[src.Addr.Sym.Name] { return riscvTLSBytes(riscvMovEnc(mnem, false), false, rd, src.Addr.Sym, relocs), nil } return encodeRISCVSBLoad(src.Addr.Sym, rd, relocs), nil } if err := riscvWantMemBase(mnem, "rs1", src); err != nil { return nil, err } if err := riscvWantMemOffset(mnem, src, fi); err != nil { return nil, err } rs1, off := memFromOperandWithFrame(src, fi) if rs1 < 0 { return nil, fmt.Errorf("%s: expected integer register in rs1 position", mnem) } return riscvFrameMemOp(riscvMovEnc(mnem, false), false, rd, rs1, off), nil } // Register → memory (store). if !isMemOperand(src) && isMemOperand(dst) { // The zero-extending widths synthesise their sign correction after // a load; no store form exists, exactly as the toolchain rejects. if mnem == "MOVBU" || mnem == "MOVHU" || mnem == "MOVWU" { return nil, fmt.Errorf("%s: unsupported unsigned store", mnem) } rs2, err := riscvWantMovReg(mnem, "rs2", src) if err != nil { return nil, err } // MOV rd, sym(SB), store to static data. A TLSBSS symbol takes // the local-exec sequence with the store through TMP. if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" { if tlsSyms[dst.Addr.Sym.Name] { return riscvTLSBytes(riscvMovEnc(mnem, true), true, rs2, dst.Addr.Sym, relocs), nil } return encodeRISCVSBStore(dst.Addr.Sym, rs2, relocs), nil } if err := riscvWantMemBase(mnem, "rs1", dst); err != nil { return nil, err } if err := riscvWantMemOffset(mnem, dst, fi); err != nil { return nil, err } rs1, off := memFromOperandWithFrame(dst, fi) if rs1 < 0 { return nil, fmt.Errorf("%s: expected integer register in rs1 position", mnem) } return riscvFrameMemOp(riscvMovEnc(mnem, true), true, rs2, rs1, off), nil } // Register → register. The width suffix selects the toolchain's // synthesis: MOVF/MOVD between the integer and FP banks are FMV and // inside the FP bank FSGNJ with rs2 = rs1; MOVW is ADDIW $0; MOVBU is // ANDI $255; MOVB and MOVH sign-extend through SLLI+SRAI and MOVHU/ // MOVWU zero-extend through SLLI+SRLI; bare MOV is ADDI $0, which the // RVC pass compresses. { srcF, dstF := riscvIsFloatRegOperand(src), riscvIsFloatRegOperand(dst) if mnem == "MOVF" || mnem == "MOVD" { if !srcF && !dstF { return nil, fmt.Errorf("%s: expected float register in rd position but got non-float register %s", mnem, operandRegName(dst)) } } else { // The integer widths move through the integer file alone: a // float or vector register in either slot is the toolchain's // bank rejection, the destination reported first. if _, err := riscvWantMovReg(mnem, "rd", dst); err != nil { return nil, err } if _, err := riscvWantMovReg(mnem, "rs1", src); err != nil { return nil, err } } rs1 := regFromOperand(src) rd := regFromOperand(dst) switch mnem { case "MOVF", "MOVD": switch { case srcF && dstF: op := uint32(0x20000053) // FSGNJ.S if mnem == "MOVD" { op = 0x22000053 // FSGNJ.D } return wordLE(op | uint32(rs1)<<15 | uint32(rs1)<<20 | uint32(rd)<<7), nil case dstF && !srcF: // The FMV spellings move the bit pattern through the // integer file: a vector register in the source slot // names no integer register. if _, err := riscvWantMovReg("MOV", "rs1", src); err != nil { return nil, err } op := uint32(0x78) << 25 // FMV.W.X if mnem == "MOVD" { op = uint32(0x79) << 25 // FMV.D.X } return wordLE(op | uint32(rs1)<<15 | uint32(rd)<<7 | 0x53), nil case srcF && !dstF: if _, err := riscvWantMovReg("MOV", "rd", dst); err != nil { return nil, err } op := uint32(0x70) << 25 // FMV.X.W if mnem == "MOVD" { op = uint32(0x71) << 25 // FMV.X.D } return wordLE(op | uint32(rs1)<<15 | uint32(rd)<<7 | 0x53), nil } return nil, fmt.Errorf("%s: both registers must be in the same bank", mnem) case "MOVW": // ADDIW $0, rs, rd; the two-operand-only form never has // rd == rs1 in real sources, and the toolchain's C.ADDIW // forbids a zero immediate, so the word stays uncompressed. if srcF || dstF { return nil, fmt.Errorf("%s: expected integer register in rd position", mnem) } return wordLE(riscvIType(riscvEnc{0x1B, 0x0, 0x00}, rd, rs1, 0)), nil case "MOVBU": // ANDI $255, rs, rd; 255 never fits C.ANDI's six signed bits. if srcF || dstF { return nil, fmt.Errorf("%s: expected integer register in rd position", mnem) } return wordLE(riscvIType(riscvEnc{0x13, 0x7, 0x00}, rd, rs1, 0xFF)), nil case "MOVB", "MOVH": if srcF || dstF { return nil, fmt.Errorf("%s: expected integer register in rd position", mnem) } shamt := 56 if mnem == "MOVH" { shamt = 48 } return riscvExtendBytes(rd, rs1, shamt, true), nil case "MOVHU", "MOVWU": if srcF || dstF { return nil, fmt.Errorf("%s: expected integer register in rd position", mnem) } shamt := 48 if mnem == "MOVWU" { shamt = 32 } return riscvExtendBytes(rd, rs1, shamt, false), nil } word := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs1, 0) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } } // riscvMovEnc returns the load (store=false) or store (store=true) opcode for // a MOV-family mnemonic: the suffix selects the access width, MOVD and MOVF // select the FP load/store opcodes, and bare MOV is the 64-bit integer form. func riscvMovEnc(mnem string, store bool) riscvEnc { if store { switch mnem { case "MOVB": return riscvEnc{0x23, 0x0, 0x00} // SB case "MOVH": return riscvEnc{0x23, 0x1, 0x00} // SH case "MOVW": return riscvEnc{0x23, 0x2, 0x00} // SW case "MOVF": return riscvEnc{0x27, 0x2, 0x00} // FSW case "MOVD": return riscvEnc{0x27, 0x3, 0x00} // FSD } return riscvEnc{0x23, 0x3, 0x00} // SD } switch mnem { case "MOVB": return riscvEnc{0x03, 0x0, 0x00} // LB case "MOVBU": return riscvEnc{0x03, 0x4, 0x00} // LBU case "MOVH": return riscvEnc{0x03, 0x1, 0x00} // LH case "MOVHU": return riscvEnc{0x03, 0x5, 0x00} // LHU case "MOVW": return riscvEnc{0x03, 0x2, 0x00} // LW case "MOVWU": return riscvEnc{0x03, 0x6, 0x00} // LWU case "MOVF": return riscvEnc{0x07, 0x2, 0x00} // FLW case "MOVD": return riscvEnc{0x07, 0x3, 0x00} // FLD } return riscvEnc{0x03, 0x3, 0x00} // LD } // riscvFrameMemOp encodes a register-relative load (store=false, I-type // width 0x03) or store (store=true, S-type width 0x23) of the 64-bit width // at off(rs1). Offsets beyond the signed 12-bit range materialise the // address in X31 first: LUI hi (the rounding split), then ADD X31, rs1, // matching the toolchain's large-frame addressing; the access uses the // sign-extended low part, which always fits. func riscvFrameMemOp(enc riscvEnc, store bool, reg, rs1 int, off int32) []byte { if fits12(off) { var word uint32 if store { word = riscvSType(enc, rs1, reg, off) } else { word = riscvIType(enc, reg, rs1, off) } return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)} } lo := off - (splitHi(off) << 12) out := riscvAddressInX31WithBase(off, rs1) var word uint32 if store { word = riscvSType(enc, 31, reg, lo) } else { word = riscvIType(enc, reg, 31, lo) } return append(out, wordLE(word)...) } // riscvFrameMemSize returns the encoded size of a frame-relative MOV for the // layout pass: 4 bytes when the offset fits, otherwise the X31 // materialisation plus the access. func riscvFrameMemSize(op *ast.Operand, fi riscvFrameInfo) int { rs1, off := memFromOperandWithFrame(op, fi) if fits12(off) { return 4 } return len(riscvAddressInX31WithBase(off, rs1)) + 4 } // encodeRISCVGetCallerPC encodes the toolchain's GETCALLERPC rewrite: on // entry the caller's address sits in the link register, so a leaf reads it // straight from X1 (MOV X1, rd, which the compress pass turns into C.MV); // a body that calls out has clobbered X1, and reads the prologue's save at // 0(SP) instead (LD rd, 0(SP), compressed to C.LDSP). A memory destination // stores the address there through the MOV store paths, which the leaf form // shares with every width and offset; the framed form is the toolchain's // own rejection, its rewrite reading (SP) into a MOV with two memory ends. func encodeRISCVGetCallerPC(ops []*ast.Operand, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) { if len(ops) == 0 { return nil, fmt.Errorf("GETCALLERPC: unsupported MOV") } dst := ops[len(ops)-1] if isMemOperand(dst) { if !fi.leaf { return nil, fmt.Errorf("GETCALLERPC: unsupported MOV") } if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" { // A static destination rides the external store path, exactly // as MOV X1, sym(SB) does. return encodeRISCVSBStore(dst.Addr.Sym, 1, relocs), nil } if err := riscvWantMemBase("GETCALLERPC", "rs1", dst); err != nil { return nil, err } rs1, off, err := riscvAccessMem("GETCALLERPC", dst, fi) if err != nil { return nil, err } return riscvFrameMemOp(riscvEnc{0x23, 0x3, 0x00}, true, 1, rs1, off), nil } n, bank := riscvBankedRegNum(operandRegName(dst)) if bank != riscvBankInt { return nil, fmt.Errorf("GETCALLERPC: expected integer register in rd position but got non-integer register %s", operandRegName(dst)) } if fi.leaf { return wordLE(riscvIType(riscvInstrTable["ADDI"], n, 1, 0)), nil } return wordLE(riscvIType(riscvInstrTable["LD"], n, 2, 0)), nil } // riscvGetCallerPCSize returns the encoded size of a GETCALLERPC for the // layout pass: one word for the register reads, the store's own size // (expanded where the offset leaves imm12) for a memory destination. func riscvGetCallerPCSize(ops []*ast.Operand, fi riscvFrameInfo) int { if len(ops) == 0 { return 4 } dst := ops[len(ops)-1] if isMemOperand(dst) && (dst.Addr.Sym == nil || dst.Addr.Sym.Pseudo != "SB") { if err := riscvWantMemOffset("GETCALLERPC", dst, fi); err == nil { if rs1, off := memFromOperandWithFrame(dst, fi); rs1 >= 0 && !fits12(off) { return len(riscvAddressInX31WithBase(off, rs1)) + 4 } } } return 4 } // riscvMemInstrSize returns the encoded size of a plain load or store // (integer and FP widths, one register end and one memory end) for the layout // pass: 4 bytes when the offset fits the signed 12-bit span, otherwise the // X31 materialisation plus the access, exactly what riscvFrameMemOp emits // for them. A constant beyond the signed 32-bit span encodes never: the 4 // bytes guessed here are never emitted, because the encode pass rejects the // instruction with the toolchain's "constant too large" diagnostic first. func riscvMemInstrSize(mnem string, ops []*ast.Operand, fi riscvFrameInfo) (int, bool) { var mem *ast.Operand switch { case len(ops) == 2 && (isLoadInstr(mnem) || isFPLoadInstr(mnem)): mem = ops[0] case len(ops) == 2 && (isStoreInstr(mnem) || isFPStoreInstr(mnem)): mem = ops[1] default: return 0, false } if err := riscvWantMemOffset(mnem, mem, fi); err != nil { return 4, true } rs1, off := memFromOperandWithFrame(mem, fi) if rs1 < 0 || fits12(off) { return 4, true } return len(riscvAddressInX31WithBase(off, rs1)) + 4, true } // riscvWantMemOffset rejects a memory operand whose byte offset leaves the // signed 32-bit span, the point where the toolchain's Split32BitImmediate // stops and reports "constant %d too large". The frame pseudo-registers // resolve against the frame first, so their offsets are checked resolved, // exactly as the toolchain's stackOffset feeds Split32BitImmediate the // adjusted address; an SB reference rides the relocation paths and is never // this check's subject. func riscvWantMemOffset(mnem string, op *ast.Operand, fi riscvFrameInfo) error { var off int64 switch { case op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP": off = op.Addr.Offset + int64(fi.autosize) + 8 case op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SP": off = op.Addr.Offset + int64(fi.autosize) case op.Addr.Sym != nil: return nil default: off = op.Addr.Offset } if off < math.MinInt32 || off > math.MaxInt32 { return fmt.Errorf("%s: constant %d too large", mnem, off) } return nil } // riscvAccessMem resolves the memory end of a plain load or store: the base // register and the byte offset, range-checked. The access itself is emitted // by riscvFrameMemOp, whose hi/lo split matches the toolchain's // instructionsForLoad and instructionsForStore: the high part materialises in // the assembler's temporary register (X31) and the access reads or writes // through it. func riscvAccessMem(mnem string, op *ast.Operand, fi riscvFrameInfo) (int, int32, error) { if err := riscvWantMemOffset(mnem, op, fi); err != nil { return -1, 0, err } rs1, off := memFromOperandWithFrame(op, fi) if rs1 < 0 { return -1, 0, fmt.Errorf("%s: expected integer register in rs1 position", mnem) } return rs1, off, nil } // encodeRISCVLoadImm encodes loading an immediate into a register (MOV $imm, // rd), matching the toolchain's instructionsForMOVConst. For 12-bit // immediates it emits ADDI $imm, ZERO, rd (compressed to C.LI when it fits // six signed bits); for larger immediates it emits LUI + [ADDIW], with the LUI // and ADDIW compressed to C.LUI / C.ADDIW when their immediate fits. func encodeRISCVLoadImm(rd int, imm int32) []byte { if imm >= -2048 && imm <= 2047 { if rd != 0 && imm >= -32 && imm <= 31 { return word16(rvcCI(0x2, uint32(rd), uint32(imm)&0x3F)) // C.LI } return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, 0, imm)) } low, high := splitRISCV32Imm(imm) var out []byte if rd != 0 && rd != 2 && high >= -32 && high <= 31 { out = append(out, word16(rvcCI(0x3, uint32(rd), uint32(high)&0x3F))...) // C.LUI } else { out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, rd, high<<12))...) } if low != 0 { if low >= -32 && low <= 31 { out = append(out, word16(rvcCI(0x1, uint32(rd), uint32(low)&0x3F))...) // C.ADDIW } else { out = append(out, wordLE(riscvIType(riscvEnc{0x1B, 0x0, 0x00}, rd, rd, low))...) } } return out } // riscvMovImmSize returns the encoded byte length of MOV $imm, rd, mirroring // encodeRISCVLoadImm's expansion and compression. func riscvMovImmSize(rd int, imm int32) int { if imm >= -2048 && imm <= 2047 { if rd != 0 && imm >= -32 && imm <= 31 { return 2 // C.LI } return 4 // ADDI } low, high := splitRISCV32Imm(imm) size := 0 if rd != 0 && rd != 2 && high >= -32 && high <= 31 { size += 2 // C.LUI } else { size += 4 // LUI } if low != 0 { if low >= -32 && low <= 31 { size += 2 // C.ADDIW } else { size += 4 // ADDIW } } return size } // splitRISCV32Imm splits a signed 32-bit immediate into a signed 12-bit low // part and a signed 20-bit high part, mirroring cmd/internal/obj/riscv's // Split32BitImmediate. The high part is returned unshifted; callers place it // in the upper bits of LUI (or its compressed C.LUI form). func splitRISCV32Imm(imm int32) (low, high int32) { if imm >= -2048 && imm <= 2047 { return imm, 0 } h := int64(imm) >> 12 if imm&(1<<11) != 0 { h++ } low = int32((int64(imm) << 52) >> 52) // sign extend 12 bits high = int32((h << 44) >> 44) // sign extend 20 bits return low, high } // riscvNormalisePseudo rewrites the toolchain's UNDEF spelling onto EBREAK: // the assembler accepts UNDEF where the hardware wants the trap instruction // and emits ebreak (compressed to C.EBREAK under RVC), so every pass sees the // canonical name. The privileged aliases fold the same way: SCALL and // SBREAK are the supervisor spellings of ECALL and EBREAK and encode // identically. func riscvNormalisePseudo(mnem string) string { if strings.EqualFold(mnem, "UNDEF") { return "EBREAK" } switch mnem { case "SCALL": return "ECALL" case "SBREAK": return "EBREAK" } return mnem } // riscvOperandImm64 reads an immediate operand as a full signed 64-bit value, // where immFromOperand would truncate to int32; the MOV immediate path uses // it to classify the wide constants. func riscvOperandImm64(op *ast.Operand) int64 { if !op.Imm.HasVal { return 0 } v := op.Imm.Val if op.Imm.Neg { v = -v } return v } // riscvSplitShiftConst mirrors cmd/internal/obj/riscv's splitShiftConst: it // looks for the signed 32-bit integer a constant can be rebuilt from with a // left shift, a left-and-right shift pair (a run of ones), or a zero-extended // 32-bit pattern. A constant that fits none of the shapes is materialised // from the pooled $i64 data symbol instead. func riscvSplitShiftConst(v int64) (imm int64, lsh int, rsh int, ok bool) { // Rebuild from a signed 32-bit integer shifted left. lsh = bits.TrailingZeros64(uint64(v)) c := v >> lsh if int64(int32(c)) == c { return c, lsh, 0, true } // Rebuild from a small negative constant: shift left into place, then // shift the sign-extended ones run right. rsh = bits.LeadingZeros64(uint64(v)) ones := bits.OnesCount64((uint64(v) >> lsh) >> 11) if rsh+ones+lsh+11 == 64 { c = (1<<11 | ((v >> lsh) & 0x7ff)) << 52 >> 52 // sign extend 12 bits if lsh > 0 || c != -1 { lsh += rsh } return c, lsh, rsh, true } // Rebuild from a zero-extended signed 32-bit integer. if int64(uint32(c)) == c { c = int64(int32(c)) lsh, rsh = 32, 32-lsh return c, lsh, rsh, true } return 0, 0, 0, false } // riscvSPAddiBytes encodes ADDI rd, SP, imm for the frame-address immediates // (the MOV $sym+off(FP|SP) form), using the compressed forms the toolchain // picks under RVC: C.ADDI4SPN for a positive 4-byte multiple that fits, C.MV // for the zero offset, the plain ADDI otherwise. func riscvSPAddiBytes(rd int, imm int32) []byte { if rd != 0 && imm == 0 { return word16(rvcCR(0x8, uint32(rd), 2)) // C.MV rd, SP } if isRVCIntReg(rd) && imm > 0 && imm < 1024 && imm%4 == 0 { return word16(rvcCIW(0x0, rvcReg3(rd), uint32(imm))) } return wordLE(riscvIType(riscvInstrTable["ADDI"], rd, 2, imm)) } // riscvMovImm64Size returns the encoded byte length of MOV $imm, rd when the // immediate sits outside the signed 32-bit span: the shifted-part sequences // of riscvLoadImm64, or the 8-byte AUIPC+LD pool load. func riscvMovImm64Size(rd int, imm int64) int { c, lsh, rsh, ok := riscvSplitShiftConst(imm) if !ok { return 8 // AUIPC + LD against the $i64 pool symbol } size := riscvMovImmSize(rd, int32(c)) if lsh > 0 { size += riscvShiftImmSize(rd, true) } if rsh > 0 { size += riscvShiftImmSize(rd, false) } return size } // riscvShiftImmSize returns the encoded size of one SLLI/SRLI expansion // part: two bytes under RVC when the destination can carry a compressed // shift (C.SLLI admits every register but X0, C.SRLI only X8 to X15), four // otherwise. func riscvShiftImmSize(rd int, left bool) int { if rd != 0 && (left || isRVCIntReg(rd)) { return 2 } return 4 } // riscvLoadImm64 encodes MOV $imm, rd for an immediate beyond the signed // 32-bit span, mirroring the toolchain's instructionsForMOVConst: when a // shifted 32-bit part rebuilds the value it emits that part (compressed like // any written MOV) followed by the SLLI and SRLI shifts; otherwise it loads // the constant from the pooled read-only $i64. symbol via AUIPC + LD // and registers the literal so the data section carries its bytes. func riscvLoadImm64(rd int, imm int64, lits *riscvLiterals, relocs *[]Reloc) []byte { c, lsh, rsh, ok := riscvSplitShiftConst(imm) if !ok { name := fmt.Sprintf("$i64.%016x", uint64(imm)) if lits != nil { lits.add(name, riscvLiteralBytes(imm)) } return encodeRISCVSBLoad(&ast.Symbol{Name: name}, rd, relocs) } out := encodeRISCVLoadImm(rd, int32(c)) if lsh > 0 { out = append(out, riscvShiftImmBytes(rd, lsh, true)...) } if rsh > 0 { out = append(out, riscvShiftImmBytes(rd, rsh, false)...) } return out } // riscvShiftImmBytes encodes one SLLI (left) or SRLI expansion part, using // the compressed form the toolchain picks under RVC: C.SLLI admits every // register but X0, C.SRLI only X8 to X15. func riscvShiftImmBytes(rd, shamt int, left bool) []byte { if rd != 0 && shamt >= 1 && shamt <= 63 && (left || isRVCIntReg(rd)) { if left { return word16(rvcSLLI(uint32(rd), uint32(shamt)&0x3F)) } return word16(rvcCBShift(0x0, rvcReg3(rd), uint32(shamt)&0x3F)) } enc := riscvEnc{0x13, 0x1, 0x00} // SLLI imm := int32(shamt) if !left { enc = riscvEnc{0x13, 0x5, 0x00} // SRLI: funct6 000000, funct3 101 } return wordLE(riscvIType(enc, rd, rd, imm)) } // riscvLiteralBytes renders a 64-bit constant as the little-endian bytes the // $i64 pool symbol holds. func riscvLiteralBytes(v int64) []byte { return []byte{byte(v), byte(v >> 8), byte(v >> 16), byte(v >> 24), byte(v >> 32), byte(v >> 40), byte(v >> 48), byte(v >> 56)} } // riscvIsFloatRegOperand reports whether the operand spells a floating-point // register: every F-bank spelling starts with F, and the one integer name // that does (FP, the frame pointer alias of X8) is excluded. func riscvIsFloatRegOperand(op *ast.Operand) bool { name := "" if op.Addr.Base != "" { name = op.Addr.Base } else if op.Addr.Sym != nil { name = op.Addr.Sym.Name } return name != "FP" && strings.HasPrefix(name, "F") } // riscvFPConstBits resolves an FP constant operand to the bit pattern the // toolchain moves or pools: MOVF narrows through float32 first, MOVD keeps // the float64 bits. The bare spelling fills Imm.Float and the parenthesised // one ($ (709.78…)) leaves only the raw text, exactly as on arm64. func riscvFPConstBits(mnem string, src *ast.Operand) (uint64, bool, error) { text := src.Imm.Float if text == "" { s := strings.Join(strings.Fields(src.Raw), "") s = strings.TrimPrefix(s, "$") if strings.HasPrefix(s, "(") && strings.HasSuffix(s, ")") && strings.ContainsAny(s[1:len(s)-1], ".eE") { text = s[1 : len(s)-1] } } if text == "" { return 0, false, fmt.Errorf("not a floating-point constant") } f, err := strconv.ParseFloat(text, 64) if err != nil { return 0, false, err } if src.Imm.Neg { f = -f } if mnem == "MOVF" { return uint64(math.Float32bits(float32(f))), false, nil } return math.Float64bits(f), true, nil } // riscvExtendBytes emits the SLLI + SRAI/SRLI pair the toolchain synthesises // for the MOVB/MOVH/MOVHU/MOVWU register moves, with the per-half compression // its compress pass applies: the SLLI compresses only in place (rd == rs1), // the SRAI/SRLI only for the prime registers X8 to X15. func riscvExtendBytes(rd, rs1, shamt int, arithmetic bool) []byte { var out []byte if rd == rs1 && rd != 0 { out = word16(rvcSLLI(uint32(rd), uint32(shamt))) } else { out = wordLE(riscvIType(riscvEnc{0x13, 0x1, 0x00}, rd, rs1, int32(shamt))) } if isRVCIntReg(rd) { funct2 := uint32(0x0) if arithmetic { funct2 = 0x1 } out = append(out, word16(rvcCBShift(funct2, rvcReg3(rd), uint32(shamt)))...) } else { imm := int32(shamt) if arithmetic { imm = 0x400 | int32(shamt) // funct6 010000, the SRAI half } out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x5, 0x00}, rd, rd, imm))...) } return out } // encodeRISCVSBFPLoad emits AUIPC X31 + FLW/FLD against a pooled constant // symbol, the toolchain's form for an FP destination: the address lands in // TMP because the destination register is not an integer one. The single // R_RISCV_PCREL_ITYPE relocation covers the pair. func encodeRISCVSBFPLoad(name string, rd int, double bool, relocs *[]Reloc) []byte { if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType}) } auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, 31, 0) width := uint32(0x2) // FLW if double { width = 0x3 // FLD } fl := riscvIType(riscvEnc{0x07, width, 0x00}, rd, 31, 0) return append(wordLE(auipc), wordLE(fl)...) } // riscvTLSBytes emits the toolchain's local-exec TLS sequence for an SB // reference to a TLSBSS symbol: LUI TMP + ADDIW TMP (the 8-byte // R_RISCV_TLS_LE field the linker patches as the offset from the thread // pointer), ADD TMP, TP, TMP, then the access at zero offset through TMP. func riscvTLSBytes(enc riscvEnc, store bool, reg int, sym *ast.Symbol, relocs *[]Reloc) []byte { if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelRISCVTLSLE, Addend: sym.Offset}) } out := wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, 0)) // LUI X31, hi out = append(out, wordLE(riscvIType(riscvEnc{0x1b, 0x0, 0x00}, 31, 31, 0))...) // ADDIW X31, X31, lo out = append(out, wordLE(riscvRType(riscvInstrTable["ADD"], 31, 31, 4))...) // ADD X31, X31, X4(TP) if store { return append(out, wordLE(riscvSType(enc, 31, reg, 0))...) } return append(out, wordLE(riscvIType(enc, reg, 31, 0))...) } // RiscvLiteral is one pooled 64-bit constant: a MOV whose immediate sits // beyond both the 32-bit span and the shift sequences loads its bits from a // read-only data symbol named like the toolchain's $i64 pool. type RiscvLiteral struct { Name string Data []byte } // riscvLiterals collects the pooled constants the MOV expansions refer to, // deduplicated by name, in first-use order. type riscvLiterals struct { order []RiscvLiteral seen map[string]bool } func (l *riscvLiterals) add(name string, data []byte) { if l.seen == nil { l.seen = map[string]bool{} } if !l.seen[name] { l.seen[name] = true l.order = append(l.order, RiscvLiteral{Name: name, Data: data}) } } func (l *riscvLiterals) list() []RiscvLiteral { return l.order } // encodeRISCVItypeImmediate encodes an I-type arithmetic instruction, expanding // large immediates for ADDI/ANDI/ORI/XORI into LUI+ADDIW+op (or two ADDIs for // ADDI), matching the Go assembler. func encodeRISCVItypeImmediate(mnem string, enc riscvEnc, rd, rs1 int, imm int32) ([]byte, error) { if imm >= -2048 && imm <= 2047 { return wordLE(riscvIType(enc, rd, rs1, imm)), nil } var opMn string switch mnem { case "ADDI": opMn = "ADD" case "ANDI": opMn = "AND" case "ORI": opMn = "OR" case "XORI": opMn = "XOR" default: return nil, fmt.Errorf("%s: immediate %d does not fit 12 bits", mnem, imm) } // ADDI with a small-ish immediate splits into two ADDIs. if mnem == "ADDI" && imm >= -4096 && imm < 4095 { imm0 := imm / 2 imm1 := imm - imm0 var out []byte out = append(out, wordLE(riscvIType(enc, rd, rs1, imm0))...) out = append(out, wordLE(riscvIType(enc, rd, rd, imm1))...) return out, nil } // LUI $high, TMP; [ADDIW $low, TMP, TMP]; op TMP, rs1, rd. The LUI and // ADDIW compress to their RVC forms (C.LUI / C.ADDIW) when the immediate // fits 6 signed bits, matching the toolchain's compress pass. low, high := splitRISCV32Imm(imm) tmp := 31 // X31 = T6 = TMP var out []byte if high != 0 && high >= -32 && high <= 31 { out = append(out, word16(rvcCI(0x3, uint32(tmp), uint32(high)&0x3F))...) } else { out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, tmp, high<<12))...) } if low != 0 { if low >= -32 && low <= 31 { out = append(out, word16(rvcCI(0x1, uint32(tmp), uint32(low)&0x3F))...) } else { out = append(out, wordLE(riscvIType(riscvEnc{0x1B, 0x0, 0x00}, tmp, tmp, low))...) } } opEnc, ok := riscvInstrTable[opMn] if !ok { return nil, fmt.Errorf("%s: unsupported operation %q", mnem, opMn) } // The toolchain's compress pass runs over the expansion's instructions, // and the final ADD takes the C.ADD form whenever rd == rs1 (the two- // operand ADDI spelling); the other ops carry TMP (X31) as rs2, which // only C.ADD's full-width rs2 field can hold. if opMn == "ADD" && rd == rs1 && rd != 0 { c := rvcCR(0x9, uint32(rd), uint32(tmp)) out = append(out, byte(c), byte(c>>8)) return out, nil } out = append(out, wordLE(riscvRType(opEnc, rd, rs1, tmp))...) return out, nil } // riscvItypeImmediateSize returns the encoded byte length of an I-type // immediate instruction, accounting for the large-immediate expansion and // the C.ADD the compress pass gives the expansion's final ADD when rd == rs1. func riscvItypeImmediateSize(mnem string, rd, rs1 int, imm int32) int { if imm >= -2048 && imm <= 2047 { return 4 } switch mnem { case "ADDI", "ANDI", "ORI", "XORI": default: return 4 } if mnem == "ADDI" && imm >= -4096 && imm < 4095 { return 8 } low, high := splitRISCV32Imm(imm) // The R-type op: TMP is X31, whose full-width rs2 only C.ADD can hold, // and only when rd == rs1 (the two-operand ADDI spelling). size := 4 if mnem == "ADDI" && rd == rs1 && rd != 0 { size = 2 } if high != 0 && high >= -32 && high <= 31 { size += 2 // C.LUI } else { size += 4 // LUI } if low != 0 { if low >= -32 && low <= 31 { size += 2 // C.ADDIW } else { size += 4 // ADDIW } } return size } // encodeRISCVSBAddr emits AUIPC + ADDI to load the address of a static // symbol into rd, recording the single R_RISCV_PCREL_ITYPE relocation the Go // toolchain uses for the pair (the object-file emitters expand or map it). func encodeRISCVSBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { name := sym.Name if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType, Addend: sym.Offset}) } auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0) addi := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rd, 0) return append(wordLE(auipc), wordLE(addi)...) } // encodeRISCVSBLoad emits AUIPC + LD to load from a static symbol into rd, // recording the single R_RISCV_PCREL_ITYPE relocation for the pair. func encodeRISCVSBLoad(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { name := sym.Name if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELIType, Addend: sym.Offset}) } auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, rd, 0) ld := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rd, 0) return append(wordLE(auipc), wordLE(ld)...) } // encodeRISCVSBStore emits AUIPC + SD to store a register into a static symbol, // recording the single R_RISCV_PCREL_STYPE relocation for the pair. func encodeRISCVSBStore(sym *ast.Symbol, rs2 int, relocs *[]Reloc) []byte { tmp := 31 // X31 = T6 name := sym.Name if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 8, Name: name, Kind: RelRISCVPCRELSType, Addend: sym.Offset}) } auipc := riscvUType(riscvEnc{0x17, 0x0, 0x00}, tmp, 0) sd := riscvSType(riscvEnc{0x23, 0x3, 0x00}, tmp, rs2, 0) var out []byte out = append(out, wordLE(auipc)...) out = append(out, wordLE(sd)...) return out } // wordLE encodes a uint32 as 4 little-endian bytes. func wordLE(w uint32) []byte { return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)} } // word16 encodes a uint16 as 2 little-endian bytes. func word16(w uint16) []byte { return []byte{byte(w), byte(w >> 8)} } // encodeRISCVJALR encodes the JALR indirect jump/call instruction. // Plan 9: JALR rs1, rd (2 regs), JALR rd, offset(rs1) (the trampoline // form), or JALR offset(rs1) (memory → rd=X1). func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) { ops := instr.Operands // The toolchain's I-type validation bounds the JALR displacement to the // signed 12-bit span and rejects the rest; a wider one would truncate // silently into a jump somewhere else entirely. checkImm := func(imm int32) error { if imm < -2048 || imm > 2047 { return fmt.Errorf("JALR: signed immediate %d must be in range [-2048, 2047] (12 bits)", imm) } return nil } // JALR rd, offset(rs1): the memory operand's base is the jump-target // register, not the destination. if len(ops) == 2 && isMemOperand(ops[1]) { rd := regFromOperand(ops[0]) rs1, imm := memFromOperandWithFrame(ops[1], fi) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("JALR: invalid register operand") } if err := checkImm(imm); err != nil { return nil, err } return wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, rd, rs1, imm)), nil } if len(ops) == 2 { rs1 := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("JALR: invalid register operand") } return wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, rd, rs1, 0)), nil } if len(ops) == 1 { rs1, imm := memFromOperandWithFrame(ops[0], fi) if rs1 < 0 { return nil, fmt.Errorf("JALR: invalid memory operand") } if err := checkImm(imm); err != nil { return nil, err } return wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 1, rs1, imm)), nil } return nil, fmt.Errorf("JALR expects 1 or 2 operands, got %d", len(ops)) } // tryCompressRVC attempts to compress a RISC-V instruction to its 16-bit // RVC form. It returns the compressed instruction word and true on success. func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) { mnem := riscvCompressMnem(instr) mnem = riscvNormalisePseudo(mnem) ops := instr.Operands // The immediate aliases fold onto their I-type mnemonics before // compression: the toolchain compresses ADD $imm, rd as c.addi, exactly // as it compresses the spelling ADDI. var immNeg bool mnem, immNeg = riscvNormaliseImmAlias(mnem, ops) switch mnem { case "LD", "MOV": // LD rd, offset(SP) → C.LDSP when rd≠0 and uimm[8:3] fits. // MOV name+off(FP), rd → load, same compression. if mnem == "MOV" && len(ops) == 2 && isImmOperand(ops[0]) { return 0, false } // MOV reg, reg → C.MV (CR-type: funct4=0x8); the X0 source is ADDI // $0, X0, rd, which the toolchain's compress pass turns into C.LI $0. if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && !isMemOperand(ops[1]) && !isImmOperand(ops[0]) { rs1 := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rs1 != -1 && rd != -1 && rs1 != 0 && rd != 0 { return rvcCR(0x8, uint32(rd), uint32(rs1)), true } if rs1 == 0 && rd > 0 { return rvcCI(0x2, uint32(rd), 0), true } } rd, rs1, imm := extractLDParams(instr, fi) if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcLSP(0x3, uint32(rd), uint32(imm)), true } // Register-relative C.LD: both in prime regs, 8-byte scaled offset. if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 { return rvcCL(0x3, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true } // MOV reg, mem → store, try C.SDSP. if mnem == "MOV" && len(ops) == 2 && !isMemOperand(ops[0]) && isMemOperand(ops[1]) { rs2, rs1, imm := extractSDParams(instr, fi) if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcSSP(0x7, uint32(rs2), uint32(imm)), true } } case "SD": // SD rs2, offset(SP) → C.SDSP when uimm[8:3] fits (CSS-type). rs2, rs1, imm := extractSDParams(instr, fi) if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcSSP(0x7, uint32(rs2), uint32(imm)), true } // Register-relative C.SD: base and source in prime regs. if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 256 && imm%8 == 0 { return rvcCS(0x7, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true } case "GETCALLERPC": // The rewrite lands on MOV X1, rd (leaf), LD rd, 0(SP) (framed) or // SD X1, off(base) (leaf, memory destination); each compresses the // way its own shape does, the toolchain's compress() rules verbatim. if len(ops) == 0 { return 0, false } dst := ops[len(ops)-1] if !isMemOperand(dst) { rd := regFromOperand(dst) if fi.leaf { if rd > 0 { return rvcCR(0x8, uint32(rd), 1), true // C.MV rd, X1 } return 0, false } // LD rd, 0(SP): the SP form takes every destination but X0. if rd > 0 { return rvcLSP(0x3, uint32(rd), 0), true // C.LDSP rd, 0(SP) } return 0, false } if fi.leaf { if rs1, off := memFromOperandWithFrame(dst, fi); rs1 == 2 && off >= 0 && off < 512 && off%8 == 0 { return rvcSSP(0x7, 1, uint32(off)), true // C.SDSP X1, off } } case "LW": rd, rs1, imm := extractLDParams(instr, fi) if rs1 == 2 && rd != 0 && rd != -1 && imm >= 0 && imm < 256 && imm%4 == 0 { return rvcLSP(0x2, uint32(rd), uint32(imm)), true } if rs1 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs1) && imm >= 0 && imm < 128 && imm%4 == 0 { return rvcCL(0x2, rvcReg3(rd), rvcReg3(rs1), uint32(imm)), true } case "SW": rs2, rs1, imm := extractSDParams(instr, fi) if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 256 && imm%4 == 0 { return rvcSSP(0x6, uint32(rs2), uint32(imm)), true } if rs1 != -1 && rs2 != -1 && isRVCIntReg(rs1) && isRVCIntReg(rs2) && imm >= 0 && imm < 128 && imm%4 == 0 { return rvcCS(0x6, rvcReg3(rs2), rvcReg3(rs1), uint32(imm)), true } case "ADDI": rd, rs1, imm := extractITypeParams(instr) if immNeg { imm = -imm } if rd == -1 || rs1 == -1 { return 0, false } if rd == 2 && rs1 == 2 && imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 { // C.ADDI16SP: ADDI to SP by a nonzero 16-byte multiple. return rvcADDI16SP(2, imm), true } if rd == rs1 && rd != 0 && imm != 0 && imm >= -32 && imm <= 31 { // C.ADDI: funct3=0x0, rs1/rd, nzimm[5:0] return rvcCI(0x0, uint32(rd), uint32(imm)&0x3F), true } if isRVCIntReg(rd) && rs1 == 2 && imm != 0 && imm >= 0 && imm < 1024 && imm%4 == 0 { // C.ADDI4SPN: ADDI $imm, SP, rd for a prime rd. return rvcCIW(0x0, rvcReg3(rd), uint32(imm)), true } if rs1 == 0 && rd != 0 && imm >= -32 && imm <= 31 { // C.LI: funct3=0x2, rd, imm[5:0] return rvcCI(0x2, uint32(rd), uint32(imm)&0x3F), true } if rs1 != 0 && rd != 0 && imm == 0 { // C.MV: funct4=0x8, rd, rs1 (CR-type) return rvcCR(0x8, uint32(rd), uint32(rs1)), true } if rd == 0 && rs1 == 0 && imm == 0 { // C.NOP return 0x0001, true } case "JAL": // JAL/JMP are never compressed to C.J by the Go assembler. return 0, false case "JMP": // JAL/JMP are never compressed to C.J by the Go assembler. return 0, false case "BEQ": // Branches are never compressed to C.BEQZ/C.BNEZ. return 0, false case "BNE": // Branches are never compressed to C.BEQZ/C.BNEZ. return 0, false case "ADD": // ADD rs2, rs1, rd → C.ADD (CR-type, funct4=0x9) when rd == rs1; ADD // is commutative, so if rd == rs2, swap. ADD rs2, X0, rd is C.MV. // The two-operand form ADD rs2, rd reads rd as rs1. if len(ops) == 2 { rs2 := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rd != -1 && rs2 != -1 && rd != 0 && rs2 != 0 { return rvcCR(0x9, uint32(rd), uint32(rs2)), true } } if len(ops) == 3 { rs2 := regFromOperand(ops[0]) rs1 := regFromOperand(ops[1]) rd := regFromOperand(ops[2]) if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 { if rd == rs1 && rs2 != 0 { return rvcCR(0x9, uint32(rd), uint32(rs2)), true } if rd == rs2 && rs1 != 0 { return rvcCR(0x9, uint32(rd), uint32(rs1)), true } if rs1 == 0 && rs2 != 0 { // ADD rs2, X0, rd → C.MV rd, rs2. return rvcCR(0x8, uint32(rd), uint32(rs2)), true } } } case "SUB", "XOR", "OR", "AND": // C.SUB (0x23,0), C.XOR (0x23,1), C.OR (0x23,2), C.AND (0x23,3) funct2 := map[string]uint32{"SUB": 0x0, "XOR": 0x1, "OR": 0x2, "AND": 0x3}[mnem] if len(ops) == 2 { rs2 := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rd != -1 && rs2 != -1 && isRVCIntReg(rd) && isRVCIntReg(rs2) && rs2 != 0 { return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs2)), true } } if len(ops) == 3 { rs2 := regFromOperand(ops[0]) rs1 := regFromOperand(ops[1]) rd := regFromOperand(ops[2]) if rd != -1 && rs1 != -1 && rs2 != -1 && rd != 0 { if rd == rs1 && isRVCIntReg(rd) && isRVCIntReg(rs2) && rs2 != 0 { return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs2)), true } // AND/OR/XOR are commutative; SUB is not. if mnem != "SUB" && rd == rs2 && isRVCIntReg(rd) && isRVCIntReg(rs1) && rs1 != 0 { return rvcCA(0x23, funct2, rvcReg3(rd), rvcReg3(rs1)), true } } } case "ADDW", "SUBW": // C.ADDW (0x27,1) / C.SUBW (0x27,0), CA-type, prime regs. funct2 := uint32(0x0) if mnem == "ADDW" { funct2 = 0x1 } if len(ops) == 2 { rs2 := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rs2 != -1 && rd != -1 && isRVCIntReg(rd) && isRVCIntReg(rs2) { return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs2)), true } } if len(ops) == 3 { rs2 := regFromOperand(ops[0]) rs1 := regFromOperand(ops[1]) rd := regFromOperand(ops[2]) if rd != -1 && rs1 != -1 && rs2 != -1 && isRVCIntReg(rd) { if rd == rs1 && isRVCIntReg(rs2) { return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs2)), true } // ADDW is commutative; SUBW is not. if mnem == "ADDW" && isRVCIntReg(rs1) && rd == rs2 { return rvcCA(0x27, funct2, rvcReg3(rd), rvcReg3(rs1)), true } } } case "FLD": // FLD rd, imm(SP) → C.FLDSP (CI-type, funct3=0x1). rd, rs1, imm := extractLDParams(instr, fi) if rs1 == 2 && rd != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcLSP(0x1, uint32(rd), uint32(imm)), true } // Register-relative C.FLD: rd in F8-F15, base in X8-X15. if rs1 != -1 && rd != -1 && rd >= 8 && rd <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 { return rvcCL(0x1, uint32(rd-8), rvcReg3(rs1), uint32(imm)), true } case "FSD": // FSD rs2, imm(SP) → C.FSDSP (CSS-type, funct3=0x5). rs2, rs1, imm := extractSDParams(instr, fi) if rs1 == 2 && rs2 != -1 && imm >= 0 && imm < 512 && imm%8 == 0 { return rvcSSP(0x5, uint32(rs2), uint32(imm)), true } // Register-relative C.FSD: source in F8-F15, base in X8-X15. if rs1 != -1 && rs2 != -1 && rs2 >= 8 && rs2 <= 15 && isRVCIntReg(rs1) && imm >= 0 && imm < 256 && imm%8 == 0 { return rvcCS(0x5, uint32(rs2-8), rvcReg3(rs1), uint32(imm)), true } case "LUI": // LUI rd, imm → C.LUI when rd≠0, rd≠SP, imm nonzero and fits in six // signed bits (matching the toolchain's compress pass). if len(ops) == 2 { rd := regFromOperand(ops[0]) imm := immFromOperand(ops[1]) if rd != -1 && rd != 0 && rd != 2 && imm != 0 && imm >= -32 && imm <= 31 { return rvcCI(0x3, uint32(rd), uint32(imm)&0x3F), true } } case "ADDIW": rd, rs1, imm := extractITypeParams(instr) if immNeg { // SUBW $imm, rd arrives as ADDIW with the negated immediate. imm = -imm } if rd == rs1 && rd != 0 && imm >= -32 && imm <= 31 { return rvcCI(0x1, uint32(rd), uint32(imm)&0x3F), true } case "SLLI", "SRLI", "SRAI": rd, rs1, imm := extractITypeParams(instr) if rd == rs1 && rd != 0 && imm != 0 && imm >= 1 && imm <= 63 { if mnem == "SLLI" { // C.SLLI: funct3=0, op=10 quadrant, shamt in bits [12|6:2]. return rvcSLLI(uint32(rd), uint32(imm)&0x3F), true } if isRVCIntReg(rd) { funct2 := uint32(0x0) if mnem == "SRAI" { funct2 = 0x1 } // C.SRLI/C.SRAI: CB-type, funct3=0x4. return rvcCBShift(funct2, rvcReg3(rd), uint32(imm)&0x3F), true } } case "ANDI": rd, rs1, imm := extractITypeParams(instr) if isRVCIntReg(rd) && rd == rs1 && imm >= -32 && imm <= 31 { // C.ANDI: CB-type, funct3=0x4, funct2=0x2. return rvcCBShift(0x2, rvcReg3(rd), uint32(imm)&0x3F), true } case "EBREAK": // C.EBREAK: CR-type, funct4=0x9, rd=0, rs2=0. return rvcCR(0x9, 0, 0), true } return 0, false } // riscvCompressMnem maps a MOV-family load or store onto the base mnemonic // the toolchain lowers it to (MOVW 4(SP), X9 is LW under another name), so // the width spellings compress exactly like their base forms. Register and // immediate forms keep their own mnemonic: the C.MV path matches "MOV" and // nothing else in the switch has a width case. func riscvCompressMnem(instr *ast.Instr) string { mnem := instr.Mnemonic.Text ops := instr.Operands if !strings.HasPrefix(mnem, "MOV") || len(ops) != 2 { return mnem } load := isMemOperand(ops[0]) && !isMemOperand(ops[1]) store := !isMemOperand(ops[0]) && isMemOperand(ops[1]) if !load && !store { return mnem } switch mnem { case "MOVW": if load { return "LW" } return "SW" case "MOVF": if load { return "FLW" } return "FSW" case "MOVD": if load { return "FLD" } return "FSD" case "MOV": if load { return "LD" } return "SD" } // MOVB/MOVBU/MOVH/MOVHU/MOVWU have no compressed form; their base // mnemonics (LB/LBU/LH/LHU/LWU, SB/SH) match no case either. return mnem } // extractLDParams extracts rd, rs1, and immediate offset for a load instruction. func extractLDParams(instr *ast.Instr, fi riscvFrameInfo) (rd, rs1 int, imm int32) { ops := instr.Operands if len(ops) != 2 { return -1, -1, 0 } if instr.Mnemonic.Text == "MOV" { if isMemOperand(ops[0]) { rs1, imm = memFromOperandWithFrame(ops[0], fi) rd = regFromOperand(ops[1]) } else { return -1, -1, 0 } } else { rs1, imm = memFromOperandWithFrame(ops[0], fi) rd = regFromOperand(ops[1]) } return } // extractSDParams extracts rs2, rs1, and immediate offset for a store instruction. func extractSDParams(instr *ast.Instr, fi riscvFrameInfo) (rs2, rs1 int, imm int32) { ops := instr.Operands if len(ops) != 2 { return -1, -1, 0 } rs2 = regFromOperand(ops[0]) rs1, imm = memFromOperandWithFrame(ops[1], fi) return } // extractITypeParams extracts rd, rs1, and immediate for an I-type // instruction. The Plan 9 order is INSTR $imm, rs1, rd (3 operands) or // INSTR $imm, rd (2 operands, rd is also the source). func extractITypeParams(instr *ast.Instr) (rd, rs1 int, imm int32) { ops := instr.Operands switch len(ops) { case 3: imm = immFromOperand(ops[0]) rs1 = regFromOperand(ops[1]) rd = regFromOperand(ops[2]) case 2: imm = immFromOperand(ops[0]) rd = regFromOperand(ops[1]) rs1 = rd default: return -1, -1, 0 } return } // ---- toolchain-synthesised instructions and the RVV slice ---- // encodeRISCVExtended encodes the instructions the Go toolchain synthesises // from other instructions (ANDN/ORN, MIN/MAX, ROR and friends, the branch // pseudos and FABSD), the CSR read RDTIME, and the RVV vector slice the // compiler's kernels use. handled reports whether the mnemonic belongs to // this group; err carries the diagnostic when it does but cannot be encoded. // Each expansion reproduces the toolchain's instruction-for-instruction // sequence, including its use of X31 (TMP) and its RVC compression. func encodeRISCVExtended(mnem string, instr *ast.Instr, pc int, offsets map[string]int, pcRelPcs map[*ast.Instr]int) ([]byte, bool, error) { ops := instr.Operands switch { case isRVCInstr(mnem): code, err := encodeRISCVCompressed(mnem, instr, pc, offsets, pcRelPcs) return code, true, err } switch mnem { case "NOP": if len(ops) != 0 { return nil, true, fmt.Errorf("NOP takes no operands") } // The toolchain drops a bare NOP: no bytes at all. return nil, true, nil case "RDTIME": // RDTIME rd reads the time CSR through CSRRS with a zero source. if len(ops) != 1 { return nil, true, fmt.Errorf("RDTIME expects 1 operand, got %d", len(ops)) } rd := regFromOperand(ops[0]) if rd < 0 { return nil, true, fmt.Errorf("RDTIME: invalid register") } return wordLE(riscvIType(riscvEnc{0x73, 0x2, 0x00}, rd, 0, 0xC01)), true, nil case "NOT": // NEG and SEQZ have their own cases in encodeRISCVInstr's pseudo // switch; NOT reads as XORI $-1. if len(ops) != 1 && len(ops) != 2 { return nil, true, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops)) } rs := regFromOperand(ops[0]) rd := rs if len(ops) == 2 { rd = regFromOperand(ops[1]) } if rs < 0 || rd < 0 { return nil, true, fmt.Errorf("%s: invalid register", mnem) } var word uint32 switch mnem { case "NOT": word = riscvIType(riscvInstrTable["XORI"], rd, rs, -1) } return wordLE(word), true, nil case "ANDN", "ORN": if len(ops) != 2 && len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } rs2 := regFromOperand(ops[0]) // the operand to invert rs1 := regFromOperand(ops[1]) rd := rs1 if len(ops) == 3 { rd = regFromOperand(ops[2]) } if rs1 < 0 || rs2 < 0 || rd < 0 { return nil, true, fmt.Errorf("%s: invalid register", mnem) } notReg := rd if rs1 == notReg { notReg = 31 // TMP, when the destination would be clobbered } out := wordLE(riscvIType(riscvInstrTable["XORI"], notReg, rs2, -1)) op := riscvInstrTable["AND"] if mnem == "ORN" { op = riscvInstrTable["OR"] } return append(out, wordLE(riscvRType(op, rd, rs1, notReg))...), true, nil case "XNOR": // ~(rs1 ^ rs2): the toolchain XORs into the destination and inverts // it in place, no temporary. if len(ops) != 2 && len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } rs2 := regFromOperand(ops[0]) rs1 := regFromOperand(ops[1]) rd := rs1 if len(ops) == 3 { rd = regFromOperand(ops[2]) } if rs1 < 0 || rs2 < 0 || rd < 0 { return nil, true, fmt.Errorf("%s: invalid register", mnem) } out := wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rs2)) return append(out, wordLE(riscvIType(riscvInstrTable["XORI"], rd, rd, -1))...), true, nil case "MAX", "MAXU", "MIN", "MINU": if len(ops) != 2 && len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } rs2 := regFromOperand(ops[0]) rs1 := regFromOperand(ops[1]) rd := rs1 if len(ops) == 3 { rd = regFromOperand(ops[2]) } if rs1 < 0 || rs2 < 0 || rd < 0 { return nil, true, fmt.Errorf("%s: invalid register", mnem) } if rs1 == rd { // Process the destination-identical source first, as the // toolchain does, so the sequence stays in place. rs1, rs2 = rs2, rs1 } if rs1 == rs2 { // Identical inputs fold to ADDI $0 (compressed to C.MV and // friends by the toolchain's compressor). return riscvFoldedMove(rd, rs1), true, nil } slt1, slt2 := rs2, rs1 cmp := riscvInstrTable["SLT"] if mnem == "MAX" || mnem == "MAXU" { slt1, slt2 = slt2, slt1 } if mnem == "MAXU" || mnem == "MINU" { cmp = riscvInstrTable["SLTU"] } var out []byte out = append(out, wordLE(riscvRType(cmp, 31, slt1, slt2))...) // the compare into TMP out = append(out, wordLE(riscvRType(riscvInstrTable["SUB"], 31, 0, 31))...) // NEG TMP out = append(out, wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rs2))...) out = append(out, wordLE(riscvRType(riscvInstrTable["AND"], rd, 31, rd))...) out = append(out, wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rd))...) return out, true, nil case "BCLR", "BEXT", "BINV", "BSET": // The immediate spelling lowers to the shift-immediate entry, as the // toolchain does: BCLR $63, X24 is BCLRI $63, X24, X24. The register // spelling falls through to the main table's R-type path. if len(ops) == 0 || !isImmOperand(ops[0]) { return nil, false, nil } if len(ops) != 2 && len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } shamt, ok := riscvRawImm(ops[0]) if !ok || shamt < 0 || shamt > 63 { return nil, true, fmt.Errorf("%s: immediate out of range 0 to 63", mnem) } rs1 := regFromOperand(ops[1]) rd := rs1 if len(ops) == 3 { rd = regFromOperand(ops[2]) } if rs1 < 0 || rd < 0 { return nil, true, fmt.Errorf("%s: invalid register", mnem) } immForm := map[string]string{"BCLR": "BCLRI", "BEXT": "BEXTI", "BINV": "BINVI", "BSET": "BSETI"}[mnem] return wordLE(riscvRType(riscvInstrTable[immForm], rd, rs1, int(shamt))), true, nil case "ROL", "ROLW", "ROR", "RORI", "RORW", "RORIW": if len(ops) != 2 && len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } if isImmOperand(ops[0]) { // Immediate rotate: SRLI the amount, SLLI the complement, OR. // The immediate spellings are ROR's: ROL takes a register amount // only, as the toolchain's own expansion requires. if mnem == "ROL" || mnem == "ROLW" { return nil, true, fmt.Errorf("%s takes a register shift amount", mnem) } imm := int(immFromOperand(ops[0])) shiftW := 63 srlEnc := riscvInstrTable["SRLI"] sllEnc := riscvInstrTable["SLLI"] if mnem == "RORW" || mnem == "RORIW" { shiftW = 31 srlEnc = riscvInstrTable["SRLIW"] sllEnc = riscvInstrTable["SLLIW"] } if imm < 0 || imm > shiftW { return nil, true, fmt.Errorf("%s: shift amount out of range [0, %d]", mnem, shiftW) } rs1 := regFromOperand(ops[1]) rd := rs1 if len(ops) == 3 { rd = regFromOperand(ops[2]) } if rs1 < 0 || rd < 0 { return nil, true, fmt.Errorf("%s: invalid register", mnem) } var out []byte out = append(out, wordLE(riscvRType(srlEnc, 31, rs1, imm))...) sll := (-imm) & shiftW if mnem != "RORW" && mnem != "RORIW" && rd == rs1 && rd != 0 && sll >= 1 && sll <= 63 { out = append(out, word16(rvcSLLI(uint32(rd), uint32(sll)))...) // C.SLLI } else { out = append(out, wordLE(riscvRType(sllEnc, rd, rs1, sll))...) } return append(out, wordLE(riscvRType(riscvInstrTable["OR"], rd, 31, rd))...), true, nil } // Register rotate: OR of the two opposite shifts through TMP. RORI // and RORIW are the immediate spellings and take no register amount. if mnem == "RORIW" || mnem == "RORI" { return nil, true, fmt.Errorf("%s takes an immediate shift amount", mnem) } rs2 := regFromOperand(ops[0]) rs1 := regFromOperand(ops[1]) rd := rs1 if len(ops) == 3 { rd = regFromOperand(ops[2]) } if rs1 < 0 || rs2 < 0 || rd < 0 { return nil, true, fmt.Errorf("%s: invalid register", mnem) } // ROR shifts right by the amount and left by its complement; ROL // swaps the two. wide := mnem == "ROL" || mnem == "ROR" shiftLeft := riscvInstrTable["SLL"] shiftRight := riscvInstrTable["SRL"] shiftLeftW := riscvInstrTable["SLLW"] shiftRightW := riscvInstrTable["SRLW"] tmpShift, rdShift := shiftLeft, shiftRight if mnem == "ROL" || mnem == "ROLW" { tmpShift, rdShift = shiftRight, shiftLeft } if !wide { tmpShift, rdShift = shiftLeftW, shiftRightW if mnem == "ROLW" { tmpShift, rdShift = shiftRightW, shiftLeftW } } var out []byte out = append(out, wordLE(riscvRType(riscvInstrTable["SUB"], 31, 0, rs2))...) // NEG out = append(out, wordLE(riscvRType(tmpShift, 31, rs1, 31))...) out = append(out, wordLE(riscvRType(rdShift, rd, rs1, rs2))...) out = append(out, wordLE(riscvRType(riscvInstrTable["OR"], rd, 31, rd))...) return out, true, nil // BGT/BGTU/BLE/BLEU have no extended handler: the main table's // branch path owns them, including the N(PC) forms. case "FABSS", "FABSD", "FNEGS", "FNEGD": // INSTR fs, fd: the sign-injection pseudos, the source in both the // rs1 and rs2 fields (AFABSS: FSGNJXS rs, rs, rd; AFNEGS: FSGNJNS). // The toolchain pins the bytes: FABSS F0, F1 is 0x200020d3. if len(ops) != 2 { return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rs := regFromOperand(ops[0]) rd := regFromOperand(ops[1]) if rs < 0 || rd < 0 { return nil, true, fmt.Errorf("%s: invalid register", mnem) } enc := map[string]riscvEnc{ "FABSS": {0x53, 0x2, 0x10}, // fsgnjx.s "FABSD": {0x53, 0x2, 0x11}, // fsgnjx.d "FNEGS": {0x53, 0x1, 0x10}, // fsgnjn.s "FNEGD": {0x53, 0x1, 0x11}, // fsgnjn.d }[mnem] return wordLE(riscvRType(enc, rd, rs, rs)), true, nil case "FNES", "FNED": // INSTR fs1, fs2, xrd: the not-equal pseudos read as FEQ.S/FEQ.D // followed by XORI $1 on the result register (AFNES), two words. // The toolchain's preprocess always reads three operands: two leave // the FEQ source slot empty, and a memory destination is refused // before the validation even runs. if len(ops) == 2 { if isMemOperand(ops[1]) { return nil, true, fmt.Errorf("%s needs an integer register output", mnem) } return nil, true, fmt.Errorf("%s: expected float register in rs1 position", mnem) } if len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rs2, err := riscvWantFloatReg(mnem, "rs2", ops[0]) if err != nil { return nil, true, err } rs1, err := riscvWantFloatReg(mnem, "rs1", ops[1]) if err != nil { return nil, true, err } rd, err := riscvWantIntReg(mnem, "rd", ops[2]) if err != nil { return nil, true, err } feq := riscvEnc{0x53, 0x2, 0x50} // feq.s if mnem == "FNED" { feq = riscvEnc{0x53, 0x2, 0x51} // feq.d } eq := riscvRType(feq, rd, rs1, rs2) not := riscvIType(riscvEnc{0x13, 0x4, 0x00}, rd, rd, 1) // xori $1 return []byte{byte(eq), byte(eq >> 8), byte(eq >> 16), byte(eq >> 24), byte(not), byte(not >> 8), byte(not >> 16), byte(not >> 24)}, true, nil default: return encodeRISCVVector(mnem, ops) } } // riscvFoldedMove emits the ADDI $0, rs, rd the toolchain folds identical // MIN/MAX inputs into, with the same compression its compressor applies to // the folded form. func riscvFoldedMove(rd, rs int) []byte { switch { case rd != 0 && rs != 0: return word16(rvcCR(0x8, uint32(rd), uint32(rs))) // C.MV case rd == 0 && rs == 0: return word16(0x0001) // C.NOP case rs == 0: return word16(rvcCI(0x2, uint32(rd), 0)) // C.LI rd, $0 default: return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs, 0)) } } // isRVCInstr reports whether m is one of the explicit compressed-instruction // mnemonics: the toolchain's own spellings, encoded directly rather than // reached by compressing a 32-bit form. func isRVCInstr(m string) bool { switch m { case "CLWSP", "CLDSP", "CFLDSP", "CSWSP", "CSDSP", "CFSDSP", "CLW", "CLD", "CFLD", "CSW", "CSD", "CFSD", "CJ", "CJR", "CJALR", "CBEQZ", "CBNEZ", "CLI", "CLUI", "CADD", "CADDI", "CADDW", "CADDIW", "CADDI16SP", "CADDI4SPN", "CSLLI", "CSRLI", "CSRAI", "CANDI", "CMV", "CAND", "COR", "CXOR", "CSUB", "CSUBW", "CNOP", "CEBREAK": return true } return false } // encodeRISCVCompressed encodes one explicit RVC mnemonic to its 16-bit // halfword, with the toolchain's operand spellings and its validation: // stack-relative loads and stores pin their base to SP, the register-based // ones and the CA arithmetic to the prime registers x8-x15, and every // immediate carries its instruction's own range and scale. func encodeRISCVCompressed(mnem string, instr *ast.Instr, pc int, offsets map[string]int, pcRelPcs map[*ast.Instr]int) ([]byte, error) { ops := instr.Operands immOf := func(op *ast.Operand) (int64, error) { v, ok := riscvRawImm(op) if !ok { return 0, fmt.Errorf("%s expects an immediate", mnem) } return v, nil } // stackMem accepts a bare offset(SP) reference: the explicit compressed // stack instructions pin their base to the hardware SP, so a frame // reference (name+off(SP)) is not one. stackMem := func(op *ast.Operand) (int64, bool) { if op.Addr.Sym != nil || op.Addr.Base != "SP" { return 0, false } return op.Addr.Offset, true } branchTarget := func(op *ast.Operand) (int, error) { if op.Addr.Sym == nil && op.Addr.Base == "PC" { n := int(op.Addr.Offset) // The target lands in the final layout; pass 2 encodes ahead of // it with a placeholder, so the missing map is a range error like // any unresolved branch. if pcRelPcs == nil { return 0, &riscvRangeError{fmt.Sprintf("%s: PC-relative target %d out of range", mnem, n)} } target, ok := pcRelPcs[instr] if !ok { return 0, &riscvRangeError{fmt.Sprintf("%s: PC-relative target %d out of range", mnem, n)} } return target - pc, nil } target := labelFromOperand(op) off, ok := offsets[target] if !ok { return 0, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } return off - pc, nil } switch { // Compressed stack-pointer-based loads and stores: offset(SP), rd. case mnem == "CLWSP" || mnem == "CLDSP" || mnem == "CFLDSP": if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } off, ok := stackMem(ops[0]) if !ok { return nil, fmt.Errorf("%s: rs2 must be SP/X2", mnem) } var rd int var err error if mnem == "CFLDSP" { rd, err = riscvWantFloatReg(mnem, "rd", ops[1]) } else { rd, err = riscvWantIntReg(mnem, "rd", ops[1]) } if err != nil { return nil, err } if mnem != "CFLDSP" && rd == 0 { return nil, fmt.Errorf("%s: cannot use register X0", mnem) } scale, hi := int64(4), int64(255) funct3 := uint32(0x2) if mnem != "CLWSP" { scale, hi, funct3 = 8, 511, 0x3 } if mnem == "CFLDSP" { funct3 = 0x1 } if off < 0 || off > hi { return nil, fmt.Errorf("%s: offset %d must be in range [0, %d]", mnem, off, hi) } if off%scale != 0 { return nil, fmt.Errorf("%s: offset %d must be a multiple of %d", mnem, off, scale) } return word16(rvcLSP(funct3, uint32(rd), uint32(off))), nil case mnem == "CSWSP" || mnem == "CSDSP" || mnem == "CFSDSP": if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } var rs2 int var err error if mnem == "CFSDSP" { rs2, err = riscvWantFloatReg(mnem, "rs2", ops[0]) } else { rs2, err = riscvWantIntReg(mnem, "rs2", ops[0]) } if err != nil { return nil, err } off, ok := stackMem(ops[1]) if !ok { return nil, fmt.Errorf("%s: rd must be SP/X2", mnem) } if mnem != "CFSDSP" && rs2 == 0 { return nil, fmt.Errorf("%s: cannot use register X0", mnem) } scale, hi, funct3 := int64(4), int64(255), uint32(0x6) if mnem != "CSWSP" { scale, hi, funct3 = 8, 511, 0x7 } if mnem == "CFSDSP" { funct3 = 0x5 } if off < 0 || off > hi { return nil, fmt.Errorf("%s: offset %d must be in range [0, %d]", mnem, off, hi) } if off%scale != 0 { return nil, fmt.Errorf("%s: offset %d must be a multiple of %d", mnem, off, scale) } return word16(rvcSSP(funct3, uint32(rs2), uint32(off))), nil // Compressed register-based loads and stores: offset(rs), rd, all prime. case mnem == "CLW" || mnem == "CLD" || mnem == "CFLD": if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rs1, err := riscvWantBaseReg(mnem, "rs1", "integer prime", ops[0], riscvBankInt, 8, 15) if err != nil { return nil, err } off := ops[0].Addr.Offset var rd int if mnem == "CFLD" { rd, err = riscvWantFloatPrimeReg(mnem, "rd", ops[1]) } else { rd, err = riscvWantIntPrimeReg(mnem, "rd", ops[1]) } if err != nil { return nil, err } scale, hi, funct3 := int64(4), int64(127), uint32(0x2) if mnem != "CLW" { scale, hi, funct3 = 8, 255, 0x3 } if mnem == "CFLD" { funct3 = 0x1 } if off < 0 || off > hi { return nil, fmt.Errorf("%s: offset %d must be in range [0, %d]", mnem, off, hi) } if off%scale != 0 { return nil, fmt.Errorf("%s: offset %d must be a multiple of %d", mnem, off, scale) } return word16(rvcCL(funct3, uint32(rvcReg3(rd)), uint32(rvcReg3(rs1)), uint32(off))), nil case mnem == "CSW" || mnem == "CSD" || mnem == "CFSD": if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } var rs2 int var err error if mnem == "CFSD" { rs2, err = riscvWantFloatPrimeReg(mnem, "rs2", ops[0]) } else { rs2, err = riscvWantIntPrimeReg(mnem, "rs2", ops[0]) } if err != nil { return nil, err } rs1, err := riscvWantBaseReg(mnem, "rs1", "integer prime", ops[1], riscvBankInt, 8, 15) if err != nil { return nil, err } off := ops[1].Addr.Offset scale, hi, funct3 := int64(4), int64(127), uint32(0x6) if mnem != "CSW" { scale, hi, funct3 = 8, 255, 0x7 } if mnem == "CFSD" { funct3 = 0x5 } if off < 0 || off > hi { return nil, fmt.Errorf("%s: offset %d must be in range [0, %d]", mnem, off, hi) } if off%scale != 0 { return nil, fmt.Errorf("%s: offset %d must be a multiple of %d", mnem, off, scale) } return word16(rvcCS(funct3, uint32(rvcReg3(rs2)), uint32(rvcReg3(rs1)), uint32(off))), nil // Compressed control transfer. case mnem == "CJ" || mnem == "CBEQZ" || mnem == "CBNEZ": if mnem == "CJ" && len(ops) != 1 { return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) } if mnem != "CJ" && len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rs1 := 0 if mnem != "CJ" { var err error if rs1, err = riscvWantIntPrimeReg(mnem, "rs1", ops[0]); err != nil { return nil, err } } off, err := branchTarget(ops[len(ops)-1]) if err != nil { return nil, err } hi, lo := 2046, -2048 if mnem != "CJ" { hi, lo = 254, -256 } if off > hi || off < lo || off%2 != 0 { return nil, fmt.Errorf("%s: branch target %d out of range [%d, %d]", mnem, off, lo, hi) } if mnem == "CJ" { return word16(rvcCJ(int32(off))), nil } funct3 := uint32(0x6) if mnem == "CBNEZ" { funct3 = 0x7 } return word16(rvcCB(funct3, uint32(rvcReg3(rs1)), int32(off))), nil case mnem == "CJR" || mnem == "CJALR": if len(ops) == 2 { pos := "rs2" if mnem == "CJALR" { pos = "rd" } if err := riscvWantNoReg(mnem, pos, ops[1]); err != nil { return nil, err } } if len(ops) != 1 { return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) } rs1, err := riscvWantIntReg(mnem, "rs1", ops[0]) if err != nil { return nil, err } if rs1 == 0 { return nil, fmt.Errorf("%s: cannot use register X0 in rs1", mnem) } funct4 := uint32(0x8) if mnem == "CJALR" { funct4 = 0x9 } return word16(rvcCR(funct4, uint32(rs1), 0)), nil // Compressed constant generation. case mnem == "CLI": if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } imm, err := immOf(ops[0]) if err != nil { return nil, err } if imm < -32 || imm > 31 { return nil, fmt.Errorf("%s: immediate %d must be in range [-32, 31]", mnem, imm) } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } if rd == 0 { return nil, fmt.Errorf("%s: cannot use register X0 in rd", mnem) } return word16(rvcCI(0x2, uint32(rd), uint32(imm)&0x3F)), nil case mnem == "CLUI": if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } imm, err := immOf(ops[0]) if err != nil { return nil, err } if imm == 0 { return nil, fmt.Errorf("%s: immediate cannot be zero", mnem) } if imm < -32 || imm > 31 { return nil, fmt.Errorf("%s: immediate %d must be in range [-32, 31]", mnem, imm) } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } if rd == 0 { return nil, fmt.Errorf("%s: cannot use register X0 in rd", mnem) } if rd == 2 { return nil, fmt.Errorf("%s: cannot use register SP/X2 in rd", mnem) } return word16(rvcCI(0x3, uint32(rd), uint32(imm)&0x3F)), nil // Compressed integer register-immediate operations. case (mnem == "CADD" || mnem == "CADDI") && len(ops) >= 1 && isImmOperand(ops[0]), (mnem == "CADDW" || mnem == "CADDIW") && len(ops) >= 1 && isImmOperand(ops[0]): if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } imm, err := immOf(ops[0]) if err != nil { return nil, err } if imm < -32 || imm > 31 { return nil, fmt.Errorf("%s: immediate %d must be in range [-32, 31]", mnem, imm) } if (mnem == "CADD" || mnem == "CADDI") && imm == 0 { return nil, fmt.Errorf("%s: immediate cannot be zero", mnem) } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } if len(ops) == 3 { if rs1, err := riscvWantIntReg(mnem, "rs1", ops[2]); err != nil || rs1 != rd { return nil, fmt.Errorf("%s: rd must be the same as rs1", mnem) } } funct3 := uint32(0x0) if mnem == "CADDW" || mnem == "CADDIW" { funct3 = 0x1 } return word16(rvcCI(funct3, uint32(rd), uint32(imm)&0x3F)), nil case mnem == "CADDI16SP": if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } imm, err := immOf(ops[0]) if err != nil { return nil, err } if imm == 0 { return nil, fmt.Errorf("%s: immediate cannot be zero", mnem) } if imm < -512 || imm > 511 { return nil, fmt.Errorf("%s: immediate %d must be in range [-512, 511]", mnem, imm) } if imm%16 != 0 { return nil, fmt.Errorf("%s: immediate %d must be a multiple of 16", mnem, imm) } if n, b := riscvBankedRegNum(operandRegName(ops[1])); b != riscvBankInt || n != 2 { return nil, fmt.Errorf("%s: rd must be SP/X2", mnem) } return word16(rvcADDI16SP(2, int32(imm))), nil case mnem == "CADDI4SPN": if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } imm, err := immOf(ops[0]) if err != nil { return nil, err } if imm == 0 { return nil, fmt.Errorf("%s: immediate cannot be zero", mnem) } if imm < 0 || imm > 1023 { return nil, fmt.Errorf("%s: immediate %d must be in range [0, 1023]", mnem, imm) } if imm%4 != 0 { return nil, fmt.Errorf("%s: immediate %d must be a multiple of 4", mnem, imm) } if n, b := riscvBankedRegNum(operandRegName(ops[1])); b != riscvBankInt || n != 2 { return nil, fmt.Errorf("%s: SP/X2 must be in rs1", mnem) } rd, err := riscvWantIntPrimeReg(mnem, "rd", ops[2]) if err != nil { return nil, err } return word16(rvcCIW(0x0, uint32(rvcReg3(rd)), uint32(imm))), nil // Compressed shifts and the immediate C.ANDI: rd is the source too. // CAND with an immediate first operand is the toolchain's C.ANDI // spelling (CANDI $imm and CAND $imm encode identically). case mnem == "CSLLI" || mnem == "CSRLI" || mnem == "CSRAI" || mnem == "CANDI", mnem == "CAND" && len(ops) >= 1 && isImmOperand(ops[0]): if mnem == "CAND" { mnem = "CANDI" } if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } imm, err := immOf(ops[0]) if err != nil { return nil, err } if imm == 0 && mnem != "CANDI" { return nil, fmt.Errorf("%s: immediate cannot be zero", mnem) } if mnem == "CANDI" { if imm < -32 || imm > 31 { return nil, fmt.Errorf("%s: immediate %d must be in range [-32, 31]", mnem, imm) } } else { if imm < 0 || imm > 63 { return nil, fmt.Errorf("%s: immediate %d must be in range [0, 63]", mnem, imm) } } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } if len(ops) == 3 { if rs1, err := riscvWantIntReg(mnem, "rs1", ops[2]); err != nil || rs1 != rd { return nil, fmt.Errorf("%s: rd must be the same as rs1", mnem) } } if mnem == "CSLLI" { if rd == 0 { return nil, fmt.Errorf("%s: cannot use register X0 in rd", mnem) } return word16(rvcSLLI(uint32(rd), uint32(imm)&0x3F)), nil } if rd < 8 || rd > 15 { return nil, fmt.Errorf("%s: expected integer prime register in rd position but got non-integer prime register %s", mnem, operandRegName(ops[1])) } funct2 := uint32(0x0) switch mnem { case "CSRAI": funct2 = 0x1 case "CANDI": funct2 = 0x2 } return word16(rvcCBShift(funct2, uint32(rvcReg3(rd)), uint32(imm)&0x3F)), nil // Compressed integer register-register operations: destination last. case mnem == "CMV": if len(ops) == 3 { if err := riscvWantNoReg(mnem, "rs1", ops[2]); err != nil { return nil, err } } if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rs2, err := riscvWantIntReg(mnem, "rs2", ops[0]) if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } if rs2 == 0 { return nil, fmt.Errorf("%s: cannot use register X0 in rs2", mnem) } if rd == 0 { return nil, fmt.Errorf("%s: cannot use register X0 in rd", mnem) } return word16(rvcCR(0x8, uint32(rd), uint32(rs2))), nil case mnem == "CADD" || mnem == "CAND" || mnem == "COR" || mnem == "CXOR" || mnem == "CSUB" || mnem == "CSUBW": if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } rs2, err := riscvWantIntReg(mnem, "rs2", ops[0]) if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } if len(ops) == 3 { if rs1, err := riscvWantIntReg(mnem, "rs1", ops[2]); err != nil || rs1 != rd { return nil, fmt.Errorf("%s: rd must be the same as rs1", mnem) } } if rs2 == 0 { return nil, fmt.Errorf("%s: cannot use register X0 in rs2", mnem) } if rd == 0 { return nil, fmt.Errorf("%s: cannot use register X0 in rd", mnem) } if mnem == "CADD" { return word16(rvcCR(0x9, uint32(rd), uint32(rs2))), nil } // The CA forms carry both fields in the three-bit prime encoding. if rs2 < 8 || rs2 > 15 { return nil, fmt.Errorf("%s: expected integer prime register in rs2 position but got non-integer prime register %s", mnem, operandRegName(ops[0])) } if rd < 8 || rd > 15 { return nil, fmt.Errorf("%s: expected integer prime register in rd position but got non-integer prime register %s", mnem, operandRegName(ops[1])) } funct6 := uint32(0x23) funct2 := uint32(0x0) switch mnem { case "CAND": funct2 = 0x3 case "COR": funct2 = 0x2 case "CXOR": funct2 = 0x1 case "CSUBW": funct6 = 0x27 } return word16(rvcCA(funct6, funct2, uint32(rvcReg3(rd)), uint32(rvcReg3(rs2)))), nil case mnem == "CADDW": if len(ops) < 1 { return nil, fmt.Errorf("%s expects operands", mnem) } if isImmOperand(ops[0]) { if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } imm, err := immOf(ops[0]) if err != nil { return nil, err } if imm < -32 || imm > 31 { return nil, fmt.Errorf("%s: immediate %d must be in range [-32, 31]", mnem, imm) } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } if len(ops) == 3 { if rs1, err := riscvWantIntReg(mnem, "rs1", ops[2]); err != nil || rs1 != rd { return nil, fmt.Errorf("%s: rd must be the same as rs1", mnem) } } return word16(rvcCI(0x1, uint32(rd), uint32(imm)&0x3F)), nil } if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } rs2, err := riscvWantIntReg(mnem, "rs2", ops[0]) if err != nil { return nil, err } rd, err := riscvWantIntReg(mnem, "rd", ops[1]) if err != nil { return nil, err } if len(ops) == 3 { if rs1, err := riscvWantIntReg(mnem, "rs1", ops[2]); err != nil || rs1 != rd { return nil, fmt.Errorf("%s: rd must be the same as rs1", mnem) } } if rs2 < 8 || rs2 > 15 { return nil, fmt.Errorf("%s: expected integer prime register in rs2 position but got non-integer prime register %s", mnem, operandRegName(ops[0])) } if rd < 8 || rd > 15 { return nil, fmt.Errorf("%s: expected integer prime register in rd position but got non-integer prime register %s", mnem, operandRegName(ops[1])) } return word16(rvcCA(0x27, 0x1, uint32(rvcReg3(rd)), uint32(rvcReg3(rs2)))), nil case mnem == "CNOP": if len(ops) == 1 { if err := riscvWantNoReg(mnem, "rs2", ops[0]); err != nil { return nil, err } } if len(ops) != 0 { return nil, fmt.Errorf("%s expects no operands", mnem) } return word16(0x0001), nil case mnem == "CEBREAK": if len(ops) == 1 { if err := riscvWantNoReg(mnem, "rs2", ops[0]); err != nil { return nil, err } } if len(ops) != 0 { return nil, fmt.Errorf("%s expects no operands", mnem) } return word16(0x9002), nil } return nil, fmt.Errorf("unsupported RISC-V instruction %q", mnem) } // rvcCJ encodes a CJ-type compressed jump: the 11-bit displacement in the // order [11|4|9:8|10|6|7|3:1|5], funct3 5, op 01. func rvcCJ(off int32) uint16 { packed := encodeRVCPattern(uint32(off), []int{11, 4, 9, 8, 10, 6, 7, 3, 2, 1, 5}) return uint16((0x5 << 13) | packed<<2 | 0x1) } // rvcCB encodes a CB-type compressed branch: the 8-bit displacement in the // order [8|4:3|7:6|2:1|5], funct3 6 (C.BEQZ) or 7 (C.BNEZ), op 01. func rvcCB(funct3, rs1 uint32, off int32) uint16 { packed := encodeRVCPattern(uint32(off), []int{8, 4, 3, 7, 6, 2, 1, 5}) return uint16((funct3 << 13) | ((packed>>5)&0x7)<<10 | rs1<<7 | (packed&0x1F)<<2 | 0x1) } // encodeRISCVVector encodes the RVV slice GOROOT's kernels use. Registers // are accepted in either spelling: the vector V registers and the integer // registers share their 5-bit numbers, and the superset keeps hand-written // probes simple. handled is always true: every name reaching here is one of // the vector mnemonics. func encodeRISCVVector(mnem string, ops []*ast.Operand) ([]byte, bool, error) { reg := regFromOperand // The general vector load and store families: unit, constant-stride and // indexed, with and without segments, the fault-only-first loads and the // whole-register moves. riscvIsVecLS parses the mnemonic. if riscvIsVecLS(mnem) { return encodeRISCVVecLS(mnem, ops) } switch mnem { case "VSETVL": // INSTR rs2, rs1, rd: the register form of the configuration // setting. The toolchain writes funct7 0x40 above the standard // fields, its own disambiguator against the immediate forms. if len(ops) == 2 { // The assembler binds rs1 to the third slot, so a two-operand // VSETVL leaves rs1 unset: the toolchain's rs1 rejection. return nil, true, fmt.Errorf("%s: expected integer register in rs1 position", mnem) } if len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rs2, err := riscvWantIntReg(mnem, "rs2", ops[0]) if err != nil { return nil, true, err } rs1, err := riscvWantIntReg(mnem, "rs1", ops[1]) if err != nil { return nil, true, err } rd, err := riscvWantIntReg(mnem, "rd", ops[2]) if err != nil { return nil, true, err } return wordLE(riscvRType(riscvEnc{0x57, 0x7, 0x40}, rd, rs1, rs2)), true, nil case "VSETVLI", "VSETIVLI": // INSTR avl, vsew, vlmul, vta, vma, rd. if len(ops) != 6 { return nil, true, fmt.Errorf("%s expects 6 operands, got %d", mnem, len(ops)) } avl := 0 if isImmOperand(ops[0]) { avl = int(immFromOperand(ops[0])) if avl < 0 || avl > 31 { return nil, true, fmt.Errorf("%s: avl immediate %d must be in range [0, 31] (5 bits)", mnem, avl) } } else { avl = reg(ops[0]) if avl < 0 { return nil, true, fmt.Errorf("%s: invalid avl register", mnem) } } if mnem == "VSETIVLI" && !isImmOperand(ops[0]) { return nil, true, fmt.Errorf("%s: expected immediate value", mnem) } vsew, err := riscvVTypeToken(operandRegName(ops[1]), "E", map[string]int{"8": 0, "16": 1, "32": 2, "64": 3}) if err != nil { return nil, true, fmt.Errorf("%s: %w", mnem, err) } vlmul, err := riscvVTypeToken(operandRegName(ops[2]), "M", map[string]int{"1": 0, "2": 1, "4": 2, "8": 3, "F8": 5, "F4": 6, "F2": 7}) if err != nil { return nil, true, fmt.Errorf("%s: %w", mnem, err) } vta := 0 switch operandRegName(ops[3]) { case "TA": vta = 1 case "TU": default: return nil, true, fmt.Errorf("%s: invalid tail policy %q (want TA or TU)", mnem, operandRegName(ops[3])) } vma := 0 switch operandRegName(ops[4]) { case "MA": vma = 1 case "MU": default: return nil, true, fmt.Errorf("%s: invalid mask policy %q (want MA or MU)", mnem, operandRegName(ops[4])) } rd, err := riscvWantIntReg(mnem, "rd", ops[5]) if err != nil { return nil, true, err } // An immediate avl always encodes as vsetivli, even under the // VSETVLI spelling: the toolchain canonicalises the pair, and // `VSETVLI $15` and `VSETIVLI $15` come out byte-identical // (0xcd07f657) from GOARCH=riscv64 go tool asm. ivli := mnem == "VSETIVLI" || isImmOperand(ops[0]) return wordLE(riscvVSetEnc(ivli, avl, riscvVType(vsew, vlmul, vta, vma), rd)), true, nil } // Every remaining OP-V mnemonic the toolchain knows: the arithmetic // table, dispatched by operand class. return encodeRISCVVecOp(mnem, ops) } // encodeRISCVVecOp encodes one vector arithmetic instruction through the // extracted table. The entry's class places the operands in the rs1 and vs2 // fields, an optional V0 between the sources and the destination clears the // vm bit, and the transform classes rewrite the pseudo forms the toolchain // expands before encoding (the swapped comparisons, VNEGV and friends). func encodeRISCVVecOp(mnem string, ops []*ast.Operand) ([]byte, bool, error) { op, ok := riscvVecOps[mnem] if !ok { return nil, false, nil } // vecSrcSlot answers the register bank and field name the mnemonic's // source suffix selects (VADDVV takes a vector in the vs1 field, VADDVX // an integer in rs1, VADDVF a float in rs1) and whether that suffix // names an immediate instead. vecSrcSlot := func() (riscvRegBank, bool, string) { s := mnem // The trailing M of the mask-mandatory forms names the mask, not // the source: VADCVVM sources a vector, VADCVXM an integer, // VADCVIM an immediate, VFMERGEVFM a float. if n := len(s); n > 2 && s[n-1] == 'M' && (s[n-2] == 'V' || s[n-2] == 'X' || s[n-2] == 'I' || s[n-2] == 'F') { s = s[:n-1] } switch { case strings.HasSuffix(s, "VI") || strings.HasSuffix(s, "I"): return riscvBankNone, true, "" case strings.HasSuffix(s, "VX") || strings.HasSuffix(s, "X"): return riscvBankInt, false, "rs1" case strings.HasSuffix(s, "VF") || strings.HasSuffix(s, "F"): return riscvBankFloat, false, "rs1" } return riscvBankVec, false, "vs1" } // vecSrc reads the source operand against its suffix's bank and name. vecSrc := func(o *ast.Operand) (int, error) { bank, _, pos := vecSrcSlot() return riscvWantRegDescr(mnem, pos, bank.String(), o, bank, 0, 31) } // vecImm reads the immediate the entry's form bounds: signed five bits // [-16, 15], or unsigned [0, 31] for the shifts and slides. The // toolchain's own table labels four of the unsigned forms signed (the // narrowing shifts and clips), and the parity catalogue pins that // wording, so the label follows the mnemonic. vecImm := func(o *ast.Operand) (int32, error) { if !isImmOperand(o) { return 0, fmt.Errorf("%s expects an immediate first operand", mnem) } v := immFromOperand(o) if op.immU { label := "unsigned" switch mnem { case "VSSRLVI", "VSSRAVI", "VNCLIPUWI", "VNCLIPWI": label = "signed" } if v < 0 || v > 31 { return 0, fmt.Errorf("%s: %s immediate %d must be in range [0, 31] (5 bits)", mnem, label, v) } } else if v < -16 || v > 15 { return 0, fmt.Errorf("%s: signed immediate %d must be in range [-16, 15] (5 bits)", mnem, v) } return int32(v), nil } // vecMask reads the optional mask operand: only V0 is lawful. vecMask := func(o *ast.Operand) error { if n, b := riscvBankedRegNum(operandRegName(o)); b != riscvBankVec || n != 0 { return fmt.Errorf("%s: invalid vector mask register", mnem) } return nil } // vm carries the funct7 with the vm bit set for the unmasked form: the // toolchain ORs 1 when no V0 follows the sources. vm := func(masked bool) uint32 { if !masked { return op.funct7 | 1 } return op.funct7 } // word builds the instruction from the entry's fields. word := func(funct7 uint32, rs1Field int32, vs2 int, funct3 uint32, vd int) ([]byte, bool, error) { return wordLE(riscvVecWord(funct7, rs1Field, vs2, funct3, vd)), true, nil } // rename resolves a transform to its target table entry. rename := func(to string) (riscvVecOp, error) { t, ok := riscvVecOps[to] if !ok { return riscvVecOp{}, fmt.Errorf("%s: transform target %q not in the table", mnem, to) } return t, nil } switch op.class { case vecVV: // INSTR vs1|$imm, vs2 [, V0], vd. if len(ops) != 3 && len(ops) != 4 { return nil, true, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops)) } masked := len(ops) == 4 if masked { if err := vecMask(ops[2]); err != nil { return nil, true, err } } vs2, err := riscvWantVecReg(mnem, "vs2", ops[1]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[len(ops)-1]) if err != nil { return nil, true, err } var rs1Field int32 if op.imm { if rs1Field, err = vecImm(ops[0]); err != nil { return nil, true, err } } else { var vs1 int if vs1, err = vecSrc(ops[0]); err != nil { return nil, true, err } rs1Field = int32(vs1) } return word(vm(masked), rs1Field, vs2, op.funct3, vd) case vecMACC: // INSTR vs2, vs1 [, V0], vd: the multiply-accumulate order, the // addend in the rs1 field and the multiplicand in vs2. if len(ops) != 3 && len(ops) != 4 { return nil, true, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops)) } masked := len(ops) == 4 if masked { if err := vecMask(ops[2]); err != nil { return nil, true, err } } vs2, err := riscvWantVecReg(mnem, "vs2", ops[0]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[len(ops)-1]) if err != nil { return nil, true, err } var rs1Field int32 if op.imm { if rs1Field, err = vecImm(ops[1]); err != nil { return nil, true, err } } else { var vs1 int if vs1, err = vecSrc(ops[1]); err != nil { return nil, true, err } rs1Field = int32(vs1) } return word(vm(masked), rs1Field, vs2, op.funct3, vd) case vecSWAPVV: // VMSGT*/VMSGE*/VMFGT*/VMFGE* swap the two sources and lower to the // VMSLT*/VMSLE*/VMFLT*/VMFLE* entries the table carries. if len(ops) != 3 && len(ops) != 4 { return nil, true, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops)) } masked := len(ops) == 4 if masked { if err := vecMask(ops[2]); err != nil { return nil, true, err } } t, err := rename(map[string]string{ "VMSGTVV": "VMSLTVV", "VMSGTUVV": "VMSLTUVV", "VMSGEVV": "VMSLEVV", "VMSGEUVV": "VMSLEUVV", "VMFGTVV": "VMFLTVV", "VMFGEVV": "VMFLEVV", }[mnem]) if err != nil { return nil, true, err } vs2, err := riscvWantVecReg(mnem, "vs2", ops[0]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[len(ops)-1]) if err != nil { return nil, true, err } vs1, err := vecSrc(ops[1]) if err != nil { return nil, true, err } f7 := t.funct7 if !masked { f7 |= 1 } return word(f7, int32(vs1), vs2, t.funct3, vd) case vecSWAPVI: // VMSLTVI and the VMSGE*VI forms subtract one from the immediate and // lower to the VMSLE*/VMSGT* entries. if len(ops) != 3 && len(ops) != 4 { return nil, true, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops)) } masked := len(ops) == 4 if masked { if err := vecMask(ops[2]); err != nil { return nil, true, err } } t, err := rename(map[string]string{ "VMSLTVI": "VMSLEVI", "VMSLTUVI": "VMSLEUVI", "VMSGEVI": "VMSGTVI", "VMSGEUVI": "VMSGTUVI", }[mnem]) if err != nil { return nil, true, err } // The swap validates the immediate after the decrement, exactly as // the toolchain reports it: VMSLTVI $-16 fails on the -17 it would // encode. if !isImmOperand(ops[0]) { return nil, true, fmt.Errorf("%s expects an immediate first operand", mnem) } imm := immFromOperand(ops[0]) - 1 if imm < -16 || imm > 15 { return nil, true, fmt.Errorf("%s: signed immediate %d must be in range [-16, 15] (5 bits)", mnem, imm) } vs2, err := riscvWantVecReg(mnem, "vs2", ops[1]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[len(ops)-1]) if err != nil { return nil, true, err } f7 := t.funct7 if !masked { f7 |= 1 } return word(f7, imm, vs2, t.funct3, vd) case vecUNARY, vecM2I: // INSTR vs2 [, V0], vd: one vector source, the fixed rs1 field; the // m2i members take the destination in the integer file. A fourth // operand has no rs3 slot to fill. if len(ops) > 3 { if err := riscvWantNoReg(mnem, "rs3", ops[2]); err != nil { return nil, true, err } } if len(ops) != 2 && len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } masked := len(ops) == 3 if masked { if err := vecMask(ops[1]); err != nil { return nil, true, err } } vs2, err := riscvWantVecReg(mnem, "vs2", ops[0]) if err != nil { return nil, true, err } var vd int if mnem == "VCPOPM" || mnem == "VFIRSTM" { // The popcount and first-bit reads land in an integer register; // the mask-step family writes a vector destination. vd, err = riscvWantIntReg(mnem, "rd", ops[len(ops)-1]) } else { vd, err = riscvWantVecReg(mnem, "vd", ops[len(ops)-1]) } if err != nil { return nil, true, err } return word(vm(masked), int32(op.rs1), vs2, op.funct3, vd) case vecNEG: // VNEGV, VWCVTXXV, VWCVTUXXV and VNCVTXXW read as one-operand forms // of VRSUBVX, VWADDVX, VWADDUVX and VNSRLWX with X0 in the rs1 field. if len(ops) < 2 { return nil, true, fmt.Errorf("%s: expected vector register in vd position", mnem) } if len(ops) > 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } masked := len(ops) == 3 if masked { if err := vecMask(ops[1]); err != nil { return nil, true, err } } t, err := rename(map[string]string{ "VNEGV": "VRSUBVX", "VWCVTXXV": "VWADDVX", "VWCVTUXXV": "VWADDUVX", "VNCVTXXW": "VNSRLWX", }[mnem]) if err != nil { return nil, true, err } vs2, err := riscvWantVecReg(mnem, "vs2", ops[0]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[len(ops)-1]) if err != nil { return nil, true, err } f7 := t.funct7 if !masked { f7 |= 1 } return word(f7, 0, vs2, t.funct3, vd) case vecVNOT: // VNOTV reads as VXORVI with the all-ones immediate. if len(ops) < 2 { return nil, true, fmt.Errorf("%s: expected vector register in vd position", mnem) } if len(ops) > 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } masked := len(ops) == 3 if masked { if err := vecMask(ops[1]); err != nil { return nil, true, err } } t, err := rename("VXORVI") if err != nil { return nil, true, err } vs2, err := riscvWantVecReg(mnem, "vs2", ops[0]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[len(ops)-1]) if err != nil { return nil, true, err } f7 := t.funct7 if !masked { f7 |= 1 } return word(f7, -1, vs2, t.funct3, vd) case vecVFABS: // VFABSV and VFNEGV read as VFSGNJXVV/VFSGNJNVVV with the source in // both the rs1 and vs2 fields. if len(ops) != 2 && len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } masked := len(ops) == 3 if masked { if err := vecMask(ops[1]); err != nil { return nil, true, err } } t, err := rename(map[string]string{ "VFABSV": "VFSGNJXVV", "VFNEGV": "VFSGNJNVV", }[mnem]) if err != nil { return nil, true, err } vs2, err := riscvWantVecReg(mnem, "vs2", ops[0]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[len(ops)-1]) if err != nil { return nil, true, err } f7 := t.funct7 if !masked { f7 |= 1 } return word(f7, int32(vs2), vs2, t.funct3, vd) case vecVMVV: // INSTR vs2|xs2, vd (vmv.v.v/vmv.v.x): the source in the rs1 field, // V0 fixed in vs2, the vm bit from the table. The suffix picks the // source bank: VMVVV wants a vector, VMVVX an integer. if len(ops) != 2 { return nil, true, fmt.Errorf("%s: too many operands for instruction", mnem) } vs2, err := vecSrc(ops[0]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[1]) if err != nil { return nil, true, err } return word(op.funct7, int32(vs2), 0, op.funct3, vd) case vecVMVI: // INSTR $imm, vd (vmv.v.i): the immediate in the rs1 field, V0 in // vs2, the vm bit from the table. if len(ops) != 2 { return nil, true, fmt.Errorf("%s: too many operands for instruction", mnem) } imm, err := vecImm(ops[0]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[1]) if err != nil { return nil, true, err } return word(op.funct7, imm, 0, op.funct3, vd) case vecVFMVVF: // INSTR fs1, vd (vfmv.v.f): the scalar in the rs1 field, V0 in vs2. if len(ops) != 2 { return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } fs1, err := riscvWantFloatReg(mnem, "rs1", ops[0]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[1]) if err != nil { return nil, true, err } return word(op.funct7, int32(fs1), 0, op.funct3, vd) case vecTWO: // INSTR vs2, vd: two-operand forms with the fixed rs1 field (the // extensions and conversions, the whole-register moves, the scalar // reads). The scalar reads turn the destination around: VMVXS // reads a vector into an integer register, VFMVFS into a float. if len(ops) != 2 { return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } vs2, err := riscvWantVecReg(mnem, "vs2", ops[0]) if err != nil { return nil, true, err } var vd int switch mnem { case "VMVXS": vd, err = riscvWantIntReg(mnem, "rd", ops[1]) case "VFMVFS": vd, err = riscvWantFloatReg(mnem, "rd", ops[1]) default: vd, err = riscvWantVecReg(mnem, "vd", ops[1]) } if err != nil { return nil, true, err } return word(op.funct7, int32(op.rs1), vs2, op.funct3, vd) case vecTWOX: // INSTR xs1|fs1, vd: two-operand forms with the fixed vs2 field // (vmv.s.x and vfmv.s.f). The suffix picks the scalar's bank and // the toolchain's field name for it: rs2 in both spellings. if len(ops) != 2 { return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } vd, err := riscvWantVecReg(mnem, "vd", ops[1]) if err != nil { return nil, true, err } var rs1 int if mnem == "VFMVSF" { rs1, err = riscvWantFloatReg(mnem, "rs2", ops[0]) } else { rs1, err = riscvWantIntReg(mnem, "rs2", ops[0]) } if err != nil { return nil, true, err } return word(op.funct7, int32(rs1), int(op.rs1), op.funct3, vd) case vecADC: // INSTR vs1|$imm, vs2, V0, vd: the carry forms, the mask mandatory, // V0 rejected as the destination. if len(ops) != 4 { return nil, true, fmt.Errorf("%s: invalid vector mask register", mnem) } if err := vecMask(ops[2]); err != nil { return nil, true, err } vs2, err := riscvWantVecReg(mnem, "vs2", ops[1]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[3]) if err != nil { return nil, true, err } if vd == 0 { return nil, true, fmt.Errorf("%s: invalid destination register V0", mnem) } var rs1Field int32 if op.imm { if rs1Field, err = vecImm(ops[0]); err != nil { return nil, true, err } } else { var vs1 int if vs1, err = vecSrc(ops[0]); err != nil { return nil, true, err } rs1Field = int32(vs1) } return word(op.funct7, rs1Field, vs2, op.funct3, vd) case vecMERGE: // INSTR vs1|fs1|$imm, vs2, V0, vd: the merge forms, the mask // mandatory, V0 allowed as the destination. if len(ops) != 4 { return nil, true, fmt.Errorf("%s: invalid vector mask register", mnem) } if err := vecMask(ops[2]); err != nil { return nil, true, err } vs2, err := riscvWantVecReg(mnem, "vs2", ops[1]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[3]) if err != nil { return nil, true, err } var rs1Field int32 if op.imm { if rs1Field, err = vecImm(ops[0]); err != nil { return nil, true, err } } else { var vs1 int if vs1, err = vecSrc(ops[0]); err != nil { return nil, true, err } rs1Field = int32(vs1) } return word(op.funct7, rs1Field, vs2, op.funct3, vd) case vecVMADC: // INSTR vs1|$imm, vs2, vd: the carry-producing forms; the third // operand names the destination and may be V0. A fourth operand // has no rs3 slot to fill, exactly as the toolchain writes it. if len(ops) > 3 { if err := riscvWantNoReg(mnem, "rs3", ops[3]); err != nil { return nil, true, err } } if len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } vs2, err := riscvWantVecReg(mnem, "vs2", ops[1]) if err != nil { return nil, true, err } vd, err := riscvWantVecReg(mnem, "vd", ops[2]) if err != nil { return nil, true, err } var rs1Field int32 if op.imm { if rs1Field, err = vecImm(ops[0]); err != nil { return nil, true, err } } else { var vs1 int if vs1, err = vecSrc(ops[0]); err != nil { return nil, true, err } rs1Field = int32(vs1) } return word(op.funct7, rs1Field, vs2, op.funct3, vd) case vecMM: // INSTR vs1, vs2, vd: the mask-mask forms. VMMVM and VMNOTM take // two operands and fold the second source into the first; the vm // bit stays as the table carries it. folded := mnem == "VMMVM" || mnem == "VMNOTM" if (folded && len(ops) != 2) || (!folded && len(ops) != 3) { return nil, true, fmt.Errorf("%s expects %d operands, got %d", mnem, map[bool]int{true: 2, false: 3}[folded], len(ops)) } vs1, err := vecSrc(ops[0]) if err != nil { return nil, true, err } vs2 := vs1 if !folded { if vs2, err = riscvWantVecReg(mnem, "vs2", ops[1]); err != nil { return nil, true, err } } vd, err := riscvWantVecReg(mnem, "vd", ops[len(ops)-1]) if err != nil { return nil, true, err } return word(op.funct7, int32(vs1), vs2, op.funct3, vd) case vecVMCLR: // INSTR vd: the whole-mask clears and sets, one register in all // three fields. if len(ops) != 1 { return nil, true, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) } t, err := rename(map[string]string{ "VMCLRM": "VMXORMM", "VMSETM": "VMXNORMM", }[mnem]) if err != nil { return nil, true, err } r, err := riscvWantVecReg(mnem, "vd", ops[0]) if err != nil { return nil, true, err } return word(t.funct7, int32(r), r, t.funct3, r) case vecVID: // INSTR [V0,] vd: the element index, the mask before the destination. if len(ops) != 1 && len(ops) != 2 { return nil, true, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops)) } masked := len(ops) == 2 if masked { if err := vecMask(ops[0]); err != nil { return nil, true, err } } vd, err := riscvWantVecReg(mnem, "vd", ops[len(ops)-1]) if err != nil { return nil, true, err } return word(vm(masked), int32(op.rs1), 0, op.funct3, vd) } return nil, true, fmt.Errorf("%s: unhandled vector operand class", mnem) } // riscvVTypeToken parses a vsetvli configuration token (E8, M8, MF2 and // friends): the letter prefix selects the field and the suffix its value // through the given table. func riscvVTypeToken(name, prefix string, codes map[string]int) (int, error) { if len(name) <= len(prefix) || name[:len(prefix)] != prefix { return 0, fmt.Errorf("invalid vtype token %q (want %s)", name, prefix) } code, ok := codes[name[len(prefix):]] if !ok { return 0, fmt.Errorf("invalid vtype token %q", name) } return code, nil } // riscvVecMem reads a vector memory operand: a bare base register, the only // addressing form the vector loads and stores carry. Frame-pseudo bases are // rejected: the toolchain resolves no frame reference on the vector forms. func riscvVecMem(op *ast.Operand) (rs1 int, ok bool) { if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { return -1, false } if op.Addr.Base == "" || op.Addr.Offset != 0 { return -1, false } rs1 = riscvRegNum(op.Addr.Base) return rs1, rs1 >= 0 } // riscvVecLS is one parsed vector load/store mnemonic: the direction, the // field counts and the fixed rs2 content (0 for plain forms, the // fault-only-first marker, the mask pair's 11 or the whole-register marker). type riscvVecLS struct { load bool // true for the VL families, false for the VS families nf int // segment count minus one mop int // 0 unit, 1 indexed-ux, 2 constant-stride, 3 indexed-ox width int // 0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit ff bool // fault-only-first: the fixed rs2 field carries 16 rs2f int // fixed rs2 field: the whole-register and mask markers } // riscvVecWidths maps the width segment of a vector load/store name onto the // instruction's width field. var riscvVecWidths = map[string]int{"8": 0, "16": 5, "32": 6, "64": 7} // riscvParseVecLS parses a vector load/store mnemonic into its fields. The // families the toolchain spells: the unit, constant-stride and indexed // accesses (VLE8V, VLSE8V, VLUXEI8V, VLOXEI8V and the stores), each with its // segment variants (VLSEG2E8V, VLSSEG2E8V, VLUXSEG2EI8V, ...), the // fault-only-first loads (VLE8FFV, VLSEG2E8FFV), the whole-register moves // (VL1RV, VL2RE64V, VS8RV) and the bit-mask pair (VLMV, VSMV). func riscvParseVecLS(m string) (riscvVecLS, bool) { // The whole-register spellings and the mask pair: exact names. whole := func(load bool, nf, rs2f int) (riscvVecLS, bool) { return riscvVecLS{load: load, nf: nf, rs2f: rs2f}, true } switch m { case "VLMV": return whole(true, 0, 11) case "VSMV": return whole(false, 0, 11) case "VL1RV": return whole(true, 0, 8) case "VS1RV": return whole(false, 0, 8) case "VL2RV": return whole(true, 1, 8) case "VS2RV": return whole(false, 1, 8) case "VL4RV": return whole(true, 3, 8) case "VS4RV": return whole(false, 3, 8) case "VL8RV": return whole(true, 7, 8) case "VS8RV": return whole(false, 7, 8) } // VL{n}RE{w}V: the whole-register loads with an explicit width; the // encoding is the width-less spelling's with the width field filled. if len(m) >= 7 && m[1] == 'L' && m[2] >= '1' && m[2] <= '8' && m[3:5] == "RE" && strings.HasSuffix(m, "V") { n := int(m[2] - '0') w, ok := riscvParseVecLSWidth(m[5 : len(m)-1]) if !ok { return riscvVecLS{}, false } rs2f := 8 return riscvVecLS{load: true, nf: n - 1, width: w, rs2f: rs2f}, true } if len(m) < 4 || m[0] != 'V' || (m[1] != 'L' && m[1] != 'S') { return riscvVecLS{}, false } v := riscvVecLS{load: m[1] == 'L'} rest := m[2:] // The segment families carry the count: SEGE, SSEGE, UXSEGEI, // OXSEGEI. for _, fam := range []struct { prefix string mop int ei bool }{ {"SSEG", 2, false}, {"UXSEG", 1, true}, {"OXSEG", 3, true}, {"SEG", 0, false}, } { if !strings.HasPrefix(rest, fam.prefix) { continue } tail := rest[len(fam.prefix):] if len(tail) < 3 || tail[0] < '2' || tail[0] > '8' || tail[1] != 'E' { return riscvVecLS{}, false } v.nf = int(tail[0] - '0') v.nf-- // the field is the count minus one tail = tail[2:] if fam.ei { if !strings.HasPrefix(tail, "I") { return riscvVecLS{}, false } tail = tail[1:] } v.mop = fam.mop rest = tail break } if v.nf == 0 { // The flat families: SEV, UXEIV, OXEIV, EV. switch { case strings.HasPrefix(rest, "SE"): v.mop = 2 rest = rest[2:] case strings.HasPrefix(rest, "UXEI"): v.mop = 1 rest = rest[4:] case strings.HasPrefix(rest, "OXEI"): v.mop = 3 rest = rest[4:] case strings.HasPrefix(rest, "E"): rest = rest[1:] default: return riscvVecLS{}, false } } // The tail: V, or FFV on the fault-only-first loads. ff := false if strings.HasSuffix(rest, "FFV") { ff = v.load rest = rest[:len(rest)-3] } else if strings.HasSuffix(rest, "V") { rest = rest[:len(rest)-1] } else { return riscvVecLS{}, false } w, ok := riscvParseVecLSWidth(rest) if !ok { return riscvVecLS{}, false } v.width = w v.ff = ff if ff { v.rs2f = 16 } return v, true } // riscvParseVecLSWidth parses a vector width segment ("8", "16", "32", "64") // onto its width field. The second result reports whether the text is a // width the families carry. func riscvParseVecLSWidth(s string) (int, bool) { w, ok := riscvVecWidths[s] return w, ok } // riscvIsVecLS reports whether m is one of the vector load/store mnemonics // encodeRISCVVecLS handles. func riscvIsVecLS(m string) bool { _, ok := riscvParseVecLS(m) return ok } // encodeRISCVVecLS encodes one vector load or store. The operand shapes are // the toolchain's: (base), vd for the unit loads; (base), rs2|vs2 [, V0], vd // for the stride, indexed and segment forms with their optional V0 mask; // stores mirror them with vs3 first and (base) last. func encodeRISCVVecLS(mnem string, ops []*ast.Operand) ([]byte, bool, error) { v, ok := riscvParseVecLS(mnem) if !ok { return nil, true, fmt.Errorf("unsupported vector load/store %q", mnem) } if len(ops) < 2 { return nil, true, fmt.Errorf("%s expects at least 2 operands, got %d", mnem, len(ops)) } op := uint32(0x27) if v.load { op = 0x07 } strided := v.mop == 2 indexed := v.mop == 1 || v.mop == 3 whole := v.rs2f == 2 || v.rs2f == 8 // Split the operands: the memory end fixes one operand, the register end // the other, and a V0 beside the register end is the mask. memIdx, regIdx := 0, len(ops)-1 if !v.load { memIdx, regIdx = len(ops)-1, 0 } // The base register is an integer file register; the unit and // whole-register stores name its field rd, every other form rs1. basePos := "rs1" if !v.load && !strided && !indexed { basePos = "rd" } if ops[memIdx].Addr.Sym != nil && ops[memIdx].Addr.Sym.Pseudo != "" { return nil, true, fmt.Errorf("%s: expected integer register in %s position", mnem, basePos) } rs1, err := riscvWantBaseReg(mnem, basePos, "integer", ops[memIdx], riscvBankInt, 0, 31) if err != nil { return nil, true, err } kind := "vd" if !v.load && !strided && !indexed { // The unit and whole-register stores name the data register vs1; // loads and the stride and index families name it vd. kind = "vs1" } vd, err := riscvWantVecReg(mnem, kind, ops[regIdx]) if err != nil { return nil, true, err } rs2 := v.rs2f masked := false if whole { // The whole-register forms take no stride, index or mask. if len(ops) != 2 { return nil, true, fmt.Errorf("%s: too many operands for instruction", mnem) } } else { rs2Filled := false for _, mid := range ops[min(memIdx, regIdx)+1 : max(memIdx, regIdx)] { name := operandRegName(mid) n, b := riscvBankedRegNum(name) // The unit forms have no rs2 slot: a middle operand is the // mask, and V0 is the only one the encoding carries. if !strided && !indexed { if b != riscvBankVec || n != 0 { return nil, true, fmt.Errorf("%s: invalid vector mask register", mnem) } masked = true continue } // The strided and indexed forms fill the rs2 field first: the // stride wants an integer register, the index a vector // register. A vector beyond it is the mask, V0 alone. if !rs2Filled { if rs2 != v.rs2f { return nil, true, fmt.Errorf("%s: too many operands for instruction", mnem) } if strided { if b != riscvBankInt { return nil, true, fmt.Errorf("%s: expected integer register in rs2 position but got non-integer register %s", mnem, name) } } else { if b != riscvBankVec { return nil, true, fmt.Errorf("%s: expected vector register in vs2 position but got non-vector register %s", mnem, name) } } rs2 = n rs2Filled = true continue } if b != riscvBankVec || n != 0 { return nil, true, fmt.Errorf("%s: invalid vector mask register", mnem) } masked = true } if strided && !rs2Filled { return nil, true, fmt.Errorf("%s: expected integer register in rs2 position", mnem) } if indexed && !rs2Filled { return nil, true, fmt.Errorf("%s: expected vector register in vs2 position", mnem) } } word := uint32(v.nf&7)<<29 | uint32(v.mop&3)<<26 | uint32(rs2&0x1F)<<20 | uint32(rs1&0x1F)<<15 | uint32(v.width&7)<<12 | uint32(vd&0x1F)<<7 | op if !masked { word |= 1 << 25 } return wordLE(word), true, nil } // Instruction type classifiers. func isRTypeInstr(m string) bool { switch m { case "ADD", "SUB", "SLL", "SLT", "SLTU", "XOR", "SRL", "SRA", "OR", "AND", "ADDW", "SUBW", "SLLW", "SRLW", "SRAW", "MUL", "MULH", "MULHSU", "MULHU", "DIV", "DIVU", "REM", "REMU", "MULW", "DIVW", "DIVUW", "REMW", "REMUW", "CZEROEQZ", "CZERONEZ", // Zba address generation, Zbc carry-less multiplication and the // Zbs single-bit register forms. "ADDUW", "SH1ADD", "SH1ADDUW", "SH2ADD", "SH2ADDUW", "SH3ADD", "SH3ADDUW", "CLMUL", "CLMULH", "CLMULR", "BCLR", "BEXT", "BINV", "BSET": return true } return false } func isShiftImmInstr(m string) bool { switch m { case "SLLI", "SRLI", "SRAI", "SLLIW", "SRLIW", "SRAIW", "BCLRI", "BEXTI", "BINVI", "BSETI", "SLLIUW": return true } return false } // isZbUnaryInstr reports whether m is a Zbb one-source bit operation: a // single source register with the rs2 field fixed, spelled INSTR rs, rd. func isZbUnaryInstr(m string) bool { switch m { case "CLZ", "CLZW", "CPOP", "CPOPW", "CTZ", "CTZW", "SEXTB", "SEXTH", "ORCB", "REV8", "ZEXTH": return true } return false } // riscvZbUnaryRS2 carries the constant each Zbb unary operation fixes in the // rs2 field: the population counts, sign extensions and byte operations // address a width or a position, not a second register. CLZ, CLZW and ZEXTH // leave the field empty and are absent from the map. var riscvZbUnaryRS2 = map[string]int{ "CPOP": 2, "CPOPW": 2, "CTZ": 1, "CTZW": 1, "SEXTB": 4, "SEXTH": 5, "ORCB": 7, "REV8": 24, } // riscvShiftMax bounds a shift immediate at the instruction's width: the // doubleword forms shift 0-63, the word forms 0-31, the toolchain's own // validation boundary. func riscvShiftMax(m string) int64 { switch m { case "SLLIW", "SRLIW", "SRAIW": return 31 } return 63 } func isITypeInstr(m string) bool { switch m { case "ADDI", "ADDIW", "SLTI", "SLTIU", "XORI", "ORI", "ANDI", "JALR": return true } return false } func isLoadInstr(m string) bool { switch m { case "LB", "LH", "LW", "LD", "LBU", "LHU", "LWU": return true } return false } func isStoreInstr(m string) bool { switch m { case "SB", "SH", "SW", "SD": return true } return false } func isBranchInstr(m string) bool { switch m { case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BGT", "BLE", "BGTU", "BLEU": return true } return false } func isUTypeInstr(m string) bool { return m == "LUI" || m == "AUIPC" } func isAMOInstr(m string) bool { switch m { case "AMOSWAPW", "AMOSWAPD", "AMOADDW", "AMOADDD", "AMOANDW", "AMOANDD", "AMOORW", "AMOORD", "AMOXORW", "AMOXORD", "AMOMAXW", "AMOMAXD", "AMOMINW", "AMOMIND", "AMOMAXUW", "AMOMAXUD", "AMOMINUW", "AMOMINUD": return true } return false } func isFPArithInstr(m string) bool { switch m { case "FADDS", "FSUBS", "FMULS", "FDIVS", "FADDD", "FSUBD", "FMULD", "FDIVD", "FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD", "FSGNJS", "FSGNJXD", "FSGNJXS", "FSGNJND", "FSGNJNS", "FADDQ", "FSUBQ", "FMULQ", "FDIVQ", "FSQRTQ", "FMINQ", "FMAXQ", "FSGNJQ", "FSGNJXQ", "FSGNJNQ": return true } return false } func isFPLoadInstr(m string) bool { return m == "FLW" || m == "FLD" || m == "FLQ" } func isFPStoreInstr(m string) bool { return m == "FSW" || m == "FSD" || m == "FSQ" } func isLRInstr(m string) bool { return m == "LRW" || m == "LRD" } func isSCInstr(m string) bool { return m == "SCW" || m == "SCD" } func isFPCmpInstr(m string) bool { switch m { case "FEQS", "FLTS", "FLES", "FEQD", "FLTD", "FLED", "FEQQ", "FLTQ", "FLEQ": return true } return false } // Operand helpers. func regFromOperand(op *ast.Operand) int { // Register is in Addr.Base (from (base) syntax) or Addr.Sym.Name (bare ident). if op.Addr.Base != "" { return riscvRegNum(op.Addr.Base) } if op.Addr.Sym != nil && op.Addr.Sym.Name != "" { return riscvRegNum(op.Addr.Sym.Name) } return -1 } // riscvWantIntReg reads one integer-bank register operand, rejecting a // floating-point or vector register in its place. func riscvWantIntReg(mnem, pos string, op *ast.Operand) (int, error) { return riscvWantRegDescr(mnem, pos, "integer", op, riscvBankInt, 0, 31) } // riscvWantFloatReg reads one floating-point register operand. func riscvWantFloatReg(mnem, pos string, op *ast.Operand) (int, error) { return riscvWantRegDescr(mnem, pos, "float", op, riscvBankFloat, 0, 31) } // riscvWantVecReg reads one vector register operand. func riscvWantVecReg(mnem, pos string, op *ast.Operand) (int, error) { return riscvWantRegDescr(mnem, pos, "vector", op, riscvBankVec, 0, 31) } // riscvWantIntPrimeReg reads one integer register operand from the prime // range X8-X15, the constraint the compressed instruction fields carry. func riscvWantIntPrimeReg(mnem, pos string, op *ast.Operand) (int, error) { return riscvWantRegDescr(mnem, pos, "integer prime", op, riscvBankInt, 8, 15) } // riscvWantFloatPrimeReg reads one floating-point register operand from the // prime range F8-F15. func riscvWantFloatPrimeReg(mnem, pos string, op *ast.Operand) (int, error) { return riscvWantRegDescr(mnem, pos, "float prime", op, riscvBankFloat, 8, 15) } // riscvWantRegDescr reads one register operand against the toolchain's // wantReg contract: the bank, the prime range and the message shape // ("expected integer prime register in rd position but got non-integer // prime register X5"). The suffix only appears when the operand names a // register at all, exactly as the toolchain writes it. func riscvWantRegDescr(mnem, pos, descr string, op *ast.Operand, bank riscvRegBank, lo, hi int) (int, error) { if op.Addr.Base != "" { // A memory operand in a register slot: the toolchain reads no // register from it and writes the message without a suffix. return 0, fmt.Errorf("%s: expected %s register in %s position", mnem, descr, pos) } n, b := riscvBankedRegNum(operandRegName(op)) if n < 0 { return 0, fmt.Errorf("%s: expected %s register in %s position", mnem, descr, pos) } if b != bank || n < lo || n > hi { return 0, fmt.Errorf("%s: expected %s register in %s position but got non-%s register %s", mnem, descr, pos, descr, operandRegName(op)) } return n, nil } // riscvWantNoReg rejects a register operand where the toolchain takes none: // "expected no register in rs2 but got register X5". func riscvWantNoReg(mnem, pos string, op *ast.Operand) error { if name := operandRegName(op); name != "" { if n, _ := riscvBankedRegNum(name); n >= 0 { return fmt.Errorf("%s: expected no register in %s but got register %s", mnem, pos, name) } } return nil } // riscvMovBank answers the register bank a MOV width suffix addresses: the // FP widths read the float file, every integer width the integer file. func riscvMovBank(mnem string) riscvRegBank { if mnem == "MOVF" || mnem == "MOVD" { return riscvBankFloat } return riscvBankInt } // riscvWantMovReg reads one register operand for a MOV-family instruction, // against the bank the width suffix selects. func riscvWantMovReg(mnem, pos string, op *ast.Operand) (int, error) { bank := riscvMovBank(mnem) return riscvWantRegDescr(mnem, pos, bank.String(), op, bank, 0, 31) } // riscvWantBaseReg reads the base register of a memory-shaped operand as a // register of the given bank and range, the way the register-based // compressed loads and stores validate the base field they share with the // offset. func riscvWantBaseReg(mnem, pos, descr string, op *ast.Operand, bank riscvRegBank, lo, hi int) (int, error) { n, b := riscvBankedRegNum(op.Addr.Base) if n < 0 { return 0, fmt.Errorf("%s: expected %s register in %s position", mnem, descr, pos) } if b != bank || n < lo || n > hi { return 0, fmt.Errorf("%s: expected %s register in %s position but got non-%s register %s", mnem, descr, pos, descr, op.Addr.Base) } return n, nil } // riscvWantMemBase validates the base register of a memory operand that is // not a frame reference: the addressing forms take integer registers only. func riscvWantMemBase(mnem, pos string, op *ast.Operand) error { if op.Addr.Base == "" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "") { return nil // a frame-pseudo or symbol reference names no register } if _, b := riscvBankedRegNum(op.Addr.Base); b != riscvBankInt { return fmt.Errorf("%s: expected integer register in %s position but got non-integer register %s", mnem, pos, op.Addr.Base) } return nil } func immFromOperand(op *ast.Operand) int32 { if op.Imm.HasVal { v := op.Imm.Val if op.Imm.Neg { v = -v } return int32(v) } return 0 } // riscvRawImm reads an immediate at its full written width: the raw-data // statements (WORD, BYTE) validate against their own ranges, so a value the // source spelled wider than int32 must reach the check whole, never truncated // through an int32 read (WORD $0xffffffff is in range, WORD $0x100000000 is // not, and neither may arrive disguised as the other). func riscvRawImm(op *ast.Operand) (int64, bool) { if !op.Imm.HasVal { return 0, false } v := op.Imm.Val if op.Imm.Neg { v = -v } return v, true } // riscvImm32FromOperand reads an immediate for the MOV/I-type paths as a // signed 32-bit value. The toolchain materialises wider constants through // its SLLI expansion, which this assembler does not implement, so values // outside the int32 span are diagnosed instead of silently truncated (MOV // $0x123456789 must not assemble as $0x3456789). The neg flag carries the // SUB $imm alias, whose negated value may fit when the written one does not. func riscvImm32FromOperand(op *ast.Operand, neg bool) (int32, error) { var v int64 if op.Imm.HasVal { v = op.Imm.Val if op.Imm.Neg { v = -v } } if neg { v = -v } if int64(int32(v)) != v { return 0, fmt.Errorf("immediate %d out of range; 64-bit materialisation not supported", v) } return int32(v), nil } func memFromOperand(op *ast.Operand) (rs1 int, imm int32) { rs1 = riscvRegNum(op.Addr.Base) imm = int32(op.Addr.Offset) return } // memFromOperandWithFrame resolves a memory operand, handling FP/SP // pseudo-registers via the frame mapping. func memFromOperandWithFrame(op *ast.Operand, fi riscvFrameInfo) (rs1 int, imm int32) { // Check for a pseudo-register reference (name+offset(FP) or name+offset(SP)). if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { return riscvResolvePseudo(op.Addr.Sym, fi) } // Plain register+offset memory reference. return memFromOperand(op) } func labelFromOperand(op *ast.Operand) string { if op.Addr.Sym != nil { return op.Addr.Sym.Name } return op.Raw } // suggestLabel returns a "did you mean" suggestion for an undefined label. func suggestLabel(target string, offsets map[string]int) string { if len(offsets) == 0 { return "" } // Find the closest matching label using Levenshtein distance. bestDist := len(target) + 1 var best string for name := range offsets { dist := levenshtein(target, name) if dist < bestDist { bestDist = dist best = name } } // Only suggest if the distance is small enough. if bestDist <= 3 && bestDist < len(target)/2+1 { return fmt.Sprintf("; did you mean %q?", best) } return "" } // levenshtein computes the Levenshtein distance between two strings. func levenshtein(a, b string) int { la, lb := len(a), len(b) if la == 0 { return lb } if lb == 0 { return la } // Create a matrix of distances. prev := make([]int, lb+1) curr := make([]int, lb+1) for j := 0; j <= lb; j++ { prev[j] = j } for i := 1; i <= la; i++ { curr[0] = i for j := 1; j <= lb; j++ { cost := 1 if a[i-1] == b[j-1] { cost = 0 } curr[j] = min3(curr[j-1]+1, prev[j]+1, prev[j-1]+cost) } prev, curr = curr, prev } return prev[lb] } func min3(a, b, c int) int { if a < b { if a < c { return a } return c } if b < c { return b } return c }