feat(riscv64,loong64): operand tail, float DATA and honest port classification

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-21 00:44:47 +02:00
parent ec1c521187
commit 1456907000
10 changed files with 573 additions and 57 deletions
+286 -35
View File
@@ -6,6 +6,7 @@ package asm
import (
"errors"
"fmt"
"math/bits"
"slices"
"strings"
@@ -14,13 +15,14 @@ import (
// assembleRISCV assembles a RISC-V TEXT function body into machine code.
// It handles the full RV64IMAFDC instruction set including RVC compression.
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, []RiscvLiteral, error) {
fi := riscvComputeFrame(t)
prologue := riscvPrologue(fi)
guardLen, err := riscvGuardLen(fi)
if err != nil {
return nil, nil, nil, nil, nil, err
return nil, nil, nil, nil, nil, nil, err
}
lits := &riscvLiterals{}
var relocs []Reloc
var spadj []SpadjStep
@@ -74,9 +76,9 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
pc := len(prologue)
for i := range recs {
branchLike := isBranchLike(recs[i].instr.Mnemonic.Text) || riscvIsCondBranch(recs[i].instr.Mnemonic.Text)
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil, nil) // no relocs in Pass 2
code, err := encodeRISCVInstr(recs[i].instr, pc, offsets, fi, nil, nil, lits) // no relocs in Pass 2
if err != nil && !(branchLike && riscvIsRangeError(err)) {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", recs[i].instr.Mnemonic.Text, err)
}
if err != nil {
code = make([]byte, 4)
@@ -178,13 +180,20 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
}
}
if !changed {
// Capture the final pcs for the N(PC) branch forms: their target
// is the instruction N source slots away, resolved by index.
// Capture the final pcs for the N(PC) branch and jump forms: the
// target is the instruction N source slots away (N=0 the branch
// itself, N negative backwards), resolved by index against the
// final layout.
pcRelPcs = map[*ast.Instr]int{}
for i := range recs {
if _, ok := riscvPCRelOffset(recs[i].instr); ok {
pcRelPcs[recs[i].instr] = pcs[i]
n, ok := riscvPCRelOffset(recs[i].instr)
if !ok {
continue
}
if i+n < 0 || i+n >= len(recs) {
continue
}
pcRelPcs[recs[i].instr] = pcs[i+n]
}
break
}
@@ -198,7 +207,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
var out []byte
guardBytes, guardReloc, err := riscvGuard(fi)
if err != nil {
return nil, nil, nil, nil, nil, err
return nil, nil, nil, nil, nil, nil, err
}
if fi.needSplit {
out = append(out, guardBytes...)
@@ -220,11 +229,11 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
// The JMP a relaxation inserted: JAL X0 to the original target.
targetOff, ok := offsets[r.jmpTo]
if !ok {
return nil, nil, nil, nil, nil, fmt.Errorf("undefined label %q", r.jmpTo)
return nil, nil, nil, nil, nil, nil, fmt.Errorf("undefined label %q", r.jmpTo)
}
offset := int32(targetOff - pc)
if err := riscvCheckJumpOffset(r.jmpTo, offset); err != nil {
return nil, nil, nil, nil, nil, err
return nil, nil, nil, nil, nil, nil, err
}
word := riscvJType(0, offset)
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
@@ -233,7 +242,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
// JMP, always the very next instruction (offset 4).
enc, rs1, rs2, ok := riscvInvertedBranchEnc(strings.ToUpper(r.instr.Mnemonic.Text), r.instr.Operands)
if !ok {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: cannot relax branch", r.instr.Mnemonic.Text)
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: cannot relax branch", r.instr.Mnemonic.Text)
}
word := riscvBType(enc, rs1, rs2, 4)
code = []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
@@ -241,9 +250,9 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
code = r.code
default:
var err error
code, err = encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs, pcRelPcs)
code, err = encodeRISCVInstr(r.instr, pc, offsets, fi, &relocs, pcRelPcs, lits)
if err != nil {
return nil, nil, nil, nil, nil, err
return nil, nil, nil, nil, nil, nil, err
}
if c16, ok := tryCompressRVC(r.instr, fi); ok {
code = []byte{byte(c16), byte(c16 >> 8)}
@@ -270,7 +279,7 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
if fi.needSplit {
relocs = append(relocs, guardReloc)
}
return out, offsets, relocs, lines, spadj, nil
return out, offsets, relocs, lines, spadj, lits.list(), nil
}
// riscvImmAlias maps the R-type ALU mnemonics onto their I-type immediate
@@ -348,6 +357,11 @@ func riscvPadBytes(pad int) []byte {
func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
mnem := instr.Mnemonic.Text
ops := instr.Operands
mnem = riscvNormalisePseudo(mnem)
if mnem == "FUNCDATA" || mnem == "PCDATA" {
// The bookkeeping statements contribute no bytes.
return 0
}
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
if mnem == "RET" {
@@ -368,7 +382,25 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
}
// MOV $imm, rd → size depends on the immediate and RVC compression.
if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil {
return riscvMovImmSize(regFromOperand(ops[1]), immFromOperand(ops[0]))
imm := riscvOperandImm64(ops[0])
if int64(int32(imm)) != imm {
return riscvMovImm64Size(regFromOperand(ops[1]), imm)
}
return riscvMovImmSize(regFromOperand(ops[1]), int32(imm))
}
// MOV $sym+off(FP|SP), rd → the frame-adjusted offset as an ADDI,
// compressed like riscvSPAddiBytes encodes it.
if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil &&
(ops[0].Imm.Sym.Pseudo == "FP" || ops[0].Imm.Sym.Pseudo == "SP") {
rd := regFromOperand(ops[1])
_, off := riscvResolvePseudo(ops[0].Imm.Sym, fi)
if rd > 0 && off == 0 {
return 2 // C.MV rd, SP
}
if isRVCIntReg(rd) && off > 0 && off < 1024 && off%4 == 0 {
return 2 // C.ADDI4SPN
}
return riscvItypeImmediateSize("ADDI", off)
}
// Frame-relative loads and stores: a frame offset beyond the signed
// 12-bit range materialises the address in X31 first.
@@ -664,9 +696,10 @@ func riscvCheckJumpOffset(target string, off int32) error {
}
// encodeRISCVInstr encodes a single RISC-V instruction.
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int, lits *riscvLiterals) ([]byte, error) {
mnem := instr.Mnemonic.Text
ops := instr.Operands
mnem = riscvNormalisePseudo(mnem)
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
var word uint32
@@ -677,6 +710,22 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// RET = epilogue (restore LR and close the frame when present) +
// uncompressed JALR X0, 0(X1) (the toolchain never compresses RET).
return riscvReturn(fi), nil
case "FUNCDATA":
// The assembler's bookkeeping statement, the expanded form of the
// GO_ARGS and NO_LOCAL_POINTERS macros: FUNCDATA $n, sym(SB)
// contributes no bytes, exactly as the toolchain's listing shows
// (the FUNCDATA entries and the instruction after them share a PC).
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
}
return nil, nil
case "PCDATA":
// The other bookkeeping statement, the expanded form of
// GO_RESULTS_INITIALIZED: PCDATA $n, $m contributes no bytes too.
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
return nil, fmt.Errorf("PCDATA expects $n, $m")
}
return nil, nil
case "WORD":
// WORD $w lays down a raw 32-bit little-endian word.
if len(ops) != 1 {
@@ -739,6 +788,22 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
target = labelFromOperand(ops[0])
// JMP N(PC): the PC-relative slot form, resolved like the
// branches (the toolchain counts source instructions at a
// uniform 4 bytes, so JMP 0(PC) is a self-loop and JMP -3(PC)
// reaches twelve bytes back). It must be recognised before the
// indirect-register form, whose operand it resembles.
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
offset := int32(off - pc)
if err := riscvCheckJumpOffset("", offset); err != nil {
return nil, err
}
word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// JMP (X5): an indirect branch, the toolchain's JALR X0, 0(X5).
if ops[0].Addr.Sym == nil && ops[0].Addr.Base != "" {
if ops[0].Addr.Offset != 0 || ops[0].Addr.Index != "" {
@@ -751,17 +816,6 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
word = riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, rs1, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
if off, isPCRel, err := riscvPCRelTargetOff(instr, pc, pcRelPcs); isPCRel {
if err != nil {
return nil, err
}
offset := int32(off - pc)
if err := riscvCheckJumpOffset("", offset); err != nil {
return nil, err
}
word = riscvJType(0, offset)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
}
targetOff, ok := offsets[target]
if !ok {
@@ -810,7 +864,7 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// (MOVB/MOVH/MOVW and unsigned forms) select the access width, and
// MOVD/MOVF address the FP registers.
case "MOV", "MOVB", "MOVBU", "MOVH", "MOVHU", "MOVW", "MOVWU", "MOVF", "MOVD":
return encodeRISCVMov(instr, fi, relocs)
return encodeRISCVMov(instr, fi, relocs, lits)
// JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1).
case "JALR":
@@ -1316,7 +1370,7 @@ func isImmOperand(op *ast.Operand) bool {
// - MOV Rs, (Rd) register-relative store
// - MOV Rs, Rd register-to-register move (ADDI $0)
// - MOV $imm, Rd load immediate (ADDI or LUI+ADDIW)
func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) {
func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc, lits *riscvLiterals) ([]byte, error) {
ops := instr.Operands
if len(ops) != 2 {
return nil, fmt.Errorf("MOV expects 2 operands, got %d", len(ops))
@@ -1335,8 +1389,21 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
}
return encodeRISCVSBAddr(src.Imm.Sym, rd, relocs), nil
}
// MOV $sym+off(FP|SP), rd: the address of a frame slot as an
// immediate is the frame-adjusted offset against the hardware SP,
// the toolchain's ADDI $adj, SP, rd (argframe+0(FP) in the runtime's
// reflect trampolines is the spelling).
if src.Imm.Sym != nil && (src.Imm.Sym.Pseudo == "FP" || src.Imm.Sym.Pseudo == "SP") {
rd := regFromOperand(dst)
if rd < 0 {
return nil, fmt.Errorf("MOV $%s(%s): invalid destination register", src.Imm.Sym.Name, src.Imm.Sym.Pseudo)
}
_, off := riscvResolvePseudo(src.Imm.Sym, fi)
return riscvSPAddiBytes(rd, off), nil
}
// MOV $sym(FP/SP), rd, not supported: immediate symbol references
// other than SB cannot be encoded as a simple immediate.
// other than the frame pseudos cannot be encoded as a simple
// immediate.
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo != "" {
return nil, fmt.Errorf("MOV $%s(%s): unsupported immediate symbol reference (only SB is supported)", src.Imm.Sym.Name, src.Imm.Sym.Pseudo)
}
@@ -1344,11 +1411,14 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
if rd < 0 {
return nil, fmt.Errorf("MOV $imm: invalid destination register")
}
imm, err := riscvImm32FromOperand(src, false)
if err != nil {
return nil, err
imm := riscvOperandImm64(src)
if int64(int32(imm)) != imm {
// Beyond the signed 32-bit span the toolchain either builds the
// value from a shifted 32-bit part or loads it from the pooled
// $i64 constant it synthesises for the purpose.
return riscvLoadImm64(rd, imm, lits, relocs), nil
}
return encodeRISCVLoadImm(rd, imm), nil
return encodeRISCVLoadImm(rd, int32(imm)), nil
}
// Memory → register (load).
@@ -1557,6 +1627,186 @@ func splitRISCV32Imm(imm int32) (low, high int32) {
return low, high
}
// riscvNormalisePseudo rewrites the toolchain's UNDEF spelling onto EBREAK:
// the assembler accepts UNDEF where the hardware wants the trap instruction
// and emits ebreak (compressed to C.EBREAK under RVC), so every pass sees the
// canonical name.
func riscvNormalisePseudo(mnem string) string {
if strings.EqualFold(mnem, "UNDEF") {
return "EBREAK"
}
return mnem
}
// riscvOperandImm64 reads an immediate operand as a full signed 64-bit value,
// where immFromOperand would truncate to int32; the MOV immediate path uses
// it to classify the wide constants.
func riscvOperandImm64(op *ast.Operand) int64 {
if !op.Imm.HasVal {
return 0
}
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return v
}
// riscvSplitShiftConst mirrors cmd/internal/obj/riscv's splitShiftConst: it
// looks for the signed 32-bit integer a constant can be rebuilt from with a
// left shift, a left-and-right shift pair (a run of ones), or a zero-extended
// 32-bit pattern. A constant that fits none of the shapes is materialised
// from the pooled $i64 data symbol instead.
func riscvSplitShiftConst(v int64) (imm int64, lsh int, rsh int, ok bool) {
// Rebuild from a signed 32-bit integer shifted left.
lsh = bits.TrailingZeros64(uint64(v))
c := v >> lsh
if int64(int32(c)) == c {
return c, lsh, 0, true
}
// Rebuild from a small negative constant: shift left into place, then
// shift the sign-extended ones run right.
rsh = bits.LeadingZeros64(uint64(v))
ones := bits.OnesCount64((uint64(v) >> lsh) >> 11)
if rsh+ones+lsh+11 == 64 {
c = (1<<11 | ((v >> lsh) & 0x7ff)) << 52 >> 52 // sign extend 12 bits
if lsh > 0 || c != -1 {
lsh += rsh
}
return c, lsh, rsh, true
}
// Rebuild from a zero-extended signed 32-bit integer.
if int64(uint32(c)) == c {
c = int64(int32(c))
lsh, rsh = 32, 32-lsh
return c, lsh, rsh, true
}
return 0, 0, 0, false
}
// riscvSPAddiBytes encodes ADDI rd, SP, imm for the frame-address immediates
// (the MOV $sym+off(FP|SP) form), using the compressed forms the toolchain
// picks under RVC: C.ADDI4SPN for a positive 4-byte multiple that fits, C.MV
// for the zero offset, the plain ADDI otherwise.
func riscvSPAddiBytes(rd int, imm int32) []byte {
if rd != 0 && imm == 0 {
return word16(rvcCR(0x8, uint32(rd), 2)) // C.MV rd, SP
}
if isRVCIntReg(rd) && imm > 0 && imm < 1024 && imm%4 == 0 {
return word16(rvcCIW(0x0, rvcReg3(rd), uint32(imm)))
}
return wordLE(riscvIType(riscvInstrTable["ADDI"], rd, 2, imm))
}
// riscvMovImm64Size returns the encoded byte length of MOV $imm, rd when the
// immediate sits outside the signed 32-bit span: the shifted-part sequences
// of riscvLoadImm64, or the 8-byte AUIPC+LD pool load.
func riscvMovImm64Size(rd int, imm int64) int {
c, lsh, rsh, ok := riscvSplitShiftConst(imm)
if !ok {
return 8 // AUIPC + LD against the $i64 pool symbol
}
size := riscvMovImmSize(rd, int32(c))
if lsh > 0 {
size += riscvShiftImmSize(rd, true)
}
if rsh > 0 {
size += riscvShiftImmSize(rd, false)
}
return size
}
// riscvShiftImmSize returns the encoded size of one SLLI/SRLI expansion
// part: two bytes under RVC when the destination can carry a compressed
// shift (C.SLLI admits every register but X0, C.SRLI only X8 to X15), four
// otherwise.
func riscvShiftImmSize(rd int, left bool) int {
if rd != 0 && (left || isRVCIntReg(rd)) {
return 2
}
return 4
}
// riscvLoadImm64 encodes MOV $imm, rd for an immediate beyond the signed
// 32-bit span, mirroring the toolchain's instructionsForMOVConst: when a
// shifted 32-bit part rebuilds the value it emits that part (compressed like
// any written MOV) followed by the SLLI and SRLI shifts; otherwise it loads
// the constant from the pooled read-only $i64.<hex> symbol via AUIPC + LD
// and registers the literal so the data section carries its bytes.
func riscvLoadImm64(rd int, imm int64, lits *riscvLiterals, relocs *[]Reloc) []byte {
c, lsh, rsh, ok := riscvSplitShiftConst(imm)
if !ok {
name := fmt.Sprintf("$i64.%016x", uint64(imm))
if lits != nil {
lits.add(name, riscvLiteralBytes(imm))
}
return encodeRISCVSBLoad(&ast.Symbol{Name: name}, rd, relocs)
}
out := encodeRISCVLoadImm(rd, int32(c))
if lsh > 0 {
out = append(out, riscvShiftImmBytes(rd, lsh, true)...)
}
if rsh > 0 {
out = append(out, riscvShiftImmBytes(rd, rsh, false)...)
}
return out
}
// riscvShiftImmBytes encodes one SLLI (left) or SRLI expansion part, using
// the compressed form the toolchain picks under RVC: C.SLLI admits every
// register but X0, C.SRLI only X8 to X15.
func riscvShiftImmBytes(rd, shamt int, left bool) []byte {
if rd != 0 && shamt >= 1 && shamt <= 63 && (left || isRVCIntReg(rd)) {
if left {
return word16(rvcSLLI(uint32(rd), uint32(shamt)&0x3F))
}
return word16(rvcCBShift(0x0, rvcReg3(rd), uint32(shamt)&0x3F))
}
enc := riscvEnc{0x13, 0x1, 0x00} // SLLI
imm := int32(shamt)
if !left {
enc = riscvEnc{0x13, 0x5, 0x00} // SRLI: funct6 000000, funct3 101
}
return wordLE(riscvIType(enc, rd, rd, imm))
}
// riscvLiteralBytes renders a 64-bit constant as the little-endian bytes the
// $i64 pool symbol holds.
func riscvLiteralBytes(v int64) []byte {
return []byte{byte(v), byte(v >> 8), byte(v >> 16), byte(v >> 24),
byte(v >> 32), byte(v >> 40), byte(v >> 48), byte(v >> 56)}
}
// RiscvLiteral is one pooled 64-bit constant: a MOV whose immediate sits
// beyond both the 32-bit span and the shift sequences loads its bits from a
// read-only data symbol named like the toolchain's $i64 pool.
type RiscvLiteral struct {
Name string
Data []byte
}
// riscvLiterals collects the pooled constants the MOV expansions refer to,
// deduplicated by name, in first-use order.
type riscvLiterals struct {
order []RiscvLiteral
seen map[string]bool
}
func (l *riscvLiterals) add(name string, data []byte) {
if l.seen == nil {
l.seen = map[string]bool{}
}
if !l.seen[name] {
l.seen[name] = true
l.order = append(l.order, RiscvLiteral{Name: name, Data: data})
}
}
func (l *riscvLiterals) list() []RiscvLiteral { return l.order }
// encodeRISCVItypeImmediate encodes an I-type arithmetic instruction, expanding
// large immediates for ADDI/ANDI/ORI/XORI into LUI+ADDIW+op (or two ADDIs for
// ADDI), matching the Go assembler.
@@ -1734,6 +1984,7 @@ func encodeRISCVJALR(instr *ast.Instr, fi riscvFrameInfo) ([]byte, error) {
// RVC form. It returns the compressed instruction word and true on success.
func tryCompressRVC(instr *ast.Instr, fi riscvFrameInfo) (uint16, bool) {
mnem := riscvCompressMnem(instr)
mnem = riscvNormalisePseudo(mnem)
ops := instr.Operands
// The immediate aliases fold onto their I-type mnemonics before
// compression: the toolchain compresses ADD $imm, rd as c.addi, exactly