feat(riscv64): GOROOT instruction shapes, DATA order and offset expressions

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-19 19:58:43 +02:00
parent 1e77e58250
commit f37f183577
7 changed files with 310 additions and 95 deletions
+20 -13
View File
@@ -440,33 +440,37 @@ type dataSym struct {
} }
// collectData gathers the file's static symbols (GLOBL) and their initial // collectData gathers the file's static symbols (GLOBL) and their initial
// contents (DATA) into byte buffers, in declaration order. // contents (DATA) into byte buffers. Two passes: the Plan 9 convention puts
// every DATA line before its symbol's GLOBL, so the symbols are registered
// before the initialisers are applied.
func collectData(f *ast.File) ([]dataSym, error) { func collectData(f *ast.File) ([]dataSym, error) {
index := map[string]int{} index := map[string]int{}
var syms []dataSym var syms []dataSym
for _, d := range f.Decls { for _, d := range f.Decls {
switch dd := d.(type) { gd, ok := d.(*ast.Globl)
case *ast.Globl: if !ok {
if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue continue
} }
name := dd.Name.Name if gd.Name == nil || gd.Name.Pseudo != "SB" {
continue
}
name := gd.Name.Name
if _, dup := index[name]; dup { if _, dup := index[name]; dup {
return nil, fmt.Errorf("duplicate GLOBL %q", name) return nil, fmt.Errorf("duplicate GLOBL %q", name)
} }
size := 0 size := 0
if dd.Size != nil && dd.Size.Imm.HasVal { if gd.Size != nil && gd.Size.Imm.HasVal {
size = int(dd.Size.Imm.Val) size = int(gd.Size.Imm.Val)
} }
index[name] = len(syms) index[name] = len(syms)
ds := dataSym{ ds := dataSym{
name: name, name: name,
pkg: dd.Name.Pkg, pkg: gd.Name.Pkg,
buf: make([]byte, size), buf: make([]byte, size),
size: size, size: size,
static: dd.Name.Static, static: gd.Name.Static,
} }
for _, f := range dd.Flags { for _, f := range gd.Flags {
switch f { switch f {
case "RODATA": case "RODATA":
ds.rodata = true ds.rodata = true
@@ -487,8 +491,12 @@ func collectData(f *ast.File) ([]dataSym, error) {
} }
} }
syms = append(syms, ds) syms = append(syms, ds)
}
case *ast.Data: for _, d := range f.Decls {
dd, ok := d.(*ast.Data)
if !ok {
continue
}
if dd.Name == nil || dd.Name.Pseudo != "SB" { if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue continue
} }
@@ -518,7 +526,6 @@ func collectData(f *ast.File) ([]dataSym, error) {
buf[off+int64(j)] = byte(v >> (8 * j)) buf[off+int64(j)] = byte(v >> (8 * j))
} }
} }
}
return syms, nil return syms, nil
} }
+143 -7
View File
@@ -135,6 +135,43 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
return out, offsets, relocs, lines, spadj, nil return out, offsets, relocs, lines, spadj, nil
} }
// riscvImmAlias maps the R-type ALU mnemonics onto their I-type immediate
// forms: the toolchain accepts ADD $imm, rj, rd and emits addi. Applied
// whenever the first operand is an immediate.
var riscvImmAlias = map[string]string{
"ADD": "ADDI",
"ADDW": "ADDIW",
"AND": "ANDI",
"OR": "ORI",
"XOR": "XORI",
"SLL": "SLLI",
"SRL": "SRLI",
"SRA": "SRAI",
"SLLW": "SLLIW",
"SRLW": "SRLIW",
"SRAW": "SRAIW",
}
// riscvNormaliseImmAlias rewrites the mnemonic to its immediate form when the
// first operand is an immediate: the toolchain accepts ADD $imm, rj, rd and
// emits addi, and SUB $imm becomes addi with the negated immediate. The
// second result reports that negation; the operand itself is left untouched
// because several passes normalise the same instruction.
func riscvNormaliseImmAlias(mnem string, ops []*ast.Operand) (string, bool) {
if len(ops) >= 2 && isImmOperand(ops[0]) {
switch strings.ToUpper(mnem) {
case "SUB":
return "ADDI", true
case "SUBW":
return "ADDIW", true
}
if alias, ok := riscvImmAlias[strings.ToUpper(mnem)]; ok {
return alias, false
}
}
return mnem, false
}
// riscvInstrSize returns the encoded size in bytes of a RISC-V instruction. // riscvInstrSize returns the encoded size in bytes of a RISC-V instruction.
// Most instructions are 4 bytes; MOV with a large immediate and I-type // Most instructions are 4 bytes; MOV with a large immediate and I-type
// arithmetic with a large immediate expand to several (possibly compressed) // arithmetic with a large immediate expand to several (possibly compressed)
@@ -142,10 +179,12 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int { func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
mnem := instr.Mnemonic.Text mnem := instr.Mnemonic.Text
ops := instr.Operands ops := instr.Operands
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
if mnem == "RET" { if mnem == "RET" {
return len(riscvReturn(fi)) return len(riscvReturn(fi))
} }
if mnem == "MOV" && len(ops) == 2 { if strings.HasPrefix(mnem, "MOV") && len(ops) == 2 {
// MOV $sym(SB), rd → 8 bytes (AUIPC + ADDI). // MOV $sym(SB), rd → 8 bytes (AUIPC + ADDI).
if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && ops[0].Imm.Sym.Pseudo == "SB" { if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && ops[0].Imm.Sym.Pseudo == "SB" {
return 8 return 8
@@ -173,7 +212,11 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
} }
// I-type arithmetic with a large immediate expands to several instructions. // I-type arithmetic with a large immediate expands to several instructions.
if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) { if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) {
return riscvItypeImmediateSize(mnem, immFromOperand(ops[0])) imm := immFromOperand(ops[0])
if immNeg {
imm = -imm
}
return riscvItypeImmediateSize(mnem, imm)
} }
return 4 return 4
} }
@@ -192,6 +235,8 @@ func isBranchLike(mnem string) bool {
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) { func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) {
mnem := instr.Mnemonic.Text mnem := instr.Mnemonic.Text
ops := instr.Operands ops := instr.Operands
var immNeg bool
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
var word uint32 var word uint32
// Handle pseudo-instructions and special cases first. // Handle pseudo-instructions and special cases first.
@@ -279,14 +324,50 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
// MOV is a pseudo-instruction that the Go assembler uses for loads, // MOV is a pseudo-instruction that the Go assembler uses for loads,
// stores, register moves and immediate loads. // stores, register moves and immediate loads. The width suffixes
case "MOV": // (MOVB/MOVH/MOVW and unsigned forms) select the access width, and
// MOVD/MOVF address the FP registers.
case "MOV", "MOVB", "MOVBU", "MOVH", "MOVHU", "MOVW", "MOVWU", "MOVF", "MOVD":
return encodeRISCVMov(instr, fi, relocs) return encodeRISCVMov(instr, fi, relocs)
// JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1). // JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1).
case "JALR": case "JALR":
return encodeRISCVJALR(instr, fi) return encodeRISCVJALR(instr, fi)
// Branch-zero pseudos: BEQZ/BNEZ compare against X0, and BLTZ/BGEZ/
// BLEZ/BGTZ reorder the register operands of BLT/BGE accordingly.
case "BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rs := regFromOperand(ops[0])
if rs < 0 {
return nil, fmt.Errorf("%s: invalid register", mnem)
}
target := labelFromOperand(ops[1])
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
var enc riscvEnc
rs1, rs2 := rs, 0
switch mnem {
case "BEQZ":
enc = riscvEnc{0x63, 0x0, 0x00} // beq rs, x0
case "BNEZ":
enc = riscvEnc{0x63, 0x1, 0x00} // bne rs, x0
case "BLTZ":
enc = riscvEnc{0x63, 0x4, 0x00} // blt rs, x0
case "BGEZ":
enc = riscvEnc{0x63, 0x5, 0x00} // bge rs, x0
case "BLEZ":
enc, rs1, rs2 = riscvEnc{0x63, 0x5, 0x00}, 0, rs // bge x0, rs
case "BGTZ":
enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, 0, rs // blt x0, rs
}
word = riscvBType(enc, rs1, rs2, int32(targetOff-pc))
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
// System instructions with no operands. // System instructions with no operands.
case "FENCE", "ECALL", "EBREAK": case "FENCE", "ECALL", "EBREAK":
enc, ok := riscvInstrTable[mnem] enc, ok := riscvInstrTable[mnem]
@@ -481,6 +562,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
// two-operand form INSTR $imm, rd uses rd as the source. // two-operand form INSTR $imm, rd uses rd as the source.
case len(ops) == 3 && isITypeInstr(mnem): case len(ops) == 3 && isITypeInstr(mnem):
imm := immFromOperand(ops[0]) // immediate imm := immFromOperand(ops[0]) // immediate
if immNeg {
imm = -imm // SUB $imm arrived through the ADDI alias
}
rs1 := regFromOperand(ops[1]) // source register rs1 := regFromOperand(ops[1]) // source register
rd := regFromOperand(ops[2]) // destination rd := regFromOperand(ops[2]) // destination
if rd < 0 || rs1 < 0 { if rd < 0 || rs1 < 0 {
@@ -490,6 +574,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
case len(ops) == 2 && isITypeInstr(mnem): case len(ops) == 2 && isITypeInstr(mnem):
imm := immFromOperand(ops[0]) imm := immFromOperand(ops[0])
if immNeg {
imm = -imm
}
rd := regFromOperand(ops[1]) rd := regFromOperand(ops[1])
if rd < 0 { if rd < 0 {
return nil, fmt.Errorf("invalid register in %s", mnem) return nil, fmt.Errorf("invalid register in %s", mnem)
@@ -626,7 +713,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
if rd < 0 || rs1 < 0 { if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("MOV load: invalid operand") return nil, fmt.Errorf("MOV load: invalid operand")
} }
return riscvFrameMemOp(riscvEnc{0x03, 0x3, 0x00}, false, rd, rs1, off), nil return riscvFrameMemOp(riscvMovEnc(strings.ToUpper(instr.Mnemonic.Text), false), false, rd, rs1, off), nil
} }
// Register → memory (store). // Register → memory (store).
@@ -643,21 +730,70 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
if rs2 < 0 || rs1 < 0 { if rs2 < 0 || rs1 < 0 {
return nil, fmt.Errorf("MOV store: invalid operand") return nil, fmt.Errorf("MOV store: invalid operand")
} }
return riscvFrameMemOp(riscvEnc{0x23, 0x3, 0x00}, true, rs2, rs1, off), nil return riscvFrameMemOp(riscvMovEnc(strings.ToUpper(instr.Mnemonic.Text), true), true, rs2, rs1, off), nil
} }
// Register → register (ADDI $0, src, dst). // Register → register: MOVD/MOVF are FP moves (fsgnj with rs2 = rs1),
// everything else is ADDI $0, src, dst.
{ {
rs1 := regFromOperand(src) rs1 := regFromOperand(src)
rd := regFromOperand(dst) rd := regFromOperand(dst)
if rd < 0 || rs1 < 0 { if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("MOV: invalid register operand") return nil, fmt.Errorf("MOV: invalid register operand")
} }
mnem := strings.ToUpper(instr.Mnemonic.Text)
if mnem == "MOVD" || mnem == "MOVF" {
op := uint32(0x20000053) // FSGNJ.S
if mnem == "MOVD" {
op = 0x22000053 // FSGNJ.D
}
return wordLE(op | uint32(rs1)<<15 | uint32(rs1)<<20 | uint32(rd)<<7), nil
}
word := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs1, 0) word := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs1, 0)
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
} }
} }
// riscvMovEnc returns the load (store=false) or store (store=true) opcode for
// a MOV-family mnemonic: the suffix selects the access width, MOVD and MOVF
// select the FP load/store opcodes, and bare MOV is the 64-bit integer form.
func riscvMovEnc(mnem string, store bool) riscvEnc {
if store {
switch mnem {
case "MOVB":
return riscvEnc{0x23, 0x0, 0x00} // SB
case "MOVH":
return riscvEnc{0x23, 0x1, 0x00} // SH
case "MOVW":
return riscvEnc{0x23, 0x2, 0x00} // SW
case "MOVF":
return riscvEnc{0x27, 0x2, 0x00} // FSW
case "MOVD":
return riscvEnc{0x27, 0x3, 0x00} // FSD
}
return riscvEnc{0x23, 0x3, 0x00} // SD
}
switch mnem {
case "MOVB":
return riscvEnc{0x03, 0x0, 0x00} // LB
case "MOVBU":
return riscvEnc{0x03, 0x4, 0x00} // LBU
case "MOVH":
return riscvEnc{0x03, 0x1, 0x00} // LH
case "MOVHU":
return riscvEnc{0x03, 0x5, 0x00} // LHU
case "MOVW":
return riscvEnc{0x03, 0x2, 0x00} // LW
case "MOVWU":
return riscvEnc{0x03, 0x6, 0x00} // LWU
case "MOVF":
return riscvEnc{0x07, 0x2, 0x00} // FLW
case "MOVD":
return riscvEnc{0x07, 0x3, 0x00} // FLD
}
return riscvEnc{0x03, 0x3, 0x00} // LD
}
// riscvFrameMemOp encodes a register-relative load (store=false, I-type // riscvFrameMemOp encodes a register-relative load (store=false, I-type
// width 0x03) or store (store=true, S-type width 0x23) of the 64-bit width // width 0x03) or store (store=true, S-type width 0x23) of the 64-bit width
// at off(rs1). Offsets beyond the signed 12-bit range materialise the // at off(rs1). Offsets beyond the signed 12-bit range materialise the
+2
View File
@@ -328,6 +328,8 @@ var riscvCvtTable = map[string]riscvCvtEnc{
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32 "FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32 "FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32 "FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64 "FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64 "FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64 "FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
+4 -2
View File
@@ -511,8 +511,10 @@ requires -p, the package path, and the installed Go toolchain).
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err) fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
return 1 return 1
} }
if len(img.Funcs) == 0 { if len(img.Funcs) == 0 && len(img.Data) == 0 {
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found") // A file with neither code nor data assembles to nothing, which is
// almost always a wrong architecture rather than an intent.
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions or GLOBL data found")
return 1 return 1
} }
for _, fn := range img.Funcs { for _, fn := range img.Funcs {
+45 -14
View File
@@ -438,25 +438,56 @@ func parseAddress(g []token.Token) ast.Address {
// Optional leading displacement before a '(' base group. A sign pushes // Optional leading displacement before a '(' base group. A sign pushes
// the parenthesis one token further out: -4(DX) has it at i+2. // the parenthesis one token further out: -4(DX) has it at i+2.
if isSignedNumber(g, i) { if isSignedNumber(g, i) {
paren := i + 1 j := i
if g[i].Kind == token.Minus || g[i].Kind == token.Plus {
paren = i + 2
}
if paren < len(g) && g[paren].Kind == token.LParen {
neg := false neg := false
if g[i].Kind == token.Minus { if g[j].Kind == token.Minus {
neg = true neg = true
i++ j++
} else if g[i].Kind == token.Plus { } else if g[j].Kind == token.Plus {
i++ j++
}
if j < len(g) && g[j].Kind == token.Number {
v := parseInt(g[j].Text)
j++
// A term may carry a *number factor: 0*8(base).
for j+1 < len(g) && g[j].Kind == token.Star && g[j+1].Kind == token.Number {
v *= parseInt(g[j+1].Text)
j += 2
} }
if i < len(g) && g[i].Kind == token.Number {
addr.Offset = parseInt(g[i].Text)
addr.HasOff = true
if neg { if neg {
addr.Offset = -addr.Offset v = -v
} }
i++ // Further +/- terms, each with its optional factor:
// 3*8+8(base), 8-4*2(base).
for {
termNeg := false
if j < len(g) && g[j].Kind == token.Minus {
termNeg = true
} else if j < len(g) && g[j].Kind == token.Plus {
} else {
break
}
if j+1 < len(g) && g[j+1].Kind == token.Number {
tv := parseInt(g[j+1].Text)
j += 2
for j+1 < len(g) && g[j].Kind == token.Star && g[j+1].Kind == token.Number {
tv *= parseInt(g[j+1].Text)
j += 2
}
if termNeg {
tv = -tv
}
v += tv
continue
}
break
}
// Commit only when the expression is followed by the base
// group; a bare number stays untouched for the caller.
if j < len(g) && g[j].Kind == token.LParen {
addr.Offset = v
addr.HasOff = true
i = j
} }
} }
} }
+36
View File
@@ -0,0 +1,36 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// GOROOT-derived shapes: the MOV width suffixes for narrow loads and the
// branch-zero pseudos. Deliberately absent: the immediate ALU aliases
// (AND/SUB $imm) and the FP memory forms, whose RVC compression the encoder
// does not reproduce yet, so a parity kernel could not hold them.
#include "textflag.h"
// func mix(x int64, y int64) int64
TEXT ·mix(SB), NOSPLIT, $0-24
MOV x+0(FP), X5
MOVWU 0(X5), X11
MOVB 1(X5), X12
MOVBU 2(X5), X13
ADD X11, X12, X14
ADD X13, X14, X15
MOV X15, ret+16(FP)
RET
// func branchy(n int64) int64
TEXT ·branchy(SB), NOSPLIT, $0-16
MOV n+0(FP), X5
BEQZ X5, zero
BNEZ X5, one
BLTZ X5, zero
BGEZ X5, one
zero:
MOV $0, X6
MOV X6, ret+8(FP)
RET
one:
MOV $1, X6
MOV X6, ret+8(FP)
RET
+1
View File
@@ -28,6 +28,7 @@ func TestGroundTruthRISCV(t *testing.T) {
"../testdata/verify/bigframe_riscv64.s", "../testdata/verify/bigframe_riscv64.s",
"../testdata/verify/guard_riscv64.s", "../testdata/verify/guard_riscv64.s",
"../testdata/verify/indirect_riscv64.s", "../testdata/verify/indirect_riscv64.s",
"../testdata/verify/misc_riscv64.s",
"trampoline_riscv64.s", "trampoline_riscv64.s",
} { } {
t.Run(path, func(t *testing.T) { t.Run(path, func(t *testing.T) {