feat(riscv64): GOROOT instruction shapes, DATA order and offset expressions
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+20
-13
@@ -440,33 +440,37 @@ type dataSym struct {
|
||||
}
|
||||
|
||||
// collectData gathers the file's static symbols (GLOBL) and their initial
|
||||
// contents (DATA) into byte buffers, in declaration order.
|
||||
// contents (DATA) into byte buffers. Two passes: the Plan 9 convention puts
|
||||
// every DATA line before its symbol's GLOBL, so the symbols are registered
|
||||
// before the initialisers are applied.
|
||||
func collectData(f *ast.File) ([]dataSym, error) {
|
||||
index := map[string]int{}
|
||||
var syms []dataSym
|
||||
for _, d := range f.Decls {
|
||||
switch dd := d.(type) {
|
||||
case *ast.Globl:
|
||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||
gd, ok := d.(*ast.Globl)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
name := dd.Name.Name
|
||||
if gd.Name == nil || gd.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
name := gd.Name.Name
|
||||
if _, dup := index[name]; dup {
|
||||
return nil, fmt.Errorf("duplicate GLOBL %q", name)
|
||||
}
|
||||
size := 0
|
||||
if dd.Size != nil && dd.Size.Imm.HasVal {
|
||||
size = int(dd.Size.Imm.Val)
|
||||
if gd.Size != nil && gd.Size.Imm.HasVal {
|
||||
size = int(gd.Size.Imm.Val)
|
||||
}
|
||||
index[name] = len(syms)
|
||||
ds := dataSym{
|
||||
name: name,
|
||||
pkg: dd.Name.Pkg,
|
||||
pkg: gd.Name.Pkg,
|
||||
buf: make([]byte, size),
|
||||
size: size,
|
||||
static: dd.Name.Static,
|
||||
static: gd.Name.Static,
|
||||
}
|
||||
for _, f := range dd.Flags {
|
||||
for _, f := range gd.Flags {
|
||||
switch f {
|
||||
case "RODATA":
|
||||
ds.rodata = true
|
||||
@@ -487,8 +491,12 @@ func collectData(f *ast.File) ([]dataSym, error) {
|
||||
}
|
||||
}
|
||||
syms = append(syms, ds)
|
||||
|
||||
case *ast.Data:
|
||||
}
|
||||
for _, d := range f.Decls {
|
||||
dd, ok := d.(*ast.Data)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if dd.Name == nil || dd.Name.Pseudo != "SB" {
|
||||
continue
|
||||
}
|
||||
@@ -518,7 +526,6 @@ func collectData(f *ast.File) ([]dataSym, error) {
|
||||
buf[off+int64(j)] = byte(v >> (8 * j))
|
||||
}
|
||||
}
|
||||
}
|
||||
return syms, nil
|
||||
}
|
||||
|
||||
|
||||
+143
-7
@@ -135,6 +135,43 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
|
||||
return out, offsets, relocs, lines, spadj, nil
|
||||
}
|
||||
|
||||
// riscvImmAlias maps the R-type ALU mnemonics onto their I-type immediate
|
||||
// forms: the toolchain accepts ADD $imm, rj, rd and emits addi. Applied
|
||||
// whenever the first operand is an immediate.
|
||||
var riscvImmAlias = map[string]string{
|
||||
"ADD": "ADDI",
|
||||
"ADDW": "ADDIW",
|
||||
"AND": "ANDI",
|
||||
"OR": "ORI",
|
||||
"XOR": "XORI",
|
||||
"SLL": "SLLI",
|
||||
"SRL": "SRLI",
|
||||
"SRA": "SRAI",
|
||||
"SLLW": "SLLIW",
|
||||
"SRLW": "SRLIW",
|
||||
"SRAW": "SRAIW",
|
||||
}
|
||||
|
||||
// riscvNormaliseImmAlias rewrites the mnemonic to its immediate form when the
|
||||
// first operand is an immediate: the toolchain accepts ADD $imm, rj, rd and
|
||||
// emits addi, and SUB $imm becomes addi with the negated immediate. The
|
||||
// second result reports that negation; the operand itself is left untouched
|
||||
// because several passes normalise the same instruction.
|
||||
func riscvNormaliseImmAlias(mnem string, ops []*ast.Operand) (string, bool) {
|
||||
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
||||
switch strings.ToUpper(mnem) {
|
||||
case "SUB":
|
||||
return "ADDI", true
|
||||
case "SUBW":
|
||||
return "ADDIW", true
|
||||
}
|
||||
if alias, ok := riscvImmAlias[strings.ToUpper(mnem)]; ok {
|
||||
return alias, false
|
||||
}
|
||||
}
|
||||
return mnem, false
|
||||
}
|
||||
|
||||
// riscvInstrSize returns the encoded size in bytes of a RISC-V instruction.
|
||||
// Most instructions are 4 bytes; MOV with a large immediate and I-type
|
||||
// arithmetic with a large immediate expand to several (possibly compressed)
|
||||
@@ -142,10 +179,12 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
|
||||
func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
|
||||
mnem := instr.Mnemonic.Text
|
||||
ops := instr.Operands
|
||||
var immNeg bool
|
||||
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
|
||||
if mnem == "RET" {
|
||||
return len(riscvReturn(fi))
|
||||
}
|
||||
if mnem == "MOV" && len(ops) == 2 {
|
||||
if strings.HasPrefix(mnem, "MOV") && len(ops) == 2 {
|
||||
// MOV $sym(SB), rd → 8 bytes (AUIPC + ADDI).
|
||||
if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && ops[0].Imm.Sym.Pseudo == "SB" {
|
||||
return 8
|
||||
@@ -173,7 +212,11 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
|
||||
}
|
||||
// I-type arithmetic with a large immediate expands to several instructions.
|
||||
if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) {
|
||||
return riscvItypeImmediateSize(mnem, immFromOperand(ops[0]))
|
||||
imm := immFromOperand(ops[0])
|
||||
if immNeg {
|
||||
imm = -imm
|
||||
}
|
||||
return riscvItypeImmediateSize(mnem, imm)
|
||||
}
|
||||
return 4
|
||||
}
|
||||
@@ -192,6 +235,8 @@ func isBranchLike(mnem string) bool {
|
||||
func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) {
|
||||
mnem := instr.Mnemonic.Text
|
||||
ops := instr.Operands
|
||||
var immNeg bool
|
||||
mnem, immNeg = riscvNormaliseImmAlias(mnem, ops)
|
||||
var word uint32
|
||||
|
||||
// Handle pseudo-instructions and special cases first.
|
||||
@@ -279,14 +324,50 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
|
||||
// MOV is a pseudo-instruction that the Go assembler uses for loads,
|
||||
// stores, register moves and immediate loads.
|
||||
case "MOV":
|
||||
// stores, register moves and immediate loads. The width suffixes
|
||||
// (MOVB/MOVH/MOVW and unsigned forms) select the access width, and
|
||||
// MOVD/MOVF address the FP registers.
|
||||
case "MOV", "MOVB", "MOVBU", "MOVH", "MOVHU", "MOVW", "MOVWU", "MOVF", "MOVD":
|
||||
return encodeRISCVMov(instr, fi, relocs)
|
||||
|
||||
// JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1).
|
||||
case "JALR":
|
||||
return encodeRISCVJALR(instr, fi)
|
||||
|
||||
// Branch-zero pseudos: BEQZ/BNEZ compare against X0, and BLTZ/BGEZ/
|
||||
// BLEZ/BGTZ reorder the register operands of BLT/BGE accordingly.
|
||||
case "BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
rs := regFromOperand(ops[0])
|
||||
if rs < 0 {
|
||||
return nil, fmt.Errorf("%s: invalid register", mnem)
|
||||
}
|
||||
target := labelFromOperand(ops[1])
|
||||
targetOff, ok := offsets[target]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||||
}
|
||||
var enc riscvEnc
|
||||
rs1, rs2 := rs, 0
|
||||
switch mnem {
|
||||
case "BEQZ":
|
||||
enc = riscvEnc{0x63, 0x0, 0x00} // beq rs, x0
|
||||
case "BNEZ":
|
||||
enc = riscvEnc{0x63, 0x1, 0x00} // bne rs, x0
|
||||
case "BLTZ":
|
||||
enc = riscvEnc{0x63, 0x4, 0x00} // blt rs, x0
|
||||
case "BGEZ":
|
||||
enc = riscvEnc{0x63, 0x5, 0x00} // bge rs, x0
|
||||
case "BLEZ":
|
||||
enc, rs1, rs2 = riscvEnc{0x63, 0x5, 0x00}, 0, rs // bge x0, rs
|
||||
case "BGTZ":
|
||||
enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, 0, rs // blt x0, rs
|
||||
}
|
||||
word = riscvBType(enc, rs1, rs2, int32(targetOff-pc))
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
|
||||
// System instructions with no operands.
|
||||
case "FENCE", "ECALL", "EBREAK":
|
||||
enc, ok := riscvInstrTable[mnem]
|
||||
@@ -481,6 +562,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
// two-operand form INSTR $imm, rd uses rd as the source.
|
||||
case len(ops) == 3 && isITypeInstr(mnem):
|
||||
imm := immFromOperand(ops[0]) // immediate
|
||||
if immNeg {
|
||||
imm = -imm // SUB $imm arrived through the ADDI alias
|
||||
}
|
||||
rs1 := regFromOperand(ops[1]) // source register
|
||||
rd := regFromOperand(ops[2]) // destination
|
||||
if rd < 0 || rs1 < 0 {
|
||||
@@ -490,6 +574,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
|
||||
case len(ops) == 2 && isITypeInstr(mnem):
|
||||
imm := immFromOperand(ops[0])
|
||||
if immNeg {
|
||||
imm = -imm
|
||||
}
|
||||
rd := regFromOperand(ops[1])
|
||||
if rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register in %s", mnem)
|
||||
@@ -626,7 +713,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
|
||||
if rd < 0 || rs1 < 0 {
|
||||
return nil, fmt.Errorf("MOV load: invalid operand")
|
||||
}
|
||||
return riscvFrameMemOp(riscvEnc{0x03, 0x3, 0x00}, false, rd, rs1, off), nil
|
||||
return riscvFrameMemOp(riscvMovEnc(strings.ToUpper(instr.Mnemonic.Text), false), false, rd, rs1, off), nil
|
||||
}
|
||||
|
||||
// Register → memory (store).
|
||||
@@ -643,21 +730,70 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
|
||||
if rs2 < 0 || rs1 < 0 {
|
||||
return nil, fmt.Errorf("MOV store: invalid operand")
|
||||
}
|
||||
return riscvFrameMemOp(riscvEnc{0x23, 0x3, 0x00}, true, rs2, rs1, off), nil
|
||||
return riscvFrameMemOp(riscvMovEnc(strings.ToUpper(instr.Mnemonic.Text), true), true, rs2, rs1, off), nil
|
||||
}
|
||||
|
||||
// Register → register (ADDI $0, src, dst).
|
||||
// Register → register: MOVD/MOVF are FP moves (fsgnj with rs2 = rs1),
|
||||
// everything else is ADDI $0, src, dst.
|
||||
{
|
||||
rs1 := regFromOperand(src)
|
||||
rd := regFromOperand(dst)
|
||||
if rd < 0 || rs1 < 0 {
|
||||
return nil, fmt.Errorf("MOV: invalid register operand")
|
||||
}
|
||||
mnem := strings.ToUpper(instr.Mnemonic.Text)
|
||||
if mnem == "MOVD" || mnem == "MOVF" {
|
||||
op := uint32(0x20000053) // FSGNJ.S
|
||||
if mnem == "MOVD" {
|
||||
op = 0x22000053 // FSGNJ.D
|
||||
}
|
||||
return wordLE(op | uint32(rs1)<<15 | uint32(rs1)<<20 | uint32(rd)<<7), nil
|
||||
}
|
||||
word := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs1, 0)
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
}
|
||||
}
|
||||
|
||||
// riscvMovEnc returns the load (store=false) or store (store=true) opcode for
|
||||
// a MOV-family mnemonic: the suffix selects the access width, MOVD and MOVF
|
||||
// select the FP load/store opcodes, and bare MOV is the 64-bit integer form.
|
||||
func riscvMovEnc(mnem string, store bool) riscvEnc {
|
||||
if store {
|
||||
switch mnem {
|
||||
case "MOVB":
|
||||
return riscvEnc{0x23, 0x0, 0x00} // SB
|
||||
case "MOVH":
|
||||
return riscvEnc{0x23, 0x1, 0x00} // SH
|
||||
case "MOVW":
|
||||
return riscvEnc{0x23, 0x2, 0x00} // SW
|
||||
case "MOVF":
|
||||
return riscvEnc{0x27, 0x2, 0x00} // FSW
|
||||
case "MOVD":
|
||||
return riscvEnc{0x27, 0x3, 0x00} // FSD
|
||||
}
|
||||
return riscvEnc{0x23, 0x3, 0x00} // SD
|
||||
}
|
||||
switch mnem {
|
||||
case "MOVB":
|
||||
return riscvEnc{0x03, 0x0, 0x00} // LB
|
||||
case "MOVBU":
|
||||
return riscvEnc{0x03, 0x4, 0x00} // LBU
|
||||
case "MOVH":
|
||||
return riscvEnc{0x03, 0x1, 0x00} // LH
|
||||
case "MOVHU":
|
||||
return riscvEnc{0x03, 0x5, 0x00} // LHU
|
||||
case "MOVW":
|
||||
return riscvEnc{0x03, 0x2, 0x00} // LW
|
||||
case "MOVWU":
|
||||
return riscvEnc{0x03, 0x6, 0x00} // LWU
|
||||
case "MOVF":
|
||||
return riscvEnc{0x07, 0x2, 0x00} // FLW
|
||||
case "MOVD":
|
||||
return riscvEnc{0x07, 0x3, 0x00} // FLD
|
||||
}
|
||||
return riscvEnc{0x03, 0x3, 0x00} // LD
|
||||
}
|
||||
|
||||
// riscvFrameMemOp encodes a register-relative load (store=false, I-type
|
||||
// width 0x03) or store (store=true, S-type width 0x23) of the 64-bit width
|
||||
// at off(rs1). Offsets beyond the signed 12-bit range materialise the
|
||||
|
||||
@@ -328,6 +328,8 @@ var riscvCvtTable = map[string]riscvCvtEnc{
|
||||
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
|
||||
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
|
||||
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
|
||||
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
|
||||
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
|
||||
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
|
||||
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
|
||||
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
|
||||
|
||||
+4
-2
@@ -511,8 +511,10 @@ requires -p, the package path, and the installed Go toolchain).
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
|
||||
return 1
|
||||
}
|
||||
if len(img.Funcs) == 0 {
|
||||
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found")
|
||||
if len(img.Funcs) == 0 && len(img.Data) == 0 {
|
||||
// A file with neither code nor data assembles to nothing, which is
|
||||
// almost always a wrong architecture rather than an intent.
|
||||
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions or GLOBL data found")
|
||||
return 1
|
||||
}
|
||||
for _, fn := range img.Funcs {
|
||||
|
||||
+45
-14
@@ -438,25 +438,56 @@ func parseAddress(g []token.Token) ast.Address {
|
||||
// Optional leading displacement before a '(' base group. A sign pushes
|
||||
// the parenthesis one token further out: -4(DX) has it at i+2.
|
||||
if isSignedNumber(g, i) {
|
||||
paren := i + 1
|
||||
if g[i].Kind == token.Minus || g[i].Kind == token.Plus {
|
||||
paren = i + 2
|
||||
}
|
||||
if paren < len(g) && g[paren].Kind == token.LParen {
|
||||
j := i
|
||||
neg := false
|
||||
if g[i].Kind == token.Minus {
|
||||
if g[j].Kind == token.Minus {
|
||||
neg = true
|
||||
i++
|
||||
} else if g[i].Kind == token.Plus {
|
||||
i++
|
||||
j++
|
||||
} else if g[j].Kind == token.Plus {
|
||||
j++
|
||||
}
|
||||
if j < len(g) && g[j].Kind == token.Number {
|
||||
v := parseInt(g[j].Text)
|
||||
j++
|
||||
// A term may carry a *number factor: 0*8(base).
|
||||
for j+1 < len(g) && g[j].Kind == token.Star && g[j+1].Kind == token.Number {
|
||||
v *= parseInt(g[j+1].Text)
|
||||
j += 2
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
addr.Offset = parseInt(g[i].Text)
|
||||
addr.HasOff = true
|
||||
if neg {
|
||||
addr.Offset = -addr.Offset
|
||||
v = -v
|
||||
}
|
||||
i++
|
||||
// Further +/- terms, each with its optional factor:
|
||||
// 3*8+8(base), 8-4*2(base).
|
||||
for {
|
||||
termNeg := false
|
||||
if j < len(g) && g[j].Kind == token.Minus {
|
||||
termNeg = true
|
||||
} else if j < len(g) && g[j].Kind == token.Plus {
|
||||
} else {
|
||||
break
|
||||
}
|
||||
if j+1 < len(g) && g[j+1].Kind == token.Number {
|
||||
tv := parseInt(g[j+1].Text)
|
||||
j += 2
|
||||
for j+1 < len(g) && g[j].Kind == token.Star && g[j+1].Kind == token.Number {
|
||||
tv *= parseInt(g[j+1].Text)
|
||||
j += 2
|
||||
}
|
||||
if termNeg {
|
||||
tv = -tv
|
||||
}
|
||||
v += tv
|
||||
continue
|
||||
}
|
||||
break
|
||||
}
|
||||
// Commit only when the expression is followed by the base
|
||||
// group; a bare number stays untouched for the caller.
|
||||
if j < len(g) && g[j].Kind == token.LParen {
|
||||
addr.Offset = v
|
||||
addr.HasOff = true
|
||||
i = j
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Vendored
+36
@@ -0,0 +1,36 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// GOROOT-derived shapes: the MOV width suffixes for narrow loads and the
|
||||
// branch-zero pseudos. Deliberately absent: the immediate ALU aliases
|
||||
// (AND/SUB $imm) and the FP memory forms, whose RVC compression the encoder
|
||||
// does not reproduce yet, so a parity kernel could not hold them.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func mix(x int64, y int64) int64
|
||||
TEXT ·mix(SB), NOSPLIT, $0-24
|
||||
MOV x+0(FP), X5
|
||||
MOVWU 0(X5), X11
|
||||
MOVB 1(X5), X12
|
||||
MOVBU 2(X5), X13
|
||||
ADD X11, X12, X14
|
||||
ADD X13, X14, X15
|
||||
MOV X15, ret+16(FP)
|
||||
RET
|
||||
|
||||
// func branchy(n int64) int64
|
||||
TEXT ·branchy(SB), NOSPLIT, $0-16
|
||||
MOV n+0(FP), X5
|
||||
BEQZ X5, zero
|
||||
BNEZ X5, one
|
||||
BLTZ X5, zero
|
||||
BGEZ X5, one
|
||||
zero:
|
||||
MOV $0, X6
|
||||
MOV X6, ret+8(FP)
|
||||
RET
|
||||
one:
|
||||
MOV $1, X6
|
||||
MOV X6, ret+8(FP)
|
||||
RET
|
||||
@@ -28,6 +28,7 @@ func TestGroundTruthRISCV(t *testing.T) {
|
||||
"../testdata/verify/bigframe_riscv64.s",
|
||||
"../testdata/verify/guard_riscv64.s",
|
||||
"../testdata/verify/indirect_riscv64.s",
|
||||
"../testdata/verify/misc_riscv64.s",
|
||||
"trampoline_riscv64.s",
|
||||
} {
|
||||
t.Run(path, func(t *testing.T) {
|
||||
|
||||
Reference in New Issue
Block a user