diff --git a/asm/link.go b/asm/link.go index 3a0565c..70a3494 100644 --- a/asm/link.go +++ b/asm/link.go @@ -440,83 +440,90 @@ type dataSym struct { } // collectData gathers the file's static symbols (GLOBL) and their initial -// contents (DATA) into byte buffers, in declaration order. +// contents (DATA) into byte buffers. Two passes: the Plan 9 convention puts +// every DATA line before its symbol's GLOBL, so the symbols are registered +// before the initialisers are applied. func collectData(f *ast.File) ([]dataSym, error) { index := map[string]int{} var syms []dataSym for _, d := range f.Decls { - switch dd := d.(type) { - case *ast.Globl: - if dd.Name == nil || dd.Name.Pseudo != "SB" { - continue - } - name := dd.Name.Name - if _, dup := index[name]; dup { - return nil, fmt.Errorf("duplicate GLOBL %q", name) - } - size := 0 - if dd.Size != nil && dd.Size.Imm.HasVal { - size = int(dd.Size.Imm.Val) - } - index[name] = len(syms) - ds := dataSym{ - name: name, - pkg: dd.Name.Pkg, - buf: make([]byte, size), - size: size, - static: dd.Name.Static, - } - for _, f := range dd.Flags { - switch f { - case "RODATA": - ds.rodata = true - case "DUPOK": - ds.dupok = true - default: - // Legacy numeric flag constants (runtime/textflag.h): - // DUPOK is 2, RODATA is 8; combinations arrive as one - // number (e.g. 10 = RODATA|DUPOK). - if n, err := strconv.Atoi(f); err == nil { - if n&2 != 0 { - ds.dupok = true - } - if n&8 != 0 { - ds.rodata = true - } + gd, ok := d.(*ast.Globl) + if !ok { + continue + } + if gd.Name == nil || gd.Name.Pseudo != "SB" { + continue + } + name := gd.Name.Name + if _, dup := index[name]; dup { + return nil, fmt.Errorf("duplicate GLOBL %q", name) + } + size := 0 + if gd.Size != nil && gd.Size.Imm.HasVal { + size = int(gd.Size.Imm.Val) + } + index[name] = len(syms) + ds := dataSym{ + name: name, + pkg: gd.Name.Pkg, + buf: make([]byte, size), + size: size, + static: gd.Name.Static, + } + for _, f := range gd.Flags { + switch f { + case "RODATA": + ds.rodata = true + case "DUPOK": + ds.dupok = true + default: + // Legacy numeric flag constants (runtime/textflag.h): + // DUPOK is 2, RODATA is 8; combinations arrive as one + // number (e.g. 10 = RODATA|DUPOK). + if n, err := strconv.Atoi(f); err == nil { + if n&2 != 0 { + ds.dupok = true + } + if n&8 != 0 { + ds.rodata = true } } } - syms = append(syms, ds) - - case *ast.Data: - if dd.Name == nil || dd.Name.Pseudo != "SB" { - continue - } - i, ok := index[dd.Name.Name] - if !ok { - return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name) - } - if dd.Value == nil || !dd.Value.Imm.HasVal { - return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name) - } - w := dd.Width - switch w { - case 1, 2, 4, 8: - default: - return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w) - } - off := dd.Name.Offset - buf := syms[i].buf - if off < 0 || off+int64(w) > int64(len(buf)) { - return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf)) - } - v := dd.Value.Imm.Val - if dd.Value.Imm.Neg { - v = -v - } - for j := range w { - buf[off+int64(j)] = byte(v >> (8 * j)) - } + } + syms = append(syms, ds) + } + for _, d := range f.Decls { + dd, ok := d.(*ast.Data) + if !ok { + continue + } + if dd.Name == nil || dd.Name.Pseudo != "SB" { + continue + } + i, ok := index[dd.Name.Name] + if !ok { + return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name) + } + if dd.Value == nil || !dd.Value.Imm.HasVal { + return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name) + } + w := dd.Width + switch w { + case 1, 2, 4, 8: + default: + return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w) + } + off := dd.Name.Offset + buf := syms[i].buf + if off < 0 || off+int64(w) > int64(len(buf)) { + return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf)) + } + v := dd.Value.Imm.Val + if dd.Value.Imm.Neg { + v = -v + } + for j := range w { + buf[off+int64(j)] = byte(v >> (8 * j)) } } return syms, nil diff --git a/asm/riscv_assemble.go b/asm/riscv_assemble.go index c33885f..9197658 100644 --- a/asm/riscv_assemble.go +++ b/asm/riscv_assemble.go @@ -135,6 +135,43 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ return out, offsets, relocs, lines, spadj, nil } +// riscvImmAlias maps the R-type ALU mnemonics onto their I-type immediate +// forms: the toolchain accepts ADD $imm, rj, rd and emits addi. Applied +// whenever the first operand is an immediate. +var riscvImmAlias = map[string]string{ + "ADD": "ADDI", + "ADDW": "ADDIW", + "AND": "ANDI", + "OR": "ORI", + "XOR": "XORI", + "SLL": "SLLI", + "SRL": "SRLI", + "SRA": "SRAI", + "SLLW": "SLLIW", + "SRLW": "SRLIW", + "SRAW": "SRAIW", +} + +// riscvNormaliseImmAlias rewrites the mnemonic to its immediate form when the +// first operand is an immediate: the toolchain accepts ADD $imm, rj, rd and +// emits addi, and SUB $imm becomes addi with the negated immediate. The +// second result reports that negation; the operand itself is left untouched +// because several passes normalise the same instruction. +func riscvNormaliseImmAlias(mnem string, ops []*ast.Operand) (string, bool) { + if len(ops) >= 2 && isImmOperand(ops[0]) { + switch strings.ToUpper(mnem) { + case "SUB": + return "ADDI", true + case "SUBW": + return "ADDIW", true + } + if alias, ok := riscvImmAlias[strings.ToUpper(mnem)]; ok { + return alias, false + } + } + return mnem, false +} + // riscvInstrSize returns the encoded size in bytes of a RISC-V instruction. // Most instructions are 4 bytes; MOV with a large immediate and I-type // arithmetic with a large immediate expand to several (possibly compressed) @@ -142,10 +179,12 @@ func assembleRISCV(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int { mnem := instr.Mnemonic.Text ops := instr.Operands + var immNeg bool + mnem, immNeg = riscvNormaliseImmAlias(mnem, ops) if mnem == "RET" { return len(riscvReturn(fi)) } - if mnem == "MOV" && len(ops) == 2 { + if strings.HasPrefix(mnem, "MOV") && len(ops) == 2 { // MOV $sym(SB), rd → 8 bytes (AUIPC + ADDI). if isImmOperand(ops[0]) && ops[0].Imm.Sym != nil && ops[0].Imm.Sym.Pseudo == "SB" { return 8 @@ -173,7 +212,11 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int { } // I-type arithmetic with a large immediate expands to several instructions. if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) { - return riscvItypeImmediateSize(mnem, immFromOperand(ops[0])) + imm := immFromOperand(ops[0]) + if immNeg { + imm = -imm + } + return riscvItypeImmediateSize(mnem, imm) } return 4 } @@ -192,6 +235,8 @@ func isBranchLike(mnem string) bool { func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscvFrameInfo, relocs *[]Reloc) ([]byte, error) { mnem := instr.Mnemonic.Text ops := instr.Operands + var immNeg bool + mnem, immNeg = riscvNormaliseImmAlias(mnem, ops) var word uint32 // Handle pseudo-instructions and special cases first. @@ -279,14 +324,50 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil // MOV is a pseudo-instruction that the Go assembler uses for loads, - // stores, register moves and immediate loads. - case "MOV": + // stores, register moves and immediate loads. The width suffixes + // (MOVB/MOVH/MOVW and unsigned forms) select the access width, and + // MOVD/MOVF address the FP registers. + case "MOV", "MOVB", "MOVBU", "MOVH", "MOVHU", "MOVW", "MOVWU", "MOVF", "MOVD": return encodeRISCVMov(instr, fi, relocs) // JALR: indirect jump/call. Plan 9: JALR rs1, rd or JALR offset(rs1). case "JALR": return encodeRISCVJALR(instr, fi) + // Branch-zero pseudos: BEQZ/BNEZ compare against X0, and BLTZ/BGEZ/ + // BLEZ/BGTZ reorder the register operands of BLT/BGE accordingly. + case "BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ": + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rs := regFromOperand(ops[0]) + if rs < 0 { + return nil, fmt.Errorf("%s: invalid register", mnem) + } + target := labelFromOperand(ops[1]) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) + } + var enc riscvEnc + rs1, rs2 := rs, 0 + switch mnem { + case "BEQZ": + enc = riscvEnc{0x63, 0x0, 0x00} // beq rs, x0 + case "BNEZ": + enc = riscvEnc{0x63, 0x1, 0x00} // bne rs, x0 + case "BLTZ": + enc = riscvEnc{0x63, 0x4, 0x00} // blt rs, x0 + case "BGEZ": + enc = riscvEnc{0x63, 0x5, 0x00} // bge rs, x0 + case "BLEZ": + enc, rs1, rs2 = riscvEnc{0x63, 0x5, 0x00}, 0, rs // bge x0, rs + case "BGTZ": + enc, rs1, rs2 = riscvEnc{0x63, 0x4, 0x00}, 0, rs // blt x0, rs + } + word = riscvBType(enc, rs1, rs2, int32(targetOff-pc)) + return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil + // System instructions with no operands. case "FENCE", "ECALL", "EBREAK": enc, ok := riscvInstrTable[mnem] @@ -481,6 +562,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv // two-operand form INSTR $imm, rd uses rd as the source. case len(ops) == 3 && isITypeInstr(mnem): imm := immFromOperand(ops[0]) // immediate + if immNeg { + imm = -imm // SUB $imm arrived through the ADDI alias + } rs1 := regFromOperand(ops[1]) // source register rd := regFromOperand(ops[2]) // destination if rd < 0 || rs1 < 0 { @@ -490,6 +574,9 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv case len(ops) == 2 && isITypeInstr(mnem): imm := immFromOperand(ops[0]) + if immNeg { + imm = -imm + } rd := regFromOperand(ops[1]) if rd < 0 { return nil, fmt.Errorf("invalid register in %s", mnem) @@ -626,7 +713,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("MOV load: invalid operand") } - return riscvFrameMemOp(riscvEnc{0x03, 0x3, 0x00}, false, rd, rs1, off), nil + return riscvFrameMemOp(riscvMovEnc(strings.ToUpper(instr.Mnemonic.Text), false), false, rd, rs1, off), nil } // Register → memory (store). @@ -643,21 +730,70 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt if rs2 < 0 || rs1 < 0 { return nil, fmt.Errorf("MOV store: invalid operand") } - return riscvFrameMemOp(riscvEnc{0x23, 0x3, 0x00}, true, rs2, rs1, off), nil + return riscvFrameMemOp(riscvMovEnc(strings.ToUpper(instr.Mnemonic.Text), true), true, rs2, rs1, off), nil } - // Register → register (ADDI $0, src, dst). + // Register → register: MOVD/MOVF are FP moves (fsgnj with rs2 = rs1), + // everything else is ADDI $0, src, dst. { rs1 := regFromOperand(src) rd := regFromOperand(dst) if rd < 0 || rs1 < 0 { return nil, fmt.Errorf("MOV: invalid register operand") } + mnem := strings.ToUpper(instr.Mnemonic.Text) + if mnem == "MOVD" || mnem == "MOVF" { + op := uint32(0x20000053) // FSGNJ.S + if mnem == "MOVD" { + op = 0x22000053 // FSGNJ.D + } + return wordLE(op | uint32(rs1)<<15 | uint32(rs1)<<20 | uint32(rd)<<7), nil + } word := riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs1, 0) return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil } } +// riscvMovEnc returns the load (store=false) or store (store=true) opcode for +// a MOV-family mnemonic: the suffix selects the access width, MOVD and MOVF +// select the FP load/store opcodes, and bare MOV is the 64-bit integer form. +func riscvMovEnc(mnem string, store bool) riscvEnc { + if store { + switch mnem { + case "MOVB": + return riscvEnc{0x23, 0x0, 0x00} // SB + case "MOVH": + return riscvEnc{0x23, 0x1, 0x00} // SH + case "MOVW": + return riscvEnc{0x23, 0x2, 0x00} // SW + case "MOVF": + return riscvEnc{0x27, 0x2, 0x00} // FSW + case "MOVD": + return riscvEnc{0x27, 0x3, 0x00} // FSD + } + return riscvEnc{0x23, 0x3, 0x00} // SD + } + switch mnem { + case "MOVB": + return riscvEnc{0x03, 0x0, 0x00} // LB + case "MOVBU": + return riscvEnc{0x03, 0x4, 0x00} // LBU + case "MOVH": + return riscvEnc{0x03, 0x1, 0x00} // LH + case "MOVHU": + return riscvEnc{0x03, 0x5, 0x00} // LHU + case "MOVW": + return riscvEnc{0x03, 0x2, 0x00} // LW + case "MOVWU": + return riscvEnc{0x03, 0x6, 0x00} // LWU + case "MOVF": + return riscvEnc{0x07, 0x2, 0x00} // FLW + case "MOVD": + return riscvEnc{0x07, 0x3, 0x00} // FLD + } + return riscvEnc{0x03, 0x3, 0x00} // LD +} + // riscvFrameMemOp encodes a register-relative load (store=false, I-type // width 0x03) or store (store=true, S-type width 0x23) of the 64-bit width // at off(rs1). Offsets beyond the signed 12-bit range materialise the diff --git a/asm/riscv_encode.go b/asm/riscv_encode.go index d794b8a..4ed7c23 100644 --- a/asm/riscv_encode.go +++ b/asm/riscv_encode.go @@ -328,6 +328,8 @@ var riscvCvtTable = map[string]riscvCvtEnc{ "FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32 "FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32 "FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32 + "FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask + "FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask "FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64 "FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64 "FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64 diff --git a/cmd/gasm/main.go b/cmd/gasm/main.go index 4aba239..5ff4f40 100644 --- a/cmd/gasm/main.go +++ b/cmd/gasm/main.go @@ -511,8 +511,10 @@ requires -p, the package path, and the installed Go toolchain). fmt.Fprintf(os.Stderr, "%s: %v\n", path, err) return 1 } - if len(img.Funcs) == 0 { - fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found") + if len(img.Funcs) == 0 && len(img.Data) == 0 { + // A file with neither code nor data assembles to nothing, which is + // almost always a wrong architecture rather than an intent. + fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions or GLOBL data found") return 1 } for _, fn := range img.Funcs { diff --git a/parser/parser.go b/parser/parser.go index b615f5a..d5751f2 100644 --- a/parser/parser.go +++ b/parser/parser.go @@ -438,25 +438,56 @@ func parseAddress(g []token.Token) ast.Address { // Optional leading displacement before a '(' base group. A sign pushes // the parenthesis one token further out: -4(DX) has it at i+2. if isSignedNumber(g, i) { - paren := i + 1 - if g[i].Kind == token.Minus || g[i].Kind == token.Plus { - paren = i + 2 + j := i + neg := false + if g[j].Kind == token.Minus { + neg = true + j++ + } else if g[j].Kind == token.Plus { + j++ } - if paren < len(g) && g[paren].Kind == token.LParen { - neg := false - if g[i].Kind == token.Minus { - neg = true - i++ - } else if g[i].Kind == token.Plus { - i++ + if j < len(g) && g[j].Kind == token.Number { + v := parseInt(g[j].Text) + j++ + // A term may carry a *number factor: 0*8(base). + for j+1 < len(g) && g[j].Kind == token.Star && g[j+1].Kind == token.Number { + v *= parseInt(g[j+1].Text) + j += 2 } - if i < len(g) && g[i].Kind == token.Number { - addr.Offset = parseInt(g[i].Text) - addr.HasOff = true - if neg { - addr.Offset = -addr.Offset + if neg { + v = -v + } + // Further +/- terms, each with its optional factor: + // 3*8+8(base), 8-4*2(base). + for { + termNeg := false + if j < len(g) && g[j].Kind == token.Minus { + termNeg = true + } else if j < len(g) && g[j].Kind == token.Plus { + } else { + break } - i++ + if j+1 < len(g) && g[j+1].Kind == token.Number { + tv := parseInt(g[j+1].Text) + j += 2 + for j+1 < len(g) && g[j].Kind == token.Star && g[j+1].Kind == token.Number { + tv *= parseInt(g[j+1].Text) + j += 2 + } + if termNeg { + tv = -tv + } + v += tv + continue + } + break + } + // Commit only when the expression is followed by the base + // group; a bare number stays untouched for the caller. + if j < len(g) && g[j].Kind == token.LParen { + addr.Offset = v + addr.HasOff = true + i = j } } } diff --git a/testdata/verify/misc_riscv64.s b/testdata/verify/misc_riscv64.s new file mode 100644 index 0000000..0c8980f --- /dev/null +++ b/testdata/verify/misc_riscv64.s @@ -0,0 +1,36 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// GOROOT-derived shapes: the MOV width suffixes for narrow loads and the +// branch-zero pseudos. Deliberately absent: the immediate ALU aliases +// (AND/SUB $imm) and the FP memory forms, whose RVC compression the encoder +// does not reproduce yet, so a parity kernel could not hold them. + +#include "textflag.h" + +// func mix(x int64, y int64) int64 +TEXT ·mix(SB), NOSPLIT, $0-24 + MOV x+0(FP), X5 + MOVWU 0(X5), X11 + MOVB 1(X5), X12 + MOVBU 2(X5), X13 + ADD X11, X12, X14 + ADD X13, X14, X15 + MOV X15, ret+16(FP) + RET + +// func branchy(n int64) int64 +TEXT ·branchy(SB), NOSPLIT, $0-16 + MOV n+0(FP), X5 + BEQZ X5, zero + BNEZ X5, one + BLTZ X5, zero + BGEZ X5, one +zero: + MOV $0, X6 + MOV X6, ret+8(FP) + RET +one: + MOV $1, X6 + MOV X6, ret+8(FP) + RET diff --git a/verify/riscv_groundtruth_test.go b/verify/riscv_groundtruth_test.go index 146a945..6c6d85b 100644 --- a/verify/riscv_groundtruth_test.go +++ b/verify/riscv_groundtruth_test.go @@ -28,6 +28,7 @@ func TestGroundTruthRISCV(t *testing.T) { "../testdata/verify/bigframe_riscv64.s", "../testdata/verify/guard_riscv64.s", "../testdata/verify/indirect_riscv64.s", + "../testdata/verify/misc_riscv64.s", "trampoline_riscv64.s", } { t.Run(path, func(t *testing.T) {