feat: gasm-devkit 0.1.0 — GAsm lexer, parser, linter, formatter, LSP and amd64 assembler

Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
2026-07-06 09:49:50 +02:00
commit d5a4a6de45
53 changed files with 13166 additions and 0 deletions
+12
View File
@@ -0,0 +1,12 @@
# Binaries
/gasm
/bin/
*.exe
# Test and coverage artefacts
coverage.out
*.test
# Editor detritus
*.swp
.DS_Store
+28
View File
@@ -0,0 +1,28 @@
BSD 3-Clause License
Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the following conditions are met:
1. Redistributions of source code must retain the above copyright notice,
this list of conditions and the following disclaimer.
2. Redistributions in binary form must reproduce the above copyright notice,
this list of conditions and the following disclaimer in the documentation
and/or other materials provided with the distribution.
3. Neither the name of the copyright holder nor the names of its
contributors may be used to endorse or promote products derived from
this software without specific prior written permission.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
+180
View File
@@ -0,0 +1,180 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Command gen regenerates the architecture instruction tables from the Go
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
// complete set of mnemonics it accepts for each architecture in
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
// names so gasm-devkit supports every instruction the real assembler does,
// with no hand-maintained (and therefore inevitably incomplete) lists.
//
// Usage (via the justfile):
//
// just gen
//
// The generated files are committed; regenerating requires a Go installation
// but the toolkit itself has no dependency on the toolchain source at runtime.
package main
import (
"fmt"
"go/ast"
"go/parser"
"go/token"
"os"
"os/exec"
"path/filepath"
"sort"
"strings"
)
// archDirs maps a gasm-devkit architecture name to its obj sub-directory.
var archDirs = []struct {
arch string
sub string
}{
{"amd64", "x86"},
{"arm64", "arm64"},
{"riscv", "riscv"},
{"loong64", "loong64"},
}
func main() {
goroot := strings.TrimSpace(runGoEnvGOROOT())
if goroot == "" {
fatal("could not determine GOROOT")
}
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
common, err := extractInstrs(commonPath)
if err != nil {
fatal("extract common: %v", err)
}
common = filterCommon(common)
if err := writeCommon(common); err != nil {
fatal("write common: %v", err)
}
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
for _, a := range archDirs {
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
names, err := extractInstrs(path)
if err != nil {
fatal("extract %s: %v", a.arch, err)
}
if err := writeGen(a.arch, a.sub, names); err != nil {
fatal("write %s: %v", a.arch, err)
}
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names), a.arch)
}
}
// filterCommon drops opcode names that are not user-writable instructions.
func filterCommon(names []string) []string {
drop := map[string]bool{"XXX": true, "LAST": true}
var out []string
for _, n := range names {
if !drop[n] {
out = append(out, n)
}
}
return out
}
// writeCommon emits arch/common_gen.go.
func writeCommon(names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n\n")
b.WriteString("package arch\n\n")
b.WriteString("// commonGeneratedInstrs is the set of opcodes shared by every architecture\n")
b.WriteString("// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).\n")
b.WriteString("var commonGeneratedInstrs = []string{\n")
for _, n := range names {
fmt.Fprintf(&b, "\t%q,\n", n)
}
b.WriteString("}\n")
return os.WriteFile(filepath.Join("arch", "common_gen.go"), []byte(b.String()), 0o644)
}
// extractInstrs parses an anames.go file and returns the sorted, de-duplicated
// instruction names from its `var Anames = []string{...}` literal.
func extractInstrs(path string) ([]string, error) {
fset := token.NewFileSet()
f, err := parser.ParseFile(fset, path, nil, 0)
if err != nil {
return nil, err
}
seen := map[string]bool{}
var names []string
for _, decl := range f.Decls {
gd, ok := decl.(*ast.GenDecl)
if !ok || gd.Tok != token.VAR {
continue
}
for _, spec := range gd.Specs {
vs, ok := spec.(*ast.ValueSpec)
if !ok || len(vs.Names) == 0 || vs.Names[0].Name != "Anames" {
continue
}
for _, val := range vs.Values {
cl, ok := val.(*ast.CompositeLit)
if !ok {
continue
}
for _, elt := range cl.Elts {
if lit := stringLit(elt); lit != "" && lit != "LAST" && !seen[lit] {
seen[lit] = true
names = append(names, lit)
}
}
}
}
}
sort.Strings(names)
return names, nil
}
// stringLit returns the string value of a composite-literal element, whether it
// is a plain literal or a keyed entry such as `obj.A_ARCHSPECIFIC: "AAA"`.
func stringLit(elt ast.Expr) string {
switch e := elt.(type) {
case *ast.BasicLit:
if e.Kind == token.STRING {
return strings.Trim(e.Value, `"`)
}
case *ast.KeyValueExpr:
return stringLit(e.Value)
}
return ""
}
// writeGen emits arch/<arch>_gen.go.
func writeGen(arch, sub string, names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n\n")
b.WriteString("package arch\n\n")
b.WriteString("// " + arch + "GeneratedInstrs is the complete set of " + arch +
" mnemonics accepted by\n// Go's Plan 9 assembler.\n")
b.WriteString("var " + arch + "GeneratedInstrs = []string{\n")
for _, n := range names {
fmt.Fprintf(&b, "\t%q,\n", n)
}
b.WriteString("}\n")
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
}
func runGoEnvGOROOT() string {
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
return ""
}
return string(out)
}
func fatal(format string, args ...any) {
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
os.Exit(1)
}
+317
View File
@@ -0,0 +1,317 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
import "fmt"
func buildAMD64() *Table {
return newTable(AMD64, amd64Registers(), mergedInstrs(amd64Summaries(), commonGeneratedInstrs, amd64GeneratedInstrs, amd64Aliases()))
}
// amd64Aliases are the traditional x86 conditional-jump spellings (plus a few
// instruction aliases) that Go's assembler accepts and maps onto its canonical
// opcodes. They are user-writable but absent from the generated opcode table,
// so they are listed explicitly here.
func amd64Aliases() []string {
return []string{
"JA", "JAE", "JB", "JBE", "JC", "JCC", "JCS", "JE", "JG", "JHI", "JHS",
"JL", "JLO", "JLS", "JMI", "JNA", "JNAE", "JNB", "JNBE", "JNC", "JNG",
"JNGE", "JNL", "JNLE", "JNO", "JNP", "JNS", "JNZ", "JO", "JOC", "JOS",
"JP", "JPC", "JPE", "JPL", "JPO", "JPS", "JS", "JZ",
"MASKMOVDQU", "MOVDQ2Q", "MOVNTDQ", "MOVOA", "PSLLDQ", "PSRLDQ",
"MOVD", "PADDD", "MOVBELL", "MOVBEQQ", "MOVBEWW",
}
}
// amd64Summaries returns the curated documentation/operand-count table keyed by
// upper-case mnemonic. It enriches the complete generated name list; names
// without a curated entry are still recognised, just without a summary.
func amd64Summaries() map[string]Instr { return toMap(amd64Curated()) }
// amd64Registers builds the amd64 register file. Numbered registers are
// generated; the irregularly named ones are listed explicitly.
func amd64Registers() []Register {
var regs []Register
add := func(name string, class RegClass, desc string) {
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
}
// 64-bit general-purpose registers.
for _, n := range []string{"AX", "BX", "CX", "DX", "SI", "DI", "BP", "SP"} {
add(n, GPR, "64-bit general-purpose register")
}
for i := 8; i <= 15; i++ {
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
}
// 8-bit low/high sub-registers.
for _, n := range []string{"AL", "BL", "CL", "DL", "SIL", "DIL", "BPL", "SPL"} {
add(n, GPRSub, "8-bit low sub-register")
}
for _, n := range []string{"AH", "BH", "CH", "DH"} {
add(n, GPRSub, "8-bit high sub-register")
}
// Sized numbered sub-registers.
for i := 8; i <= 15; i++ {
add(fmt.Sprintf("R%dB", i), GPRSub, "8-bit sub-register")
add(fmt.Sprintf("R%dW", i), GPRSub, "16-bit sub-register")
add(fmt.Sprintf("R%dD", i), GPRSub, "32-bit sub-register")
}
// SIMD vector registers: X (SSE), Y (AVX2), Z (AVX-512).
for i := 0; i <= 15; i++ {
add(fmt.Sprintf("X%d", i), Vector, "128-bit SSE/AVX vector register")
add(fmt.Sprintf("Y%d", i), Vector, "256-bit AVX2 vector register")
add(fmt.Sprintf("Z%d", i), Vector, "512-bit AVX-512 vector register")
}
for i := 16; i <= 31; i++ {
add(fmt.Sprintf("Z%d", i), Vector, "512-bit AVX-512 vector register")
}
// AVX-512 mask registers.
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
}
return regs
}
// i builds an instruction with an unknown/variable operand count.
func i(name, summary string) Instr {
return Instr{Name: name, Summary: summary, MinOps: -1, MaxOps: -1}
}
// ic builds an instruction with an explicit operand-count range.
func ic(name, summary string, min, max int) Instr {
return Instr{Name: name, Summary: summary, MinOps: min, MaxOps: max}
}
// amd64Curated returns the hand-written subset of amd64 instructions that carry
// a summary and/or an explicit operand-count range. The authoritative,
// complete instruction set is amd64GeneratedInstrs (see amd64_gen.go).
func amd64Curated() []Instr {
var t []Instr
// Data movement.
for _, s := range []string{"B", "W", "L", "Q"} {
t = append(t, ic("MOV"+s, "Move "+s+"-width value", 2, 2))
}
for _, m := range []string{
"MOVBLZX", "MOVBQSX", "MOVWLZX", "MOVWQSX", "MOVWLSX", "MOVLQSX", "MOVBLSX", "MOVQL",
} {
t = append(t, ic(m, "Sign/zero-extending move", 2, 2))
}
for _, m := range []string{"MOVO", "MOVOU"} {
t = append(t, ic(m, "Move 16-byte aligned/unaligned vector", 2, 2))
}
for _, s := range []string{"B", "W", "L", "Q"} {
t = append(t, ic("LEA"+s, "Load effective address", 2, 2))
}
for _, s := range []string{"B", "W", "L", "Q"} {
t = append(t, ic("XCHG"+s, "Exchange operands", 2, 2))
}
// Integer arithmetic and logic.
for _, op := range []string{"ADD", "SUB", "AND", "OR", "XOR", "ADC", "SBB"} {
for _, s := range []string{"B", "W", "L", "Q"} {
t = append(t, ic(op+s, op+" integer", 2, 2))
}
}
for _, s := range []string{"B", "W", "L", "Q"} {
t = append(t, ic("INC"+s, "Increment", 1, 1))
t = append(t, ic("DEC"+s, "Decrement", 1, 1))
t = append(t, ic("NEG"+s, "Two's-complement negate", 1, 1))
t = append(t, ic("NOT"+s, "Bitwise complement", 1, 1))
}
for _, s := range []string{"B", "W", "L", "Q"} {
t = append(t, i("IMUL"+s, "Signed multiply"))
t = append(t, ic("IMUL3"+s, "Signed multiply by immediate", 3, 3))
t = append(t, ic("MUL"+s, "Unsigned multiply", 1, 1))
t = append(t, ic("DIV"+s, "Unsigned divide", 1, 1))
t = append(t, ic("IDIV"+s, "Signed divide", 1, 1))
}
// Shifts and rotates.
for _, op := range []string{"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR"} {
for _, s := range []string{"B", "W", "L", "Q"} {
t = append(t, i(op+s, op+" shift/rotate"))
}
}
for _, s := range []string{"W", "L", "Q"} {
t = append(t, i("SHLD"+s, "Double-precision left shift"))
t = append(t, i("SHRD"+s, "Double-precision right shift"))
}
// Compare and test.
for _, s := range []string{"B", "W", "L", "Q"} {
t = append(t, ic("CMP"+s, "Compare (subtract, flags only)", 2, 2))
t = append(t, ic("TEST"+s, "AND, flags only", 2, 2))
}
for _, op := range []string{"BT", "BTS", "BTR", "BTC"} {
for _, s := range []string{"W", "L", "Q"} {
t = append(t, i(op+s, "Bit test"+op[1:]))
}
}
// Control flow.
t = append(t, ic("JMP", "Unconditional jump", 1, 1))
for _, cc := range []string{
"EQ", "NE", "Z", "NZ", "L", "LE", "G", "GE", "LT", "GT", "MI", "PL",
"B", "BE", "A", "AE", "CS", "CC", "HI", "LS", "C", "NC",
"S", "NS", "O", "NO", "P", "NP", "PE", "PO", "OS", "OC",
"CXZ", "ECXZ", "RCXZ",
} {
t = append(t, ic("J"+cc, "Conditional jump", 1, 1))
}
t = append(t, ic("CALL", "Call subroutine", 1, 1))
t = append(t, ic("RET", "Return from subroutine", 0, 0))
t = append(t, ic("RETF", "Far return", 0, 0))
t = append(t, ic("NOP", "No operation", 0, 1))
t = append(t, i("INT", "Software interrupt"))
t = append(t, ic("SYSCALL", "System call", 0, 0))
t = append(t, ic("HLT", "Halt", 0, 0))
t = append(t, ic("UD2", "Undefined instruction (trap)", 0, 0))
// Conditional set and move.
for _, cc := range []string{
"EQ", "NE", "L", "LE", "G", "GE", "LT", "GT", "B", "BE", "A", "AE",
"CS", "CC", "HI", "LS", "S", "NS", "O", "NO", "P", "NP", "MI", "PL",
} {
t = append(t, ic("SET"+cc, "Set byte on condition", 1, 1))
}
for _, s := range []string{"L", "Q", "W"} {
for _, cc := range []string{"EQ", "NE", "LT", "LE", "GT", "GE"} {
t = append(t, ic("CMOV"+s+cc, "Conditional move", 2, 2))
}
}
// Bit scanning and counting.
for _, op := range []string{"LZCNT", "TZCNT", "POPCNT", "BSF", "BSR"} {
for _, s := range []string{"W", "L", "Q"} {
t = append(t, ic(op+s, op+" bit operation", 2, 2))
}
}
for _, s := range []string{"L", "Q"} {
t = append(t, ic("BSWAP"+s, "Byte-swap", 1, 1))
}
for _, m := range []string{"CDQ", "CQO", "CBW", "CWDE", "CDQE"} {
t = append(t, ic(m, "Sign-extend accumulator", 0, 0))
}
for _, m := range []string{"CPUID", "RDTSC", "LFENCE", "SFENCE", "MFENCE", "PAUSE"} {
t = append(t, ic(m, "Serialising/system instruction", 0, 0))
}
// SIMD data movement.
for _, m := range []string{
"VMOVDQU", "VMOVDQA", "VMOVUPS", "VMOVUPD", "VMOVAPS", "VMOVAPD",
"VMOVSD", "VMOVSS", "VMOVQ", "VMOVD",
"VMOVDQU32", "VMOVDQU64", "VMOVDQA32", "VMOVDQA64",
"MOVDQU", "MOVDQA", "MOVUPS", "MOVUPD", "MOVAPS", "MOVAPD", "MOVSD", "MOVSS", "MOVD",
} {
t = append(t, i(m, "SIMD move"))
}
// SIMD integer logic and arithmetic.
for _, m := range []string{
"VPXOR", "VPXORD", "VPXORQ", "VPAND", "VPANDN", "VPANDD", "VPANDND", "VPOR", "VPORD", "VPORQ",
"VPADDB", "VPADDW", "VPADDD", "VPADDQ",
"VPSUBB", "VPSUBW", "VPSUBD", "VPSUBQ",
"VPMULLW", "VPMULLD", "VPMULLQ", "VPMULDQ", "VPMULUDQ", "VPMULHUW", "VPMULHW",
"VPADUSB", "VPADUSW", "VPSUBUSB", "VPSUBUSW",
"VPMINSB", "VPMINSW", "VPMINSD", "VPMAXSB", "VPMAXSW", "VPMAXSD",
"VPABSB", "VPABSW", "VPABSD", "VPABSQ",
"VPSLLW", "VPSLLD", "VPSLLQ", "VPSRLW", "VPSRLD", "VPSRLQ",
"VPSRAW", "VPSRAD", "VPSRAQ", "VPSRAVD", "VPSRAVQ", "VPSLLVD", "VPSLLVQ", "VPSRLVD", "VPSRLVQ",
"VPAVGB", "VPAVGW",
"VPACKSSDW", "VPACKSSWB", "VPACKUSDW", "VPACKUSWB",
} {
t = append(t, i(m, "Packed integer SIMD"))
}
// SIMD comparison.
for _, m := range []string{
"VPCMPEQB", "VPCMPEQW", "VPCMPEQD", "VPCMPEQQ",
"VPCMPGTB", "VPCMPGTW", "VPCMPGTD", "VPCMPGTQ",
"VPCMPB", "VPCMPW", "VPCMPD", "VPCMPQ",
} {
t = append(t, i(m, "Packed compare"))
}
// SIMD unpack, shuffle, permute, broadcast, extract, insert.
for _, m := range []string{
"VPUNPCKLBW", "VPUNPCKLWD", "VPUNPCKLDQ", "VPUNPCKLQDQ",
"VPUNPCKHBW", "VPUNPCKHWD", "VPUNPCKHDQ", "VPUNPCKHQDQ",
"VPSHUFD", "VPSHUFHW", "VPSHUFLW", "VPSHUFB",
"VEXTRACTI128", "VEXTRACTF128", "VEXTRACTI32X4", "VEXTRACTI64X4",
"VEXTRACTF32X4", "VEXTRACTF64X4", "VEXTRACTI32X8", "VEXTRACTI64X2",
"VINSERTI128", "VINSERTF128", "VINSERTI32X4", "VINSERTI64X4", "VINSERTF32X4", "VINSERTF64X4",
"VPERMQ", "VPERMD", "VPERMPS", "VPERM2I128", "VPERM2F128",
"VPBROADCASTD", "VPBROADCASTQ", "VPBROADCASTB", "VPBROADCASTW",
"VBROADCASTSD", "VBROADCASTSS", "VBROADCASTI128", "VBROADCASTI32X4",
"VALIGND", "VALIGNQ", "VPBLENDD", "VPBLENDW", "VBLENDVPD", "VBLENDVPS",
"VSHUFPD", "VSHUFPS",
} {
t = append(t, i(m, "Shuffle / permute / broadcast"))
}
// SIMD sign/zero extension and truncation.
for _, m := range []string{
"VPMOVSXBW", "VPMOVSXBD", "VPMOVSXBQ", "VPMOVSXWD", "VPMOVSXWQ", "VPMOVSXDQ",
"VPMOVZXBW", "VPMOVZXBD", "VPMOVZXBQ", "VPMOVZXWD", "VPMOVZXWQ", "VPMOVZXDQ",
"VPMOVDW", "VPMOVQW", "VPMOVQD", "VPMOVDB", "VPMOVWB", "VPMOVQB",
"VPMOVMSKB", "VMOVMSKPS", "VMOVMSKPD", "VMOVQ2DQ", "VMOVDQ2Q",
} {
t = append(t, i(m, "Packed extend / truncate / mask"))
}
// SIMD floating point.
for _, m := range []string{
"VADDPD", "VADDPS", "VADDSD", "VADDSS",
"VSUBPD", "VSUBPS", "VSUBSD", "VSUBSS",
"VMULPD", "VMULPS", "VMULSD", "VMULSS",
"VDIVPD", "VDIVPS", "VDIVSD", "VDIVSS",
"VMINPD", "VMINPS", "VMINSD", "VMINSS", "VMAXPD", "VMAXPS", "VMAXSD", "VMAXSS",
"VXORPD", "VXORPS", "VANDPD", "VANDPS", "VANDNPD", "VANDNPS", "VORPD", "VORPS",
"VUNPCKHPD", "VUNPCKLPD", "VUNPCKHPS", "VUNPCKLPS",
"VSQRTPD", "VSQRTPS", "VSQRTSD", "VSQRTSS", "VRSQRTPS", "VRCPPS",
"VCMPPD", "VCMPPS", "VCMPSD", "VCMPSS",
} {
t = append(t, i(m, "Packed/scalar floating point"))
}
// FMA.
for _, ord := range []string{"132", "213", "231"} {
for _, sfx := range []string{"PD", "PS", "SD", "SS"} {
t = append(t, i("VFMADD"+ord+sfx, "Fused multiply-add"))
t = append(t, i("VFMSUB"+ord+sfx, "Fused multiply-subtract"))
t = append(t, i("VFNMADD"+ord+sfx, "Fused negated multiply-add"))
t = append(t, i("VFNMSUB"+ord+sfx, "Fused negated multiply-subtract"))
}
}
// SIMD conversion.
for _, m := range []string{
"VCVTDQ2PD", "VCVTDQ2PS", "VCVTPD2DQ", "VCVTPS2DQ", "VCVTPD2PS", "VCVTPS2PD",
"VCVTQQ2PD", "VCVTQQ2PS", "VCVTUQQ2PD", "VCVTUQQ2PS",
"VCVTTPD2DQ", "VCVTTPS2DQ", "VCVTSI2SD", "VCVTSI2SS", "VCVTSD2SI", "VCVTSS2SI",
"VCVTSD2SS", "VCVTSS2SD",
"CVTSL2SD", "CVTSL2SS", "CVTSQ2SD", "CVTSQ2SS", "CVTTSD2SL", "CVTTSD2SQ", "CVTTSS2SL",
} {
t = append(t, i(m, "Numeric conversion"))
}
// SIMD zeroing.
t = append(t, ic("VZEROUPPER", "Zero upper halves of YMM/ZMM", 0, 0))
t = append(t, ic("VZEROALL", "Zero all YMM/ZMM state", 0, 0))
// AVX-512 mask register operations.
for _, s := range []string{"B", "W", "D", "Q"} {
t = append(t, ic("KMOV"+s, "Move mask register", 2, 2))
t = append(t, ic("KTEST"+s, "Test mask registers", 2, 2))
t = append(t, i("KAND"+s, "AND masks"))
t = append(t, i("KOR"+s, "OR masks"))
t = append(t, i("KXOR"+s, "XOR masks"))
t = append(t, i("KNOT"+s, "NOT mask"))
t = append(t, i("KANDN"+s, "AND-NOT masks"))
t = append(t, i("KUNPCK"+s, "Unpack masks"))
}
return t
}
+1609
View File
File diff suppressed because it is too large Load Diff
+275
View File
@@ -0,0 +1,275 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package arch provides architecture-specific metadata for GAsm: the register
// files and instruction tables for amd64 and arm64. The metadata powers
// completion, hover documentation, semantic highlighting and the "unknown
// instruction" lint. It is pure data with no dependency on the parser, so it
// can be consulted from any layer.
package arch
import (
"strings"
)
// Arch identifies a target instruction set.
type Arch string
// Supported architectures.
const (
AMD64 Arch = "amd64"
ARM64 Arch = "arm64"
RISCV Arch = "riscv"
LOONG64 Arch = "loong64"
Unknown Arch = ""
)
// FromFilename guesses the target architecture from a source file name. Go
// assembly files conventionally carry a GOARCH suffix such as "_amd64.s",
// "_arm64.s", "_riscv64.s" or "_loong64.s". It returns Unknown when no suffix
// matches.
func FromFilename(name string) Arch {
lower := strings.ToLower(name)
switch {
case strings.Contains(lower, "_amd64"):
return AMD64
case strings.Contains(lower, "_arm64"):
return ARM64
case strings.Contains(lower, "_riscv64"), strings.Contains(lower, "_riscv"):
return RISCV
case strings.Contains(lower, "_loong64"), strings.Contains(lower, "_loong"):
return LOONG64
default:
return Unknown
}
}
// RegClass classifies a register for highlighting and completion grouping.
type RegClass int
// Register classes.
const (
GPR RegClass = iota // general-purpose integer register
GPRSub // sized sub-register (AL, R8D, …)
Vector // SSE/AVX/AVX-512 vector (X/Y/Z)
Mask // AVX-512 mask register (K)
Float // arm64 floating-point register (F)
VecARM // arm64 SIMD/vector register (V)
Special // architecture-special register
)
// String returns a short label for the class.
func (c RegClass) String() string {
switch c {
case GPR:
return "general-purpose"
case GPRSub:
return "sub-register"
case Vector:
return "vector"
case Mask:
return "mask"
case Float:
return "float"
case VecARM:
return "vector (arm64)"
case Special:
return "special"
default:
return "register"
}
}
// Register describes one architectural register.
type Register struct {
Name string
Class RegClass
Desc string
}
// Instr describes one instruction mnemonic.
type Instr struct {
Name string
Summary string
// MinOps and MaxOps bound the operand count; -1 means "unknown/variable"
// and disables the operand-count lint for that instruction.
MinOps int
MaxOps int
}
// Table is the metadata for one architecture.
type Table struct {
Arch Arch
regs map[string]Register
regList []Register
instrs map[string]Instr
instrList []Instr
}
func newTable(a Arch, regs []Register, instrs []Instr) *Table {
t := &Table{
Arch: a,
regs: make(map[string]Register, len(regs)),
regList: regs,
instrs: make(map[string]Instr, len(instrs)),
instrList: instrs,
}
for _, r := range regs {
t.regs[strings.ToUpper(r.Name)] = r
}
for _, in := range instrs {
t.instrs[strings.ToUpper(in.Name)] = in
}
return t
}
// IsRegister reports whether name is a register of this architecture.
func (t *Table) IsRegister(name string) bool {
_, ok := t.regs[strings.ToUpper(name)]
return ok
}
// Register returns the named register.
func (t *Table) Register(name string) (Register, bool) {
r, ok := t.regs[strings.ToUpper(name)]
return r, ok
}
// Registers returns all registers in definition order.
func (t *Table) Registers() []Register { return t.regList }
// Lookup returns the metadata for a mnemonic (case-insensitive).
func (t *Table) Lookup(mnemonic string) (Instr, bool) {
key := strings.ToUpper(mnemonic)
if in, ok := t.instrs[key]; ok {
return in, true
}
// arm64 load/store instructions take a .P (post-index) or .W (pre-index)
// addressing suffix that the assembler front-end strips; mirror that so the
// base instruction is still recognised.
if t.Arch == ARM64 {
for _, suffix := range []string{".P", ".W"} {
if base, ok := strings.CutSuffix(key, suffix); ok {
if in, found := t.instrs[base]; found {
return in, true
}
}
}
}
return Instr{}, false
}
// Instructions returns all instructions in definition order.
func (t *Table) Instructions() []Instr { return t.instrList }
// pseudoRegs are the Plan 9 pseudo-registers, valid on every architecture.
var pseudoRegs = map[string]string{
"FP": "frame pointer: references function arguments and results",
"SP": "stack pointer: the top of the local stack frame",
"SB": "static base: references global symbols",
"PC": "program counter",
}
// IsPseudoReg reports whether name is a Plan 9 pseudo-register.
func IsPseudoReg(name string) bool {
_, ok := pseudoRegs[strings.ToUpper(name)]
return ok
}
// PseudoRegDesc returns the description of a pseudo-register.
func PseudoRegDesc(name string) (string, bool) {
d, ok := pseudoRegs[strings.ToUpper(name)]
return d, ok
}
var (
amd64Table *Table
arm64Table *Table
riscvTable *Table
loong64Table *Table
)
func init() {
amd64Table = buildAMD64()
arm64Table = buildARM64()
riscvTable = buildRISCV()
loong64Table = buildLOONG64()
}
// ForArch returns the table for a, or the amd64 table for Unknown so that
// callers always get a usable default.
func ForArch(a Arch) *Table {
switch a {
case ARM64:
return arm64Table
case RISCV:
return riscvTable
case LOONG64:
return loong64Table
default:
return amd64Table
}
}
// fixedArity lists the few instructions whose operand count is reliable on
// every architecture; relaxCounts leaves these untouched.
var fixedArity = map[string]bool{
"RET": true, "NOP": true, "JMP": true, "CALL": true, "UNDEF": true,
}
// relaxCounts clears operand-count bounds for every instruction except the
// fixed-arity ones. It is applied to architectures (arm64, riscv64, loong64)
// whose instructions have too many operand forms for a single fixed count to be
// reliable, so the operand-count lint stays silent rather than guess.
func relaxCounts(instrs []Instr) []Instr {
for i := range instrs {
if !fixedArity[strings.ToUpper(instrs[i].Name)] {
instrs[i].MinOps = -1
instrs[i].MaxOps = -1
}
}
return instrs
}
// mergedInstrs combines the common opcode list with an architecture-specific
// list (de-duplicated, common first) and enriches the result with the curated
// summaries map.
func mergedInstrs(summaries map[string]Instr, nameSets ...[]string) []Instr {
seen := make(map[string]bool)
var names []string
for _, set := range nameSets {
for _, n := range set {
if !seen[n] {
seen[n] = true
names = append(names, n)
}
}
}
return buildInstrs(names, summaries)
}
// buildInstrs merges the complete generated instruction name list with a
// curated summaries map (keyed by upper-case mnemonic). Instructions without a
// curated entry get an empty summary and an unknown operand count, which keeps
// the operand-count lint silent for them.
func buildInstrs(names []string, summaries map[string]Instr) []Instr {
out := make([]Instr, 0, len(names))
for _, n := range names {
if in, ok := summaries[strings.ToUpper(n)]; ok {
in.Name = n
out = append(out, in)
} else {
out = append(out, Instr{Name: n, MinOps: -1, MaxOps: -1})
}
}
return out
}
// toMap converts a curated instruction slice into an upper-case-keyed map.
func toMap(list []Instr) map[string]Instr {
m := make(map[string]Instr, len(list))
for _, in := range list {
m[strings.ToUpper(in.Name)] = in
}
return m
}
+161
View File
@@ -0,0 +1,161 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
import "testing"
func TestFromFilename(t *testing.T) {
cases := map[string]Arch{
"avx2_amd64.s": AMD64,
"foo_arm64.s": ARM64,
"portable.s": Unknown,
"decode_ARM64.S": ARM64,
"kernels_amd64.s": AMD64,
"kernel_riscv64.s": RISCV,
"kernel_loong64.s": LOONG64,
}
for name, want := range cases {
if got := FromFilename(name); got != want {
t.Errorf("FromFilename(%q) = %q, want %q", name, got, want)
}
}
}
func TestAMD64Registers(t *testing.T) {
tab := ForArch(AMD64)
for _, reg := range []string{"AX", "BX", "R15", "AL", "X0", "Y15", "Z31", "K7"} {
if !tab.IsRegister(reg) {
t.Errorf("amd64: %s should be a register", reg)
}
}
for _, not := range []string{"vec1", "R16", "Z32", "K8", "swin_base"} {
if tab.IsRegister(not) {
t.Errorf("amd64: %s should NOT be a register", not)
}
}
if r, ok := tab.Register("Y0"); !ok || r.Class != Vector {
t.Errorf("Y0 class = %+v, want Vector", r)
}
if r, ok := tab.Register("K1"); !ok || r.Class != Mask {
t.Errorf("K1 class = %+v, want Mask", r)
}
}
func TestARM64Registers(t *testing.T) {
tab := ForArch(ARM64)
for _, reg := range []string{"R0", "R30", "SP", "ZR", "F0", "V31"} {
if !tab.IsRegister(reg) {
t.Errorf("arm64: %s should be a register", reg)
}
}
if tab.IsRegister("AX") {
t.Error("arm64: AX must not be a register")
}
}
func TestAMD64Instructions(t *testing.T) {
tab := ForArch(AMD64)
// Every mnemonic used in the go-flac kernels must be known.
used := []string{
"MOVQ", "MOVL", "MOVB", "LEAQ", "ADDQ", "SUBL", "ANDQ", "ORL", "XORL",
"CMPQ", "CMPL", "TESTQ", "IMUL3L", "IMULQ", "INCW", "LZCNTL", "CMOVLGT",
"SETNE", "JMP", "JGE", "JNE", "JLE", "JLT", "JGT", "JZ", "JNZ", "RET",
"VPCMPEQD", "VPSLLD", "VPXOR", "VPSUBD", "VPADDD", "VPADDQ", "VMOVDQU",
"VEXTRACTI128", "VPSHUFD", "VMOVQ", "VMOVMSKPS", "VZEROUPPER", "VPUNPCKLDQ",
"VPUNPCKHDQ", "VPSRAD", "VPOR", "VPMOVSXDQ", "VPMULDQ", "VPCMPGTQ", "VPSRLQ",
"VPANDN", "VPMOVMSKB", "VPBROADCASTD", "VPSHUFB", "VPACKSSDW", "VPERMQ",
"VPERM2I128", "VPMOVZXDQ", "VFMADD231PD", "VCVTDQ2PD", "VMOVUPD", "VMULPD",
"VADDPD", "VADDSD", "VMOVSD", "VMULSD", "CVTSL2SD", "VEXTRACTF128",
"VPMOVSXWD", "MOVWLSX", "MOVBLZX", "MOVLQSX",
// AVX-512.
"VMOVDQU32", "VPXORD", "VALIGND", "VPERMD", "VPMULLD", "VPMOVDW", "VPSRAQ",
"KTESTW", "KMOVW", "VPXORQ", "VPMOVQD", "VEXTRACTF64X4", "VEXTRACTI64X4",
"VPBROADCASTQ", "VPMULLQ", "VPCMPEQD",
}
for _, m := range used {
if _, ok := tab.Lookup(m); !ok {
t.Errorf("amd64: instruction %s is missing from the table", m)
}
}
}
func TestOperandCounts(t *testing.T) {
tab := ForArch(AMD64)
if in, _ := tab.Lookup("RET"); in.MinOps != 0 || in.MaxOps != 0 {
t.Errorf("RET counts = %d/%d, want 0/0", in.MinOps, in.MaxOps)
}
if in, _ := tab.Lookup("JMP"); in.MinOps != 1 || in.MaxOps != 1 {
t.Errorf("JMP counts = %d/%d, want 1/1", in.MinOps, in.MaxOps)
}
if in, _ := tab.Lookup("MOVQ"); in.MinOps != 2 || in.MaxOps != 2 {
t.Errorf("MOVQ counts = %d/%d, want 2/2", in.MinOps, in.MaxOps)
}
if in, _ := tab.Lookup("IMUL3L"); in.MinOps != 3 || in.MaxOps != 3 {
t.Errorf("IMUL3L counts = %d/%d, want 3/3", in.MinOps, in.MaxOps)
}
}
func TestPseudoRegs(t *testing.T) {
for _, p := range []string{"FP", "SP", "SB", "PC"} {
if !IsPseudoReg(p) {
t.Errorf("%s should be a pseudo-register", p)
}
}
if IsPseudoReg("AX") {
t.Error("AX must not be a pseudo-register")
}
}
func TestRISCVRegisters(t *testing.T) {
tab := ForArch(RISCV)
for _, reg := range []string{"X0", "X31", "F0", "F31", "ZERO", "RA", "SP", "A0", "S11", "T6", "FA0"} {
if !tab.IsRegister(reg) {
t.Errorf("riscv: %s should be a register", reg)
}
}
if tab.IsRegister("AX") {
t.Error("riscv: AX must not be a register")
}
}
func TestLOONG64Registers(t *testing.T) {
tab := ForArch(LOONG64)
for _, reg := range []string{"R0", "R31", "F0", "F31", "V0", "V31", "X0", "X31"} {
if !tab.IsRegister(reg) {
t.Errorf("loong64: %s should be a register", reg)
}
}
if tab.IsRegister("AX") {
t.Error("loong64: AX must not be a register")
}
}
func TestRISCVInstructions(t *testing.T) {
tab := ForArch(RISCV)
for _, m := range []string{"ADD", "ADDI", "SUB", "MUL", "DIV", "BEQ", "BNE", "JAL", "JALR", "LW", "SW", "FADDD", "AMOSWAPD"} {
if _, ok := tab.Lookup(m); !ok {
t.Errorf("riscv: instruction %s is missing", m)
}
}
}
func TestLOONG64Instructions(t *testing.T) {
tab := ForArch(LOONG64)
for _, m := range []string{"ADD", "ADDD", "SUBD", "MULD", "BEQ", "BNE", "BGE", "BGEZ", "JIRL", "MOVD", "MOVW"} {
if _, ok := tab.Lookup(m); !ok {
t.Errorf("loong64: instruction %s is missing", m)
}
}
}
// TestGeneratedTableSize sanity-checks that the toolchain-derived tables are
// the full instruction sets, not a partial hand-written subset.
func TestGeneratedTableSize(t *testing.T) {
min := map[Arch]int{AMD64: 1000, ARM64: 400, RISCV: 800, LOONG64: 600}
for a, want := range min {
if n := len(ForArch(a).Instructions()); n < want {
t.Errorf("%s: only %d instructions, want >= %d (generation incomplete?)", a, n, want)
}
}
}
+181
View File
@@ -0,0 +1,181 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
import "fmt"
func buildARM64() *Table {
return newTable(ARM64, arm64Registers(), relaxCounts(mergedInstrs(arm64Summaries(), commonGeneratedInstrs, arm64GeneratedInstrs, arm64Aliases())))
}
// arm64Aliases are the branch/jump spellings the assembler front-end accepts in
// addition to the generated opcode table (notably the unconditional B and BL).
func arm64Aliases() []string {
return []string{
"B", "BL", "BCS", "BHS", "BCC", "BLO", "BMI", "BPL", "BVS", "BVC",
"BHI", "BLS", "CBZW", "CBNZW", "ADR", "ADRP",
}
}
// arm64Summaries returns the curated documentation/operand-count table keyed by
// upper-case mnemonic; it enriches the complete generated name list.
func arm64Summaries() map[string]Instr { return toMap(arm64Curated()) }
// arm64Registers builds the arm64 (AArch64) register file.
func arm64Registers() []Register {
var regs []Register
add := func(name string, class RegClass, desc string) {
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
}
// General-purpose integer registers R0–R30.
for i := 0; i <= 30; i++ {
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
}
add("ZR", Special, "zero register (reads as 0)")
add("SP", Special, "stack pointer")
add("LR", Special, "link register (alias of R30)")
add("PC", Special, "program counter")
add("RSP", Special, "stack pointer (alias)")
// Floating-point / SIMD registers: F (scalar FP) and V (vector).
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
add(fmt.Sprintf("V%d", i), VecARM, "128-bit SIMD/vector register")
}
return regs
}
// arm64Curated returns the hand-written subset of arm64 (AArch64) instructions
// that carry a summary and/or an operand-count range. 32-bit operations carry
// a W suffix. The authoritative, complete set is arm64GeneratedInstrs.
func arm64Curated() []Instr {
var t []Instr
// Data movement (loads and stores are MOVx with a memory operand).
for _, m := range []string{
"MOVB", "MOVBU", "MOVH", "MOVHU", "MOVW", "MOVWU", "MOVD",
"FMOVS", "FMOVD",
} {
t = append(t, ic(m, "Move / load / store", 2, 2))
}
for _, m := range []string{"MOVK", "MOVN", "MOVZ", "MOVKW", "MOVNW", "MOVZW"} {
t = append(t, i(m, "Move wide constant"))
}
for _, m := range []string{"ADR", "ADRP"} {
t = append(t, ic(m, "Address of label/page", 2, 2))
}
// Integer arithmetic and logic (64-bit and W 32-bit forms).
for _, op := range []string{"ADD", "ADDS", "SUB", "SUBS", "AND", "ANDS", "ORR", "ORN", "EOR", "EON", "BIC", "BICS", "ADC", "ADCS", "SBC", "SBCS"} {
t = append(t, i(op, op+" (64-bit)"))
t = append(t, i(op+"W", op+" (32-bit)"))
}
for _, op := range []string{"NEG", "NGC", "MVN"} {
t = append(t, i(op, op+" (64-bit)"))
t = append(t, i(op+"W", op+" (32-bit)"))
}
for _, op := range []string{"MUL", "MNEG", "SMULL", "UMULL", "SMULH", "UMULH", "MADD", "MSUB", "SMADDL", "UMADDL", "SMSUBL", "UMSUBL"} {
t = append(t, i(op, "Multiply / multiply-accumulate"))
}
for _, op := range []string{"UDIV", "SDIV", "UDIVW", "SDIVW"} {
t = append(t, ic(op, "Divide", 3, 3))
}
// Shifts, rotates and bit manipulation.
for _, op := range []string{"LSL", "LSR", "ASR", "ROR"} {
t = append(t, i(op, op+" shift"))
t = append(t, i(op+"W", op+" shift (32-bit)"))
}
for _, op := range []string{"LSLV", "LSRV", "ASRV", "RORV", "LSLVW", "LSRVW", "ASRVW", "RORVW"} {
t = append(t, i(op, "Variable shift"))
}
for _, op := range []string{"RBIT", "REV", "REV16", "REV32", "REV64", "CLZ", "CLS", "RBITW", "REVW", "CLZW", "CLSW"} {
t = append(t, ic(op, "Bit manipulation", 2, 2))
}
for _, op := range []string{"UBFX", "SBFX", "UBFM", "SBFM", "BFXIL", "EXTR"} {
t = append(t, i(op, "Bitfield extract"))
}
// Compare and test.
for _, op := range []string{"CMP", "CMN", "TST"} {
t = append(t, i(op, op+" (64-bit)"))
t = append(t, i(op+"W", op+" (32-bit)"))
}
// Conditional select.
for _, op := range []string{"CSEL", "CSINC", "CSINV", "CSNEG", "CSET", "CSETM", "CINC", "CINV", "CNEG"} {
t = append(t, i(op, "Conditional select"))
t = append(t, i(op+"W", "Conditional select (32-bit)"))
}
for _, op := range []string{"CCMP", "CCMN", "CCMPW", "CCMNW"} {
t = append(t, i(op, "Conditional compare"))
}
// Control flow.
t = append(t, ic("B", "Unconditional branch", 1, 1))
t = append(t, ic("BL", "Branch with link", 1, 1))
for _, cc := range []string{
"EQ", "NE", "CS", "HS", "CC", "LO", "MI", "PL", "VS", "VC",
"HI", "LS", "GE", "LT", "GT", "LE", "AL", "NV",
} {
t = append(t, ic("B"+cc, "Conditional branch", 1, 1))
}
for _, op := range []string{"CBZ", "CBNZ", "TBZ", "TBNZ"} {
t = append(t, i(op, "Compare/test and branch"))
t = append(t, i(op+"W", "Compare/test and branch (32-bit)"))
}
t = append(t, ic("RET", "Return", 0, 1))
t = append(t, ic("BR", "Branch to register", 1, 1))
t = append(t, ic("BLR", "Branch with link to register", 1, 1))
t = append(t, ic("NOP", "No operation", 0, 1))
t = append(t, ic("BRK", "Breakpoint", 0, 1))
for _, op := range []string{"SVC", "HVC", "SMC"} {
t = append(t, i(op, "Exception generation"))
}
for _, op := range []string{"DMB", "DSB", "ISB"} {
t = append(t, i(op, "Barrier"))
}
for _, op := range []string{"MRS", "MSR"} {
t = append(t, ic(op, "System register access", 2, 2))
}
// Atomics (LSE and load-exclusive/store-exclusive).
for _, op := range []string{
"LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW",
"LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW",
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL", "CASAL",
} {
t = append(t, i(op, "Atomic memory operation"))
}
// Floating-point scalar.
for _, op := range []string{
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
"FMADD", "FMSUB", "FNMADD", "FNMSUB", "FCMP", "FCMPE",
"FCVT", "FCVTZS", "FCVTZU", "FCVTNS", "FCVTNU", "FCVTAS", "FCVTAU",
"SCVTF", "UCVTF", "FRINTM", "FRINTN", "FRINTP", "FRINTZ",
} {
t = append(t, i(op, "Floating-point operation"))
}
t = append(t, i("FMOV", "Floating-point move"))
// NEON / SIMD vector (arrangement carried by the operand suffix).
for _, op := range []string{
"VADD", "VSUB", "VMUL", "VMLA", "VMLS", "VNEG", "VABS", "VMIN", "VMAX",
"VAND", "VORR", "VEOR", "VBIC", "VBIF", "VBSL", "VNOT",
"VDUP", "VMOV", "VMOVI", "VMOVQ",
"VLD1", "VLD2", "VLD3", "VLD4", "VST1", "VST2", "VST3", "VST4",
"VCNT", "VREV16", "VREV32", "VREV64", "VUZP1", "VUZP2", "VZIP1", "VZIP2", "VTRN1", "VTRN2",
"VSHL", "VSHR", "VSSHLL", "VUSHR", "VEXT", "VTBL", "VTBX",
"VADDV", "VUMAXV", "VUMINV", "VSMAXV", "VSMINV",
"VFADD", "VFSUB", "VFMUL", "VFDIV", "VFNEG", "VFABS", "VFMIN", "VFMAX",
"VFMLA", "VFMLS", "VFCVT", "VSCVTF", "VUCVTF", "VFCMEQ", "VFCMGT", "VFCMLT",
"VCMPEQ", "VCMPGT", "VCMPGE", "VSHLL",
} {
t = append(t, i(op, "NEON SIMD vector operation"))
}
return t
}
+547
View File
@@ -0,0 +1,547 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
package arch
// arm64GeneratedInstrs is the complete set of arm64 mnemonics accepted by
// Go's Plan 9 assembler.
var arm64GeneratedInstrs = []string{
"ADC",
"ADCS",
"ADCSW",
"ADCW",
"ADD",
"ADDS",
"ADDSW",
"ADDW",
"ADR",
"ADRP",
"AESD",
"AESE",
"AESIMC",
"AESMC",
"AND",
"ANDS",
"ANDSW",
"ANDW",
"ASR",
"ASRW",
"AT",
"AUTIA1716",
"AUTIASP",
"AUTIB1716",
"AUTIBSP",
"BCC",
"BCS",
"BEQ",
"BFI",
"BFIW",
"BFM",
"BFMW",
"BFXIL",
"BFXILW",
"BGE",
"BGT",
"BHI",
"BHS",
"BIC",
"BICS",
"BICSW",
"BICW",
"BLE",
"BLO",
"BLS",
"BLT",
"BMI",
"BNE",
"BPL",
"BRK",
"BTI",
"BVC",
"BVS",
"CASAD",
"CASALB",
"CASALD",
"CASALH",
"CASALW",
"CASAW",
"CASB",
"CASD",
"CASH",
"CASLD",
"CASLW",
"CASPD",
"CASPW",
"CASW",
"CBNZ",
"CBNZW",
"CBZ",
"CBZW",
"CCMN",
"CCMNW",
"CCMP",
"CCMPW",
"CINC",
"CINCW",
"CINV",
"CINVW",
"CLREX",
"CLS",
"CLSW",
"CLZ",
"CLZW",
"CMN",
"CMNW",
"CMP",
"CMPW",
"CNEG",
"CNEGW",
"CRC32B",
"CRC32CB",
"CRC32CH",
"CRC32CW",
"CRC32CX",
"CRC32H",
"CRC32W",
"CRC32X",
"CSEL",
"CSELW",
"CSET",
"CSETM",
"CSETMW",
"CSETW",
"CSINC",
"CSINCW",
"CSINV",
"CSINVW",
"CSNEG",
"CSNEGW",
"DC",
"DCPS1",
"DCPS2",
"DCPS3",
"DMB",
"DRPS",
"DSB",
"DWORD",
"EON",
"EONW",
"EOR",
"EORW",
"ERET",
"EXTR",
"EXTRW",
"FABSD",
"FABSS",
"FADDD",
"FADDS",
"FCCMPD",
"FCCMPED",
"FCCMPES",
"FCCMPS",
"FCMPD",
"FCMPED",
"FCMPES",
"FCMPS",
"FCSELD",
"FCSELS",
"FCVTDH",
"FCVTDS",
"FCVTHD",
"FCVTHS",
"FCVTSD",
"FCVTSH",
"FCVTZSD",
"FCVTZSDW",
"FCVTZSS",
"FCVTZSSW",
"FCVTZUD",
"FCVTZUDW",
"FCVTZUS",
"FCVTZUSW",
"FDIVD",
"FDIVS",
"FLDPD",
"FLDPQ",
"FLDPS",
"FMADDD",
"FMADDS",
"FMAXD",
"FMAXNMD",
"FMAXNMS",
"FMAXS",
"FMIND",
"FMINNMD",
"FMINNMS",
"FMINS",
"FMOVD",
"FMOVQ",
"FMOVS",
"FMSUBD",
"FMSUBS",
"FMULD",
"FMULS",
"FNEGD",
"FNEGS",
"FNMADDD",
"FNMADDS",
"FNMSUBD",
"FNMSUBS",
"FNMULD",
"FNMULS",
"FRINTAD",
"FRINTAS",
"FRINTID",
"FRINTIS",
"FRINTMD",
"FRINTMS",
"FRINTND",
"FRINTNS",
"FRINTPD",
"FRINTPS",
"FRINTXD",
"FRINTXS",
"FRINTZD",
"FRINTZS",
"FSQRTD",
"FSQRTS",
"FSTPD",
"FSTPQ",
"FSTPS",
"FSUBD",
"FSUBS",
"HINT",
"HLT",
"HVC",
"IC",
"ISB",
"LDADDAB",
"LDADDAD",
"LDADDAH",
"LDADDALB",
"LDADDALD",
"LDADDALH",
"LDADDALW",
"LDADDAW",
"LDADDB",
"LDADDD",
"LDADDH",
"LDADDLB",
"LDADDLD",
"LDADDLH",
"LDADDLW",
"LDADDW",
"LDAR",
"LDARB",
"LDARH",
"LDARW",
"LDAXP",
"LDAXPW",
"LDAXR",
"LDAXRB",
"LDAXRH",
"LDAXRW",
"LDCLRAB",
"LDCLRAD",
"LDCLRAH",
"LDCLRALB",
"LDCLRALD",
"LDCLRALH",
"LDCLRALW",
"LDCLRAW",
"LDCLRB",
"LDCLRD",
"LDCLRH",
"LDCLRLB",
"LDCLRLD",
"LDCLRLH",
"LDCLRLW",
"LDCLRW",
"LDEORAB",
"LDEORAD",
"LDEORAH",
"LDEORALB",
"LDEORALD",
"LDEORALH",
"LDEORALW",
"LDEORAW",
"LDEORB",
"LDEORD",
"LDEORH",
"LDEORLB",
"LDEORLD",
"LDEORLH",
"LDEORLW",
"LDEORW",
"LDORAB",
"LDORAD",
"LDORAH",
"LDORALB",
"LDORALD",
"LDORALH",
"LDORALW",
"LDORAW",
"LDORB",
"LDORD",
"LDORH",
"LDORLB",
"LDORLD",
"LDORLH",
"LDORLW",
"LDORW",
"LDP",
"LDPSW",
"LDPW",
"LDXP",
"LDXPW",
"LDXR",
"LDXRB",
"LDXRH",
"LDXRW",
"LSL",
"LSLW",
"LSR",
"LSRW",
"MADD",
"MADDW",
"MNEG",
"MNEGW",
"MOVB",
"MOVBU",
"MOVD",
"MOVH",
"MOVHU",
"MOVK",
"MOVKW",
"MOVN",
"MOVNW",
"MOVP",
"MOVPD",
"MOVPQ",
"MOVPS",
"MOVPSW",
"MOVPW",
"MOVW",
"MOVWU",
"MOVZ",
"MOVZW",
"MRS",
"MSR",
"MSUB",
"MSUBW",
"MUL",
"MULW",
"MVN",
"MVNW",
"NEG",
"NEGS",
"NEGSW",
"NEGW",
"NGC",
"NGCS",
"NGCSW",
"NGCW",
"NOOP",
"ORN",
"ORNW",
"ORR",
"ORRW",
"PACIASP",
"PACIBSP",
"PRFM",
"PRFUM",
"RBIT",
"RBITW",
"REM",
"REMW",
"REV",
"REV16",
"REV16W",
"REV32",
"REVW",
"ROR",
"RORW",
"SBC",
"SBCS",
"SBCSW",
"SBCW",
"SBFIZ",
"SBFIZW",
"SBFM",
"SBFMW",
"SBFX",
"SBFXW",
"SCVTFD",
"SCVTFS",
"SCVTFWD",
"SCVTFWS",
"SDIV",
"SDIVW",
"SEV",
"SEVL",
"SHA1C",
"SHA1H",
"SHA1M",
"SHA1P",
"SHA1SU0",
"SHA1SU1",
"SHA256H",
"SHA256H2",
"SHA256SU0",
"SHA256SU1",
"SHA512H",
"SHA512H2",
"SHA512SU0",
"SHA512SU1",
"SMADDL",
"SMC",
"SMNEGL",
"SMSUBL",
"SMULH",
"SMULL",
"STLR",
"STLRB",
"STLRH",
"STLRW",
"STLXP",
"STLXPW",
"STLXR",
"STLXRB",
"STLXRH",
"STLXRW",
"STP",
"STPW",
"STXP",
"STXPW",
"STXR",
"STXRB",
"STXRH",
"STXRW",
"SUB",
"SUBS",
"SUBSW",
"SUBW",
"SVC",
"SWPAB",
"SWPAD",
"SWPAH",
"SWPALB",
"SWPALD",
"SWPALH",
"SWPALW",
"SWPAW",
"SWPB",
"SWPD",
"SWPH",
"SWPLB",
"SWPLD",
"SWPLH",
"SWPLW",
"SWPW",
"SXTB",
"SXTBW",
"SXTH",
"SXTHW",
"SXTW",
"SYS",
"SYSL",
"TBNZ",
"TBZ",
"TLBI",
"TST",
"TSTW",
"UBFIZ",
"UBFIZW",
"UBFM",
"UBFMW",
"UBFX",
"UBFXW",
"UCVTFD",
"UCVTFS",
"UCVTFWD",
"UCVTFWS",
"UDIV",
"UDIVW",
"UMADDL",
"UMNEGL",
"UMSUBL",
"UMULH",
"UMULL",
"UREM",
"UREMW",
"UXTB",
"UXTBW",
"UXTH",
"UXTHW",
"UXTW",
"VADD",
"VADDP",
"VADDV",
"VAND",
"VBCAX",
"VBIF",
"VBIT",
"VBSL",
"VCMEQ",
"VCMTST",
"VCNT",
"VDUP",
"VEOR",
"VEOR3",
"VEXT",
"VFMLA",
"VFMLS",
"VLD1",
"VLD1R",
"VLD2",
"VLD2R",
"VLD3",
"VLD3R",
"VLD4",
"VLD4R",
"VMOV",
"VMOVD",
"VMOVI",
"VMOVQ",
"VMOVS",
"VORR",
"VPMULL",
"VPMULL2",
"VRAX1",
"VRBIT",
"VREV16",
"VREV32",
"VREV64",
"VSHL",
"VSLI",
"VSRI",
"VST1",
"VST2",
"VST3",
"VST4",
"VSUB",
"VTBL",
"VTBX",
"VTRN1",
"VTRN2",
"VUADDLV",
"VUADDW",
"VUADDW2",
"VUMAX",
"VUMIN",
"VUSHLL",
"VUSHLL2",
"VUSHR",
"VUSRA",
"VUXTL",
"VUXTL2",
"VUZP1",
"VUZP2",
"VXAR",
"VZIP1",
"VZIP2",
"WFE",
"WFI",
"WORD",
"YIELD",
}
+23
View File
@@ -0,0 +1,23 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/util.go from the Go toolchain.
package arch
// commonGeneratedInstrs is the set of opcodes shared by every architecture
// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).
var commonGeneratedInstrs = []string{
"CALL",
"DUFFCOPY",
"DUFFZERO",
"END",
"FUNCDATA",
"GETCALLERPC",
"JMP",
"NOP",
"PCALIGN",
"PCALIGNMAX",
"PCDATA",
"RET",
"TEXT",
"UNDEF",
}
+81
View File
@@ -0,0 +1,81 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
import "fmt"
func buildLOONG64() *Table {
return newTable(LOONG64, loong64Registers(), relaxCounts(mergedInstrs(loong64Summaries(), commonGeneratedInstrs, loong64GeneratedInstrs, loong64Aliases())))
}
// loong64Aliases are branch spellings the assembler front-end accepts in
// addition to the generated opcode table (notably JAL, BFPF and BFPT).
func loong64Aliases() []string {
return []string{"JAL", "BFPF", "BFPT"}
}
func loong64Summaries() map[string]Instr { return toMap(loong64Curated()) }
// loong64Registers builds the LoongArch (loong64) register file: 32 integer
// (R), 32 floating-point (F), and the LSX/LASX SIMD vector registers (V and X).
func loong64Registers() []Register {
var regs []Register
add := func(name string, class RegClass, desc string) {
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("R%d", i), GPR, "integer register")
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("V%d", i), VecARM, "LSX 128-bit vector register")
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("X%d", i), VecARM, "LASX 256-bit vector register")
}
return regs
}
// loong64Curated is a hand-written subset of common LoongArch instructions
// carrying summaries. The authoritative, complete set is loong64GeneratedInstrs.
func loong64Curated() []Instr {
return []Instr{
ic("ADD", "Integer add (word)", 3, 3), ic("ADDW", "Add word", 3, 3),
ic("ADDD", "Add doubleword", 3, 3), ic("ADDI", "Add immediate", 3, 3),
ic("SUB", "Subtract (word)", 3, 3), ic("SUBW", "Subtract word", 3, 3),
ic("SUBD", "Subtract doubleword", 3, 3),
ic("AND", "Bitwise AND", 3, 3), ic("ANDI", "AND immediate", 3, 3),
ic("OR", "Bitwise OR", 3, 3), ic("ORI", "OR immediate", 3, 3),
ic("XOR", "Bitwise XOR", 3, 3), ic("XORI", "XOR immediate", 3, 3),
ic("NOR", "Bitwise NOR", 3, 3),
ic("MUL", "Multiply (word)", 3, 3), ic("MULW", "Multiply word", 3, 3),
ic("MULD", "Multiply doubleword", 3, 3),
ic("DIV", "Divide (word)", 3, 3), ic("DIVW", "Divide word", 3, 3),
ic("DIVD", "Divide doubleword", 3, 3),
ic("MOD", "Modulo (word)", 3, 3), ic("MODW", "Modulo word", 3, 3),
ic("MODD", "Modulo doubleword", 3, 3),
ic("SLL", "Shift left logical", 3, 3), ic("SRL", "Shift right logical", 3, 3),
ic("SRA", "Shift right arithmetic", 3, 3), ic("ROTR", "Rotate right", 3, 3),
ic("SLT", "Set if less than", 3, 3), ic("SLTU", "Set if less than unsigned", 3, 3),
ic("SLTI", "Set if less than immediate", 3, 3),
ic("LD", "Load doubleword", 2, 2), ic("LDW", "Load word", 2, 2),
ic("LDH", "Load halfword", 2, 2), ic("LDB", "Load byte", 2, 2),
ic("ST", "Store doubleword", 2, 2), ic("STW", "Store word", 2, 2),
ic("STH", "Store halfword", 2, 2), ic("STB", "Store byte", 2, 2),
ic("BEQ", "Branch if equal", 3, 3), ic("BNE", "Branch if not equal", 3, 3),
ic("BLT", "Branch if less than", 3, 3), ic("BGE", "Branch if greater or equal", 3, 3),
ic("BLTU", "Branch if less than unsigned", 3, 3), ic("BGEU", "Branch if greater or equal unsigned", 3, 3),
ic("B", "Unconditional branch", 1, 1), ic("BL", "Branch with link", 1, 1),
ic("JIRL", "Jump indirect with link", 1, 3),
ic("RET", "Return", 0, 1), ic("NOP", "No operation", 0, 1),
i("SYSCALL", "System call"), i("BREAK", "Breakpoint"), i("DBAR", "Barrier"),
ic("FADDS", "FP add (single)", 3, 3), ic("FADDD", "FP add (double)", 3, 3),
ic("FSUBS", "FP subtract (single)", 3, 3), ic("FSUBD", "FP subtract (double)", 3, 3),
ic("FMULS", "FP multiply (single)", 3, 3), ic("FMULD", "FP multiply (double)", 3, 3),
ic("FDIVS", "FP divide (single)", 3, 3), ic("FDIVD", "FP divide (double)", 3, 3),
ic("MOV", "Move register", 2, 2),
}
}
+808
View File
@@ -0,0 +1,808 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
package arch
// loong64GeneratedInstrs is the complete set of loong64 mnemonics accepted by
// Go's Plan 9 assembler.
var loong64GeneratedInstrs = []string{
"ABSD",
"ABSF",
"ADD",
"ADDD",
"ADDF",
"ADDV",
"ADDV16",
"ADDVU",
"ADDW",
"ALSLV",
"ALSLW",
"ALSLWU",
"AMADDDBV",
"AMADDDBW",
"AMADDV",
"AMADDW",
"AMANDDBV",
"AMANDDBW",
"AMANDV",
"AMANDW",
"AMCASB",
"AMCASDBB",
"AMCASDBH",
"AMCASDBV",
"AMCASDBW",
"AMCASH",
"AMCASV",
"AMCASW",
"AMMAXDBV",
"AMMAXDBVU",
"AMMAXDBW",
"AMMAXDBWU",
"AMMAXV",
"AMMAXVU",
"AMMAXW",
"AMMAXWU",
"AMMINDBV",
"AMMINDBVU",
"AMMINDBW",
"AMMINDBWU",
"AMMINV",
"AMMINVU",
"AMMINW",
"AMMINWU",
"AMORDBV",
"AMORDBW",
"AMORV",
"AMORW",
"AMSWAPB",
"AMSWAPDBB",
"AMSWAPDBH",
"AMSWAPDBV",
"AMSWAPDBW",
"AMSWAPH",
"AMSWAPV",
"AMSWAPW",
"AMXORDBV",
"AMXORDBW",
"AMXORV",
"AMXORW",
"AND",
"ANDN",
"BEQ",
"BFPF",
"BFPT",
"BGE",
"BGEU",
"BGEZ",
"BGTZ",
"BITREV4B",
"BITREV8B",
"BITREVV",
"BITREVW",
"BLEZ",
"BLT",
"BLTU",
"BLTZ",
"BNE",
"BREAK",
"BSTRINSV",
"BSTRINSW",
"BSTRPICKV",
"BSTRPICKW",
"CLOV",
"CLOW",
"CLZV",
"CLZW",
"CMPEQD",
"CMPEQF",
"CMPGED",
"CMPGEF",
"CMPGTD",
"CMPGTF",
"CPUCFG",
"CRCCWBW",
"CRCCWHW",
"CRCCWVW",
"CRCCWWW",
"CRCWBW",
"CRCWHW",
"CRCWVW",
"CRCWWW",
"CTOV",
"CTOW",
"CTZV",
"CTZW",
"DBAR",
"DIV",
"DIVD",
"DIVF",
"DIVU",
"DIVV",
"DIVVU",
"DIVW",
"DIVWU",
"EXTWB",
"EXTWH",
"FCLASSD",
"FCLASSF",
"FCOPYSGD",
"FCOPYSGF",
"FFINTDV",
"FFINTDW",
"FFINTFV",
"FFINTFW",
"FLOGBD",
"FLOGBF",
"FMADDD",
"FMADDF",
"FMAXAD",
"FMAXAF",
"FMAXD",
"FMAXF",
"FMINAD",
"FMINAF",
"FMIND",
"FMINF",
"FMSUBD",
"FMSUBF",
"FNMADDD",
"FNMADDF",
"FNMSUBD",
"FNMSUBF",
"FSCALEBD",
"FSCALEBF",
"FSEL",
"FTINTRMVD",
"FTINTRMVF",
"FTINTRMWD",
"FTINTRMWF",
"FTINTRNEVD",
"FTINTRNEVF",
"FTINTRNEWD",
"FTINTRNEWF",
"FTINTRPVD",
"FTINTRPVF",
"FTINTRPWD",
"FTINTRPWF",
"FTINTRZVD",
"FTINTRZVF",
"FTINTRZWD",
"FTINTRZWF",
"FTINTVD",
"FTINTVF",
"FTINTWD",
"FTINTWF",
"JIRL",
"LL",
"LLV",
"LU12IW",
"LU32ID",
"LU52ID",
"LUI",
"MASKEQZ",
"MASKNEZ",
"MOVB",
"MOVBU",
"MOVD",
"MOVDF",
"MOVDV",
"MOVDW",
"MOVF",
"MOVFD",
"MOVFV",
"MOVFW",
"MOVH",
"MOVHU",
"MOVV",
"MOVVD",
"MOVVF",
"MOVVP",
"MOVW",
"MOVWD",
"MOVWF",
"MOVWP",
"MOVWU",
"MUL",
"MULD",
"MULF",
"MULH",
"MULHU",
"MULHV",
"MULHVU",
"MULV",
"MULVU",
"MULW",
"MULWVW",
"MULWVWU",
"NEGD",
"NEGF",
"NEGV",
"NEGW",
"NOOP",
"NOR",
"OR",
"ORN",
"PCADDU12I",
"PCALAU12I",
"PRELD",
"PRELDX",
"RDTIMED",
"RDTIMEHW",
"RDTIMELW",
"REM",
"REMU",
"REMV",
"REMVU",
"REMW",
"REMWU",
"REVB2H",
"REVB2W",
"REVB4H",
"REVBV",
"REVH2W",
"REVHV",
"RFE",
"ROTR",
"ROTRV",
"SC",
"SCV",
"SGT",
"SGTU",
"SLL",
"SLLV",
"SQRTD",
"SQRTF",
"SRA",
"SRAV",
"SRL",
"SRLV",
"SUB",
"SUBD",
"SUBF",
"SUBV",
"SUBVU",
"SUBW",
"SYSCALL",
"TEQ",
"TNE",
"TRUNCDV",
"TRUNCDW",
"TRUNCFV",
"TRUNCFW",
"VADDB",
"VADDBU",
"VADDD",
"VADDF",
"VADDH",
"VADDHU",
"VADDQ",
"VADDV",
"VADDVU",
"VADDW",
"VADDWEVHB",
"VADDWEVHBU",
"VADDWEVQV",
"VADDWEVQVU",
"VADDWEVVW",
"VADDWEVVWU",
"VADDWEVWH",
"VADDWEVWHU",
"VADDWODHB",
"VADDWODHBU",
"VADDWODQV",
"VADDWODQVU",
"VADDWODVW",
"VADDWODVWU",
"VADDWODWH",
"VADDWODWHU",
"VADDWU",
"VANDB",
"VANDNV",
"VANDV",
"VBITCLRB",
"VBITCLRH",
"VBITCLRV",
"VBITCLRW",
"VBITREVB",
"VBITREVH",
"VBITREVV",
"VBITREVW",
"VBITSETB",
"VBITSETH",
"VBITSETV",
"VBITSETW",
"VDIVB",
"VDIVBU",
"VDIVD",
"VDIVF",
"VDIVH",
"VDIVHU",
"VDIVV",
"VDIVVU",
"VDIVW",
"VDIVWU",
"VEXTRINSB",
"VEXTRINSH",
"VEXTRINSV",
"VEXTRINSW",
"VFCLASSD",
"VFCLASSF",
"VFRECIPD",
"VFRECIPF",
"VFRINTD",
"VFRINTF",
"VFRINTRMD",
"VFRINTRMF",
"VFRINTRNED",
"VFRINTRNEF",
"VFRINTRPD",
"VFRINTRPF",
"VFRINTRZD",
"VFRINTRZF",
"VFRSQRTD",
"VFRSQRTF",
"VFSQRTD",
"VFSQRTF",
"VILVHB",
"VILVHH",
"VILVHV",
"VILVHW",
"VILVLB",
"VILVLH",
"VILVLV",
"VILVLW",
"VMADDB",
"VMADDH",
"VMADDV",
"VMADDW",
"VMADDWEVHB",
"VMADDWEVHBU",
"VMADDWEVHBUB",
"VMADDWEVQV",
"VMADDWEVQVU",
"VMADDWEVQVUV",
"VMADDWEVVW",
"VMADDWEVVWU",
"VMADDWEVVWUW",
"VMADDWEVWH",
"VMADDWEVWHU",
"VMADDWEVWHUH",
"VMADDWODHB",
"VMADDWODHBU",
"VMADDWODHBUB",
"VMADDWODQV",
"VMADDWODQVU",
"VMADDWODQVUV",
"VMADDWODVW",
"VMADDWODVWU",
"VMADDWODVWUW",
"VMADDWODWH",
"VMADDWODWHU",
"VMADDWODWHUH",
"VMODB",
"VMODBU",
"VMODH",
"VMODHU",
"VMODV",
"VMODVU",
"VMODW",
"VMODWU",
"VMOVQ",
"VMSUBB",
"VMSUBH",
"VMSUBV",
"VMSUBW",
"VMUHB",
"VMUHBU",
"VMUHH",
"VMUHHU",
"VMUHV",
"VMUHVU",
"VMUHW",
"VMUHWU",
"VMULB",
"VMULD",
"VMULF",
"VMULH",
"VMULV",
"VMULW",
"VMULWEVHB",
"VMULWEVHBU",
"VMULWEVHBUB",
"VMULWEVQV",
"VMULWEVQVU",
"VMULWEVQVUV",
"VMULWEVVW",
"VMULWEVVWU",
"VMULWEVVWUW",
"VMULWEVWH",
"VMULWEVWHU",
"VMULWEVWHUH",
"VMULWODHB",
"VMULWODHBU",
"VMULWODHBUB",
"VMULWODQV",
"VMULWODQVU",
"VMULWODQVUV",
"VMULWODVW",
"VMULWODVWU",
"VMULWODVWUW",
"VMULWODWH",
"VMULWODWHU",
"VMULWODWHUH",
"VNEGB",
"VNEGH",
"VNEGV",
"VNEGW",
"VNORB",
"VNORV",
"VORB",
"VORNV",
"VORV",
"VPCNTB",
"VPCNTH",
"VPCNTV",
"VPCNTW",
"VPERMIW",
"VROTRB",
"VROTRH",
"VROTRV",
"VROTRW",
"VSADDB",
"VSADDBU",
"VSADDH",
"VSADDHU",
"VSADDV",
"VSADDVU",
"VSADDW",
"VSADDWU",
"VSEQB",
"VSEQH",
"VSEQV",
"VSEQW",
"VSETALLNEB",
"VSETALLNEH",
"VSETALLNEV",
"VSETALLNEW",
"VSETANYEQB",
"VSETANYEQH",
"VSETANYEQV",
"VSETANYEQW",
"VSETEQV",
"VSETNEV",
"VSHUF4IB",
"VSHUF4IH",
"VSHUF4IV",
"VSHUF4IW",
"VSHUFB",
"VSHUFH",
"VSHUFV",
"VSHUFW",
"VSLLB",
"VSLLH",
"VSLLV",
"VSLLW",
"VSLTB",
"VSLTBU",
"VSLTH",
"VSLTHU",
"VSLTV",
"VSLTVU",
"VSLTW",
"VSLTWU",
"VSRAB",
"VSRAH",
"VSRAV",
"VSRAW",
"VSRLB",
"VSRLH",
"VSRLV",
"VSRLW",
"VSSUBB",
"VSSUBBU",
"VSSUBH",
"VSSUBHU",
"VSSUBV",
"VSSUBVU",
"VSSUBW",
"VSSUBWU",
"VSUBB",
"VSUBBU",
"VSUBD",
"VSUBF",
"VSUBH",
"VSUBHU",
"VSUBQ",
"VSUBV",
"VSUBVU",
"VSUBW",
"VSUBWEVHB",
"VSUBWEVHBU",
"VSUBWEVQV",
"VSUBWEVQVU",
"VSUBWEVVW",
"VSUBWEVVWU",
"VSUBWEVWH",
"VSUBWEVWHU",
"VSUBWODHB",
"VSUBWODHBU",
"VSUBWODQV",
"VSUBWODQVU",
"VSUBWODVW",
"VSUBWODVWU",
"VSUBWODWH",
"VSUBWODWHU",
"VSUBWU",
"VXORB",
"VXORV",
"WORD",
"XOR",
"XVADDB",
"XVADDBU",
"XVADDD",
"XVADDF",
"XVADDH",
"XVADDHU",
"XVADDQ",
"XVADDV",
"XVADDVU",
"XVADDW",
"XVADDWEVHB",
"XVADDWEVHBU",
"XVADDWEVQV",
"XVADDWEVQVU",
"XVADDWEVVW",
"XVADDWEVVWU",
"XVADDWEVWH",
"XVADDWEVWHU",
"XVADDWODHB",
"XVADDWODHBU",
"XVADDWODQV",
"XVADDWODQVU",
"XVADDWODVW",
"XVADDWODVWU",
"XVADDWODWH",
"XVADDWODWHU",
"XVADDWU",
"XVANDB",
"XVANDNV",
"XVANDV",
"XVBITCLRB",
"XVBITCLRH",
"XVBITCLRV",
"XVBITCLRW",
"XVBITREVB",
"XVBITREVH",
"XVBITREVV",
"XVBITREVW",
"XVBITSETB",
"XVBITSETH",
"XVBITSETV",
"XVBITSETW",
"XVDIVB",
"XVDIVBU",
"XVDIVD",
"XVDIVF",
"XVDIVH",
"XVDIVHU",
"XVDIVV",
"XVDIVVU",
"XVDIVW",
"XVDIVWU",
"XVEXTRINSB",
"XVEXTRINSH",
"XVEXTRINSV",
"XVEXTRINSW",
"XVFCLASSD",
"XVFCLASSF",
"XVFRECIPD",
"XVFRECIPF",
"XVFRINTD",
"XVFRINTF",
"XVFRINTRMD",
"XVFRINTRMF",
"XVFRINTRNED",
"XVFRINTRNEF",
"XVFRINTRPD",
"XVFRINTRPF",
"XVFRINTRZD",
"XVFRINTRZF",
"XVFRSQRTD",
"XVFRSQRTF",
"XVFSQRTD",
"XVFSQRTF",
"XVILVHB",
"XVILVHH",
"XVILVHV",
"XVILVHW",
"XVILVLB",
"XVILVLH",
"XVILVLV",
"XVILVLW",
"XVMADDB",
"XVMADDH",
"XVMADDV",
"XVMADDW",
"XVMADDWEVHB",
"XVMADDWEVHBU",
"XVMADDWEVHBUB",
"XVMADDWEVQV",
"XVMADDWEVQVU",
"XVMADDWEVQVUV",
"XVMADDWEVVW",
"XVMADDWEVVWU",
"XVMADDWEVVWUW",
"XVMADDWEVWH",
"XVMADDWEVWHU",
"XVMADDWEVWHUH",
"XVMADDWODHB",
"XVMADDWODHBU",
"XVMADDWODHBUB",
"XVMADDWODQV",
"XVMADDWODQVU",
"XVMADDWODQVUV",
"XVMADDWODVW",
"XVMADDWODVWU",
"XVMADDWODVWUW",
"XVMADDWODWH",
"XVMADDWODWHU",
"XVMADDWODWHUH",
"XVMODB",
"XVMODBU",
"XVMODH",
"XVMODHU",
"XVMODV",
"XVMODVU",
"XVMODW",
"XVMODWU",
"XVMOVQ",
"XVMSUBB",
"XVMSUBH",
"XVMSUBV",
"XVMSUBW",
"XVMUHB",
"XVMUHBU",
"XVMUHH",
"XVMUHHU",
"XVMUHV",
"XVMUHVU",
"XVMUHW",
"XVMUHWU",
"XVMULB",
"XVMULD",
"XVMULF",
"XVMULH",
"XVMULV",
"XVMULW",
"XVMULWEVHB",
"XVMULWEVHBU",
"XVMULWEVHBUB",
"XVMULWEVQV",
"XVMULWEVQVU",
"XVMULWEVQVUV",
"XVMULWEVVW",
"XVMULWEVVWU",
"XVMULWEVVWUW",
"XVMULWEVWH",
"XVMULWEVWHU",
"XVMULWEVWHUH",
"XVMULWODHB",
"XVMULWODHBU",
"XVMULWODHBUB",
"XVMULWODQV",
"XVMULWODQVU",
"XVMULWODQVUV",
"XVMULWODVW",
"XVMULWODVWU",
"XVMULWODVWUW",
"XVMULWODWH",
"XVMULWODWHU",
"XVMULWODWHUH",
"XVNEGB",
"XVNEGH",
"XVNEGV",
"XVNEGW",
"XVNORB",
"XVNORV",
"XVORB",
"XVORNV",
"XVORV",
"XVPCNTB",
"XVPCNTH",
"XVPCNTV",
"XVPCNTW",
"XVPERMIQ",
"XVPERMIV",
"XVPERMIW",
"XVROTRB",
"XVROTRH",
"XVROTRV",
"XVROTRW",
"XVSADDB",
"XVSADDBU",
"XVSADDH",
"XVSADDHU",
"XVSADDV",
"XVSADDVU",
"XVSADDW",
"XVSADDWU",
"XVSEQB",
"XVSEQH",
"XVSEQV",
"XVSEQW",
"XVSETALLNEB",
"XVSETALLNEH",
"XVSETALLNEV",
"XVSETALLNEW",
"XVSETANYEQB",
"XVSETANYEQH",
"XVSETANYEQV",
"XVSETANYEQW",
"XVSETEQV",
"XVSETNEV",
"XVSHUF4IB",
"XVSHUF4IH",
"XVSHUF4IV",
"XVSHUF4IW",
"XVSHUFB",
"XVSHUFH",
"XVSHUFV",
"XVSHUFW",
"XVSLLB",
"XVSLLH",
"XVSLLV",
"XVSLLW",
"XVSLTB",
"XVSLTBU",
"XVSLTH",
"XVSLTHU",
"XVSLTV",
"XVSLTVU",
"XVSLTW",
"XVSLTWU",
"XVSRAB",
"XVSRAH",
"XVSRAV",
"XVSRAW",
"XVSRLB",
"XVSRLH",
"XVSRLV",
"XVSRLW",
"XVSSUBB",
"XVSSUBBU",
"XVSSUBH",
"XVSSUBHU",
"XVSSUBV",
"XVSSUBVU",
"XVSSUBW",
"XVSSUBWU",
"XVSUBB",
"XVSUBBU",
"XVSUBD",
"XVSUBF",
"XVSUBH",
"XVSUBHU",
"XVSUBQ",
"XVSUBV",
"XVSUBVU",
"XVSUBW",
"XVSUBWEVHB",
"XVSUBWEVHBU",
"XVSUBWEVQV",
"XVSUBWEVQVU",
"XVSUBWEVVW",
"XVSUBWEVVWU",
"XVSUBWEVWH",
"XVSUBWEVWHU",
"XVSUBWODHB",
"XVSUBWODHBU",
"XVSUBWODQV",
"XVSUBWODQVU",
"XVSUBWODVW",
"XVSUBWODVWU",
"XVSUBWODWH",
"XVSUBWODWHU",
"XVSUBWU",
"XVXORB",
"XVXORV",
}
+100
View File
@@ -0,0 +1,100 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
import "fmt"
func buildRISCV() *Table {
return newTable(RISCV, riscvRegisters(), relaxCounts(mergedInstrs(riscvSummaries(), commonGeneratedInstrs, riscvGeneratedInstrs)))
}
func riscvSummaries() map[string]Instr { return toMap(riscvCurated()) }
// riscvRegisters builds the RISC-V register file: the numbered integer (X) and
// floating-point (F) registers plus their standard ABI aliases.
func riscvRegisters() []Register {
var regs []Register
add := func(name string, class RegClass, desc string) {
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("X%d", i), GPR, "integer register")
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
}
// Integer ABI aliases.
for _, n := range []string{"ZERO", "RA", "SP", "GP", "TP", "FP", "LR", "TMP"} {
add(n, GPR, "integer ABI alias")
}
for i := 0; i <= 6; i++ {
add(fmt.Sprintf("T%d", i), GPR, "temporary")
}
for i := 0; i <= 11; i++ {
add(fmt.Sprintf("S%d", i), GPR, "saved register")
}
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("A%d", i), GPR, "argument/result register")
}
// Floating-point ABI aliases.
for i := 0; i <= 11; i++ {
add(fmt.Sprintf("FT%d", i), Float, "FP temporary")
}
for i := 0; i <= 11; i++ {
add(fmt.Sprintf("FS%d", i), Float, "FP saved register")
}
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("FA%d", i), Float, "FP argument/result register")
}
return regs
}
// riscvCurated is a hand-written subset of common RISC-V instructions carrying
// summaries. The authoritative, complete set is riscvGeneratedInstrs.
func riscvCurated() []Instr {
return []Instr{
ic("ADD", "Integer add", 3, 3), ic("ADDI", "Add immediate", 3, 3),
ic("ADDIW", "Add immediate (32-bit)", 3, 3), ic("ADDW", "Add (32-bit)", 3, 3),
ic("SUB", "Integer subtract", 3, 3), ic("SUBW", "Subtract (32-bit)", 3, 3),
ic("AND", "Bitwise AND", 3, 3), ic("ANDI", "AND immediate", 3, 3),
ic("OR", "Bitwise OR", 3, 3), ic("ORI", "OR immediate", 3, 3),
ic("XOR", "Bitwise XOR", 3, 3), ic("XORI", "XOR immediate", 3, 3),
ic("SLL", "Shift left logical", 3, 3), ic("SLLI", "Shift left logical immediate", 3, 3),
ic("SRL", "Shift right logical", 3, 3), ic("SRLI", "Shift right logical immediate", 3, 3),
ic("SRA", "Shift right arithmetic", 3, 3), ic("SRAI", "Shift right arithmetic immediate", 3, 3),
ic("SLT", "Set if less than", 3, 3), ic("SLTI", "Set if less than immediate", 3, 3),
ic("SLTU", "Set if less than unsigned", 3, 3), ic("SLTIU", "Set if less than unsigned immediate", 3, 3),
ic("MUL", "Multiply", 3, 3), ic("MULH", "Multiply high", 3, 3),
ic("MULHU", "Multiply high unsigned", 3, 3), ic("MULHSU", "Multiply high signed/unsigned", 3, 3),
ic("DIV", "Divide", 3, 3), ic("DIVU", "Divide unsigned", 3, 3),
ic("REM", "Remainder", 3, 3), ic("REMU", "Remainder unsigned", 3, 3),
ic("MULW", "Multiply (32-bit)", 3, 3), ic("DIVW", "Divide (32-bit)", 3, 3),
ic("LB", "Load byte", 2, 2), ic("LBU", "Load byte unsigned", 2, 2),
ic("LH", "Load halfword", 2, 2), ic("LHU", "Load halfword unsigned", 2, 2),
ic("LW", "Load word", 2, 2), ic("LWU", "Load word unsigned", 2, 2),
ic("LD", "Load doubleword", 2, 2),
ic("SB", "Store byte", 2, 2), ic("SH", "Store halfword", 2, 2),
ic("SW", "Store word", 2, 2), ic("SD", "Store doubleword", 2, 2),
ic("LUI", "Load upper immediate", 2, 2), ic("AUIPC", "Add upper immediate to PC", 2, 2),
ic("BEQ", "Branch if equal", 3, 3), ic("BNE", "Branch if not equal", 3, 3),
ic("BLT", "Branch if less than", 3, 3), ic("BGE", "Branch if greater or equal", 3, 3),
ic("BLTU", "Branch if less than unsigned", 3, 3), ic("BGEU", "Branch if greater or equal unsigned", 3, 3),
ic("JAL", "Jump and link", 1, 2), ic("JALR", "Jump and link register", 1, 3),
i("JMP", "Unconditional jump"), i("CALL", "Call subroutine"),
ic("RET", "Return", 0, 1), i("ECALL", "Environment call"), i("EBREAK", "Breakpoint"),
i("FENCE", "Memory barrier"), i("CSR", "Control/status register access"),
// Floating point.
ic("FADDS", "FP add (single)", 3, 3), ic("FADDD", "FP add (double)", 3, 3),
ic("FSUBS", "FP subtract (single)", 3, 3), ic("FSUBD", "FP subtract (double)", 3, 3),
ic("FMULS", "FP multiply (single)", 3, 3), ic("FMULD", "FP multiply (double)", 3, 3),
ic("FDIVS", "FP divide (single)", 3, 3), ic("FDIVD", "FP divide (double)", 3, 3),
ic("FLW", "FP load word", 2, 2), ic("FLD", "FP load doubleword", 2, 2),
ic("FSW", "FP store word", 2, 2), ic("FSD", "FP store doubleword", 2, 2),
// Atomics.
i("LRW", "Load-reserved word"), i("LRD", "Load-reserved doubleword"),
i("SCW", "Store-conditional word"), i("SCD", "Store-conditional doubleword"),
i("AMOSWAPW", "Atomic swap word"), i("AMOSWAPD", "Atomic swap doubleword"),
i("AMOADDW", "Atomic add word"), i("AMOADDD", "Atomic add doubleword"),
}
}
+970
View File
@@ -0,0 +1,970 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
package arch
// riscvGeneratedInstrs is the complete set of riscv mnemonics accepted by
// Go's Plan 9 assembler.
var riscvGeneratedInstrs = []string{
"ADD",
"ADDI",
"ADDIW",
"ADDUW",
"ADDW",
"AMOADDD",
"AMOADDW",
"AMOANDD",
"AMOANDW",
"AMOMAXD",
"AMOMAXUD",
"AMOMAXUW",
"AMOMAXW",
"AMOMIND",
"AMOMINUD",
"AMOMINUW",
"AMOMINW",
"AMOORD",
"AMOORW",
"AMOSWAPD",
"AMOSWAPW",
"AMOXORD",
"AMOXORW",
"AND",
"ANDI",
"ANDN",
"AUIPC",
"BCLR",
"BCLRI",
"BEQ",
"BEQZ",
"BEXT",
"BEXTI",
"BGE",
"BGEU",
"BGEZ",
"BGT",
"BGTU",
"BGTZ",
"BINV",
"BINVI",
"BLE",
"BLEU",
"BLEZ",
"BLT",
"BLTU",
"BLTZ",
"BNE",
"BNEZ",
"BSET",
"BSETI",
"CADD",
"CADDI",
"CADDI16SP",
"CADDI4SPN",
"CADDIW",
"CADDW",
"CAND",
"CANDI",
"CBEQZ",
"CBNEZ",
"CEBREAK",
"CFLD",
"CFLDSP",
"CFSD",
"CFSDSP",
"CJ",
"CJALR",
"CJR",
"CLD",
"CLDSP",
"CLI",
"CLUI",
"CLW",
"CLWSP",
"CLZ",
"CLZW",
"CMV",
"CNOP",
"COR",
"CPOP",
"CPOPW",
"CSD",
"CSDSP",
"CSLLI",
"CSRAI",
"CSRLI",
"CSRRC",
"CSRRCI",
"CSRRS",
"CSRRSI",
"CSRRW",
"CSRRWI",
"CSUB",
"CSUBW",
"CSW",
"CSWSP",
"CTZ",
"CTZW",
"CXOR",
"CZEROEQZ",
"CZERONEZ",
"DIV",
"DIVU",
"DIVUW",
"DIVW",
"DRET",
"EBREAK",
"ECALL",
"FABSD",
"FABSS",
"FADDD",
"FADDQ",
"FADDS",
"FCLASSD",
"FCLASSQ",
"FCLASSS",
"FCVTDL",
"FCVTDLU",
"FCVTDQ",
"FCVTDS",
"FCVTDW",
"FCVTDWU",
"FCVTLD",
"FCVTLQ",
"FCVTLS",
"FCVTLUD",
"FCVTLUQ",
"FCVTLUS",
"FCVTQD",
"FCVTQL",
"FCVTQLU",
"FCVTQS",
"FCVTQW",
"FCVTQWU",
"FCVTSD",
"FCVTSL",
"FCVTSLU",
"FCVTSQ",
"FCVTSW",
"FCVTSWU",
"FCVTWD",
"FCVTWQ",
"FCVTWS",
"FCVTWUD",
"FCVTWUQ",
"FCVTWUS",
"FDIVD",
"FDIVQ",
"FDIVS",
"FENCE",
"FEQD",
"FEQQ",
"FEQS",
"FLD",
"FLED",
"FLEQ",
"FLES",
"FLQ",
"FLTD",
"FLTQ",
"FLTS",
"FLW",
"FMADDD",
"FMADDQ",
"FMADDS",
"FMAXD",
"FMAXQ",
"FMAXS",
"FMIND",
"FMINQ",
"FMINS",
"FMSUBD",
"FMSUBQ",
"FMSUBS",
"FMULD",
"FMULQ",
"FMULS",
"FMVDX",
"FMVSX",
"FMVWX",
"FMVXD",
"FMVXS",
"FMVXW",
"FNED",
"FNEGD",
"FNEGS",
"FNES",
"FNMADDD",
"FNMADDQ",
"FNMADDS",
"FNMSUBD",
"FNMSUBQ",
"FNMSUBS",
"FSD",
"FSGNJD",
"FSGNJND",
"FSGNJNQ",
"FSGNJNS",
"FSGNJQ",
"FSGNJS",
"FSGNJXD",
"FSGNJXQ",
"FSGNJXS",
"FSQ",
"FSQRTD",
"FSQRTQ",
"FSQRTS",
"FSUBD",
"FSUBQ",
"FSUBS",
"FSW",
"JAL",
"JALR",
"LB",
"LBU",
"LD",
"LH",
"LHU",
"LRD",
"LRW",
"LUI",
"LW",
"LWU",
"MAX",
"MAXU",
"MIN",
"MINU",
"MOV",
"MOVB",
"MOVBU",
"MOVD",
"MOVF",
"MOVH",
"MOVHU",
"MOVW",
"MOVWU",
"MRET",
"MUL",
"MULH",
"MULHSU",
"MULHU",
"MULW",
"NEG",
"NEGW",
"NOT",
"OR",
"ORCB",
"ORI",
"ORN",
"RDCYCLE",
"RDINSTRET",
"RDTIME",
"REM",
"REMU",
"REMUW",
"REMW",
"REV8",
"ROL",
"ROLW",
"ROR",
"RORI",
"RORIW",
"RORW",
"SB",
"SBREAK",
"SCALL",
"SCD",
"SCW",
"SD",
"SEQZ",
"SEXTB",
"SEXTH",
"SFENCEVMA",
"SH",
"SH1ADD",
"SH1ADDUW",
"SH2ADD",
"SH2ADDUW",
"SH3ADD",
"SH3ADDUW",
"SLL",
"SLLI",
"SLLIUW",
"SLLIW",
"SLLW",
"SLT",
"SLTI",
"SLTIU",
"SLTU",
"SNEZ",
"SRA",
"SRAI",
"SRAIW",
"SRAW",
"SRET",
"SRL",
"SRLI",
"SRLIW",
"SRLW",
"SUB",
"SUBW",
"SW",
"VAADDUVV",
"VAADDUVX",
"VAADDVV",
"VAADDVX",
"VADCVIM",
"VADCVVM",
"VADCVXM",
"VADDVI",
"VADDVV",
"VADDVX",
"VANDVI",
"VANDVV",
"VANDVX",
"VASUBUVV",
"VASUBUVX",
"VASUBVV",
"VASUBVX",
"VCOMPRESSVM",
"VCPOPM",
"VDIVUVV",
"VDIVUVX",
"VDIVVV",
"VDIVVX",
"VFABSV",
"VFADDVF",
"VFADDVV",
"VFCLASSV",
"VFCVTFXUV",
"VFCVTFXV",
"VFCVTRTZXFV",
"VFCVTRTZXUFV",
"VFCVTXFV",
"VFCVTXUFV",
"VFDIVVF",
"VFDIVVV",
"VFIRSTM",
"VFMACCVF",
"VFMACCVV",
"VFMADDVF",
"VFMADDVV",
"VFMAXVF",
"VFMAXVV",
"VFMERGEVFM",
"VFMINVF",
"VFMINVV",
"VFMSACVF",
"VFMSACVV",
"VFMSUBVF",
"VFMSUBVV",
"VFMULVF",
"VFMULVV",
"VFMVFS",
"VFMVSF",
"VFMVVF",
"VFNCVTFFW",
"VFNCVTFXUW",
"VFNCVTFXW",
"VFNCVTRODFFW",
"VFNCVTRTZXFW",
"VFNCVTRTZXUFW",
"VFNCVTXFW",
"VFNCVTXUFW",
"VFNEGV",
"VFNMACCVF",
"VFNMACCVV",
"VFNMADDVF",
"VFNMADDVV",
"VFNMSACVF",
"VFNMSACVV",
"VFNMSUBVF",
"VFNMSUBVV",
"VFRDIVVF",
"VFREC7V",
"VFREDMAXVS",
"VFREDMINVS",
"VFREDOSUMVS",
"VFREDUSUMVS",
"VFRSQRT7V",
"VFRSUBVF",
"VFSGNJNVF",
"VFSGNJNVV",
"VFSGNJVF",
"VFSGNJVV",
"VFSGNJXVF",
"VFSGNJXVV",
"VFSLIDE1DOWNVF",
"VFSLIDE1UPVF",
"VFSQRTV",
"VFSUBVF",
"VFSUBVV",
"VFWADDVF",
"VFWADDVV",
"VFWADDWF",
"VFWADDWV",
"VFWCVTFFV",
"VFWCVTFXUV",
"VFWCVTFXV",
"VFWCVTRTZXFV",
"VFWCVTRTZXUFV",
"VFWCVTXFV",
"VFWCVTXUFV",
"VFWMACCVF",
"VFWMACCVV",
"VFWMSACVF",
"VFWMSACVV",
"VFWMULVF",
"VFWMULVV",
"VFWNMACCVF",
"VFWNMACCVV",
"VFWNMSACVF",
"VFWNMSACVV",
"VFWREDOSUMVS",
"VFWREDUSUMVS",
"VFWSUBVF",
"VFWSUBVV",
"VFWSUBWF",
"VFWSUBWV",
"VIDV",
"VIOTAM",
"VL1RE16V",
"VL1RE32V",
"VL1RE64V",
"VL1RE8V",
"VL1RV",
"VL2RE16V",
"VL2RE32V",
"VL2RE64V",
"VL2RE8V",
"VL2RV",
"VL4RE16V",
"VL4RE32V",
"VL4RE64V",
"VL4RE8V",
"VL4RV",
"VL8RE16V",
"VL8RE32V",
"VL8RE64V",
"VL8RE8V",
"VL8RV",
"VLE16FFV",
"VLE16V",
"VLE32FFV",
"VLE32V",
"VLE64FFV",
"VLE64V",
"VLE8FFV",
"VLE8V",
"VLMV",
"VLOXEI16V",
"VLOXEI32V",
"VLOXEI64V",
"VLOXEI8V",
"VLOXSEG2EI16V",
"VLOXSEG2EI32V",
"VLOXSEG2EI64V",
"VLOXSEG2EI8V",
"VLOXSEG3EI16V",
"VLOXSEG3EI32V",
"VLOXSEG3EI64V",
"VLOXSEG3EI8V",
"VLOXSEG4EI16V",
"VLOXSEG4EI32V",
"VLOXSEG4EI64V",
"VLOXSEG4EI8V",
"VLOXSEG5EI16V",
"VLOXSEG5EI32V",
"VLOXSEG5EI64V",
"VLOXSEG5EI8V",
"VLOXSEG6EI16V",
"VLOXSEG6EI32V",
"VLOXSEG6EI64V",
"VLOXSEG6EI8V",
"VLOXSEG7EI16V",
"VLOXSEG7EI32V",
"VLOXSEG7EI64V",
"VLOXSEG7EI8V",
"VLOXSEG8EI16V",
"VLOXSEG8EI32V",
"VLOXSEG8EI64V",
"VLOXSEG8EI8V",
"VLSE16V",
"VLSE32V",
"VLSE64V",
"VLSE8V",
"VLSEG2E16FFV",
"VLSEG2E16V",
"VLSEG2E32FFV",
"VLSEG2E32V",
"VLSEG2E64FFV",
"VLSEG2E64V",
"VLSEG2E8FFV",
"VLSEG2E8V",
"VLSEG3E16FFV",
"VLSEG3E16V",
"VLSEG3E32FFV",
"VLSEG3E32V",
"VLSEG3E64FFV",
"VLSEG3E64V",
"VLSEG3E8FFV",
"VLSEG3E8V",
"VLSEG4E16FFV",
"VLSEG4E16V",
"VLSEG4E32FFV",
"VLSEG4E32V",
"VLSEG4E64FFV",
"VLSEG4E64V",
"VLSEG4E8FFV",
"VLSEG4E8V",
"VLSEG5E16FFV",
"VLSEG5E16V",
"VLSEG5E32FFV",
"VLSEG5E32V",
"VLSEG5E64FFV",
"VLSEG5E64V",
"VLSEG5E8FFV",
"VLSEG5E8V",
"VLSEG6E16FFV",
"VLSEG6E16V",
"VLSEG6E32FFV",
"VLSEG6E32V",
"VLSEG6E64FFV",
"VLSEG6E64V",
"VLSEG6E8FFV",
"VLSEG6E8V",
"VLSEG7E16FFV",
"VLSEG7E16V",
"VLSEG7E32FFV",
"VLSEG7E32V",
"VLSEG7E64FFV",
"VLSEG7E64V",
"VLSEG7E8FFV",
"VLSEG7E8V",
"VLSEG8E16FFV",
"VLSEG8E16V",
"VLSEG8E32FFV",
"VLSEG8E32V",
"VLSEG8E64FFV",
"VLSEG8E64V",
"VLSEG8E8FFV",
"VLSEG8E8V",
"VLSSEG2E16V",
"VLSSEG2E32V",
"VLSSEG2E64V",
"VLSSEG2E8V",
"VLSSEG3E16V",
"VLSSEG3E32V",
"VLSSEG3E64V",
"VLSSEG3E8V",
"VLSSEG4E16V",
"VLSSEG4E32V",
"VLSSEG4E64V",
"VLSSEG4E8V",
"VLSSEG5E16V",
"VLSSEG5E32V",
"VLSSEG5E64V",
"VLSSEG5E8V",
"VLSSEG6E16V",
"VLSSEG6E32V",
"VLSSEG6E64V",
"VLSSEG6E8V",
"VLSSEG7E16V",
"VLSSEG7E32V",
"VLSSEG7E64V",
"VLSSEG7E8V",
"VLSSEG8E16V",
"VLSSEG8E32V",
"VLSSEG8E64V",
"VLSSEG8E8V",
"VLUXEI16V",
"VLUXEI32V",
"VLUXEI64V",
"VLUXEI8V",
"VLUXSEG2EI16V",
"VLUXSEG2EI32V",
"VLUXSEG2EI64V",
"VLUXSEG2EI8V",
"VLUXSEG3EI16V",
"VLUXSEG3EI32V",
"VLUXSEG3EI64V",
"VLUXSEG3EI8V",
"VLUXSEG4EI16V",
"VLUXSEG4EI32V",
"VLUXSEG4EI64V",
"VLUXSEG4EI8V",
"VLUXSEG5EI16V",
"VLUXSEG5EI32V",
"VLUXSEG5EI64V",
"VLUXSEG5EI8V",
"VLUXSEG6EI16V",
"VLUXSEG6EI32V",
"VLUXSEG6EI64V",
"VLUXSEG6EI8V",
"VLUXSEG7EI16V",
"VLUXSEG7EI32V",
"VLUXSEG7EI64V",
"VLUXSEG7EI8V",
"VLUXSEG8EI16V",
"VLUXSEG8EI32V",
"VLUXSEG8EI64V",
"VLUXSEG8EI8V",
"VMACCVV",
"VMACCVX",
"VMADCVI",
"VMADCVIM",
"VMADCVV",
"VMADCVVM",
"VMADCVX",
"VMADCVXM",
"VMADDVV",
"VMADDVX",
"VMANDMM",
"VMANDNMM",
"VMAXUVV",
"VMAXUVX",
"VMAXVV",
"VMAXVX",
"VMCLRM",
"VMERGEVIM",
"VMERGEVVM",
"VMERGEVXM",
"VMFEQVF",
"VMFEQVV",
"VMFGEVF",
"VMFGEVV",
"VMFGTVF",
"VMFGTVV",
"VMFLEVF",
"VMFLEVV",
"VMFLTVF",
"VMFLTVV",
"VMFNEVF",
"VMFNEVV",
"VMINUVV",
"VMINUVX",
"VMINVV",
"VMINVX",
"VMMVM",
"VMNANDMM",
"VMNORMM",
"VMNOTM",
"VMORMM",
"VMORNMM",
"VMSBCVV",
"VMSBCVVM",
"VMSBCVX",
"VMSBCVXM",
"VMSBFM",
"VMSEQVI",
"VMSEQVV",
"VMSEQVX",
"VMSETM",
"VMSGEUVI",
"VMSGEUVV",
"VMSGEVI",
"VMSGEVV",
"VMSGTUVI",
"VMSGTUVV",
"VMSGTUVX",
"VMSGTVI",
"VMSGTVV",
"VMSGTVX",
"VMSIFM",
"VMSLEUVI",
"VMSLEUVV",
"VMSLEUVX",
"VMSLEVI",
"VMSLEVV",
"VMSLEVX",
"VMSLTUVI",
"VMSLTUVV",
"VMSLTUVX",
"VMSLTVI",
"VMSLTVV",
"VMSLTVX",
"VMSNEVI",
"VMSNEVV",
"VMSNEVX",
"VMSOFM",
"VMULHSUVV",
"VMULHSUVX",
"VMULHUVV",
"VMULHUVX",
"VMULHVV",
"VMULHVX",
"VMULVV",
"VMULVX",
"VMV1RV",
"VMV2RV",
"VMV4RV",
"VMV8RV",
"VMVSX",
"VMVVI",
"VMVVV",
"VMVVX",
"VMVXS",
"VMXNORMM",
"VMXORMM",
"VNCLIPUWI",
"VNCLIPUWV",
"VNCLIPUWX",
"VNCLIPWI",
"VNCLIPWV",
"VNCLIPWX",
"VNCVTXXW",
"VNEGV",
"VNMSACVV",
"VNMSACVX",
"VNMSUBVV",
"VNMSUBVX",
"VNOTV",
"VNSRAWI",
"VNSRAWV",
"VNSRAWX",
"VNSRLWI",
"VNSRLWV",
"VNSRLWX",
"VORVI",
"VORVV",
"VORVX",
"VREDANDVS",
"VREDMAXUVS",
"VREDMAXVS",
"VREDMINUVS",
"VREDMINVS",
"VREDORVS",
"VREDSUMVS",
"VREDXORVS",
"VREMUVV",
"VREMUVX",
"VREMVV",
"VREMVX",
"VRGATHEREI16VV",
"VRGATHERVI",
"VRGATHERVV",
"VRGATHERVX",
"VRSUBVI",
"VRSUBVX",
"VS1RV",
"VS2RV",
"VS4RV",
"VS8RV",
"VSADDUVI",
"VSADDUVV",
"VSADDUVX",
"VSADDVI",
"VSADDVV",
"VSADDVX",
"VSBCVVM",
"VSBCVXM",
"VSE16V",
"VSE32V",
"VSE64V",
"VSE8V",
"VSETIVLI",
"VSETVL",
"VSETVLI",
"VSEXTVF2",
"VSEXTVF4",
"VSEXTVF8",
"VSLIDE1DOWNVX",
"VSLIDE1UPVX",
"VSLIDEDOWNVI",
"VSLIDEDOWNVX",
"VSLIDEUPVI",
"VSLIDEUPVX",
"VSLLVI",
"VSLLVV",
"VSLLVX",
"VSMULVV",
"VSMULVX",
"VSMV",
"VSOXEI16V",
"VSOXEI32V",
"VSOXEI64V",
"VSOXEI8V",
"VSOXSEG2EI16V",
"VSOXSEG2EI32V",
"VSOXSEG2EI64V",
"VSOXSEG2EI8V",
"VSOXSEG3EI16V",
"VSOXSEG3EI32V",
"VSOXSEG3EI64V",
"VSOXSEG3EI8V",
"VSOXSEG4EI16V",
"VSOXSEG4EI32V",
"VSOXSEG4EI64V",
"VSOXSEG4EI8V",
"VSOXSEG5EI16V",
"VSOXSEG5EI32V",
"VSOXSEG5EI64V",
"VSOXSEG5EI8V",
"VSOXSEG6EI16V",
"VSOXSEG6EI32V",
"VSOXSEG6EI64V",
"VSOXSEG6EI8V",
"VSOXSEG7EI16V",
"VSOXSEG7EI32V",
"VSOXSEG7EI64V",
"VSOXSEG7EI8V",
"VSOXSEG8EI16V",
"VSOXSEG8EI32V",
"VSOXSEG8EI64V",
"VSOXSEG8EI8V",
"VSRAVI",
"VSRAVV",
"VSRAVX",
"VSRLVI",
"VSRLVV",
"VSRLVX",
"VSSE16V",
"VSSE32V",
"VSSE64V",
"VSSE8V",
"VSSEG2E16V",
"VSSEG2E32V",
"VSSEG2E64V",
"VSSEG2E8V",
"VSSEG3E16V",
"VSSEG3E32V",
"VSSEG3E64V",
"VSSEG3E8V",
"VSSEG4E16V",
"VSSEG4E32V",
"VSSEG4E64V",
"VSSEG4E8V",
"VSSEG5E16V",
"VSSEG5E32V",
"VSSEG5E64V",
"VSSEG5E8V",
"VSSEG6E16V",
"VSSEG6E32V",
"VSSEG6E64V",
"VSSEG6E8V",
"VSSEG7E16V",
"VSSEG7E32V",
"VSSEG7E64V",
"VSSEG7E8V",
"VSSEG8E16V",
"VSSEG8E32V",
"VSSEG8E64V",
"VSSEG8E8V",
"VSSRAVI",
"VSSRAVV",
"VSSRAVX",
"VSSRLVI",
"VSSRLVV",
"VSSRLVX",
"VSSSEG2E16V",
"VSSSEG2E32V",
"VSSSEG2E64V",
"VSSSEG2E8V",
"VSSSEG3E16V",
"VSSSEG3E32V",
"VSSSEG3E64V",
"VSSSEG3E8V",
"VSSSEG4E16V",
"VSSSEG4E32V",
"VSSSEG4E64V",
"VSSSEG4E8V",
"VSSSEG5E16V",
"VSSSEG5E32V",
"VSSSEG5E64V",
"VSSSEG5E8V",
"VSSSEG6E16V",
"VSSSEG6E32V",
"VSSSEG6E64V",
"VSSSEG6E8V",
"VSSSEG7E16V",
"VSSSEG7E32V",
"VSSSEG7E64V",
"VSSSEG7E8V",
"VSSSEG8E16V",
"VSSSEG8E32V",
"VSSSEG8E64V",
"VSSSEG8E8V",
"VSSUBUVV",
"VSSUBUVX",
"VSSUBVV",
"VSSUBVX",
"VSUBVV",
"VSUBVX",
"VSUXEI16V",
"VSUXEI32V",
"VSUXEI64V",
"VSUXEI8V",
"VSUXSEG2EI16V",
"VSUXSEG2EI32V",
"VSUXSEG2EI64V",
"VSUXSEG2EI8V",
"VSUXSEG3EI16V",
"VSUXSEG3EI32V",
"VSUXSEG3EI64V",
"VSUXSEG3EI8V",
"VSUXSEG4EI16V",
"VSUXSEG4EI32V",
"VSUXSEG4EI64V",
"VSUXSEG4EI8V",
"VSUXSEG5EI16V",
"VSUXSEG5EI32V",
"VSUXSEG5EI64V",
"VSUXSEG5EI8V",
"VSUXSEG6EI16V",
"VSUXSEG6EI32V",
"VSUXSEG6EI64V",
"VSUXSEG6EI8V",
"VSUXSEG7EI16V",
"VSUXSEG7EI32V",
"VSUXSEG7EI64V",
"VSUXSEG7EI8V",
"VSUXSEG8EI16V",
"VSUXSEG8EI32V",
"VSUXSEG8EI64V",
"VSUXSEG8EI8V",
"VWADDUVV",
"VWADDUVX",
"VWADDUWV",
"VWADDUWX",
"VWADDVV",
"VWADDVX",
"VWADDWV",
"VWADDWX",
"VWCVTUXXV",
"VWCVTXXV",
"VWMACCSUVV",
"VWMACCSUVX",
"VWMACCUSVX",
"VWMACCUVV",
"VWMACCUVX",
"VWMACCVV",
"VWMACCVX",
"VWMULSUVV",
"VWMULSUVX",
"VWMULUVV",
"VWMULUVX",
"VWMULVV",
"VWMULVX",
"VWREDSUMUVS",
"VWREDSUMVS",
"VWSUBUVV",
"VWSUBUVX",
"VWSUBUWV",
"VWSUBUWX",
"VWSUBVV",
"VWSUBVX",
"VWSUBWV",
"VWSUBWX",
"VXORVI",
"VXORVV",
"VXORVX",
"VZEXTVF2",
"VZEXTVF4",
"VZEXTVF8",
"WFI",
"WORD",
"XNOR",
"XOR",
"XORI",
"ZEXTH",
}
+292
View File
@@ -0,0 +1,292 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// Assemble encodes the body of a TEXT function into x86-64 machine code,
// resolving local labels to relative jump offsets and translating the FP/SP
// pseudo-registers onto the hardware stack pointer (matching the Go
// assembler's default frame-pointer behaviour). Jumps always use the 32-bit
// relative form so instruction sizes are fixed and offsets resolve in a single
// layout pass.
//
// Supported operands: registers, memory (real base register), immediates,
// FP/SP frame-relative operands, and local-label jumps. SB (global symbol)
// operands require relocations and are not yet supported; SIMD (VEX/EVEX)
// instructions are pending.
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
fi := computeFrame(t)
// Pass 1: lay out instructions (including prologue/epilogue) to fix label
// offsets.
offsets := map[string]int{}
sizes := make([]int, len(t.Body))
pos := len(fi.prologue)
for i, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
sz, err := instrSize(s, fi)
if err != nil {
return nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
sizes[i] = sz
pos += sz
}
}
// Pass 2: emit.
out := append([]byte(nil), fi.prologue...)
pos = len(fi.prologue)
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
continue
}
code, err := encodeInstr(s, pos, offsets, fi)
if err != nil {
return nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
if len(code) != sizes[i] {
return nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
}
out = append(out, code...)
pos += len(code)
}
return out, offsets, nil
}
// frameInfo carries the frame layout derived from the TEXT directive.
type frameInfo struct {
size int // local frame size ($framesize)
useFP bool // a frame pointer (BP) is set up
fpAdjust int64 // added to x+N(FP) to reach the hardware SP-relative offset
spAdjust int64 // x-N(SP) becomes (spAdjust - N)(SP)
prologue []byte
epilogue []byte
}
// computeFrame derives the frame layout, matching the Go assembler's default
// (a frame pointer is used whenever the function has a non-zero frame).
func computeFrame(t *ast.Text) frameInfo {
fi := frameInfo{}
if t.Frame != nil && t.Frame.Imm.HasVal {
fi.size = int(t.Frame.Imm.Val)
}
if fi.size > 0 {
fi.useFP = true
fi.fpAdjust = int64(fi.size) + 16 // frame + saved BP + return address
fi.spAdjust = int64(fi.size)
fi.prologue = prologueBytes(fi.size)
fi.epilogue = epilogueBytes(fi.size)
} else {
fi.fpAdjust = 8 // return address only
}
return fi
}
// prologueBytes emits: PUSHQ BP; MOVQ SP, BP; SUBQ $size, SP.
func prologueBytes(size int) []byte {
out := []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
return append(out, subSP(size)...)
}
// epilogueBytes emits: ADDQ $size, SP; POPQ BP.
func epilogueBytes(size int) []byte {
out := addSP(size)
return append(out, 0x5D) // POPQ BP
}
func subSP(size int) []byte { // SUBQ $size, SP
if size >= -128 && size <= 127 {
return []byte{0x48, 0x83, 0xEC, byte(int8(size))}
}
return append([]byte{0x48, 0x81, 0xEC}, le32(int64(size))...)
}
func addSP(size int) []byte { // ADDQ $size, SP
if size >= -128 && size <= 127 {
return []byte{0x48, 0x83, 0xC4, byte(int8(size))}
}
return append([]byte{0x48, 0x81, 0xC4}, le32(int64(size))...)
}
// instrSize returns the encoded length of an instruction (pass 1). encodeInstr
// already includes the epilogue for a RET in a frame-pointer function; jumps use
// a fixed rel32 size (no epilogue).
func instrSize(s *ast.Instr, fi frameInfo) (int, error) {
mnem := strings.ToUpper(s.Mnemonic.Text)
if isJumpMnemonic(mnem) {
return jumpSize(mnem), nil
}
code, err := encodeInstr(s, 0, nil, fi)
if err != nil {
return 0, err
}
return len(code), nil
}
func isJumpMnemonic(mnem string) bool {
if mnem == "JMP" || mnem == "CALL" {
return true
}
_, ok := condCode(mnem)
return ok
}
// jumpSize returns the fixed length of a rel32 jump instruction.
func jumpSize(mnem string) int {
if mnem == "JMP" || mnem == "CALL" {
return 5 // opcode + rel32
}
return 6 // 0x0F 0x8x + rel32
}
// encodeInstr encodes one instruction, resolving jump targets against offsets
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
// function is prefixed with the epilogue.
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo) ([]byte, error) {
mnem := strings.ToUpper(s.Mnemonic.Text)
var prefix []byte
if mnem == "RET" && fi.useFP {
prefix = fi.epilogue
}
var code []byte
var err error
if isJumpMnemonic(mnem) {
code, err = encodeJump(s, mnem, pc+len(prefix), offsets)
} else {
code, err = encodeNormal(s, fi)
}
if err != nil {
return nil, err
}
return append(prefix, code...), nil
}
func encodeNormal(s *ast.Instr, fi frameInfo) ([]byte, error) {
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
if size == 0 {
size = 8
}
ops := make([]Operand, len(s.Operands))
for i, op := range s.Operands {
o, err := operandFromAST(op, size, fi)
if err != nil {
return nil, err
}
ops[i] = o
}
return Encode(s.Mnemonic.Text, ops...)
}
// encodeJump encodes a JMP/CALL/Jcc with a rel32 offset resolved from the
// target label.
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int) ([]byte, error) {
if len(s.Operands) != 1 {
return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands))
}
name, ok := labelName(s.Operands[0])
if !ok {
return nil, fmt.Errorf("jump target must be a local label")
}
target, ok := offsets[name]
if !ok {
return nil, fmt.Errorf("undefined label %q", name)
}
rel := int64(target - (pc + jumpSize(mnem)))
switch mnem {
case "JMP":
return append([]byte{0xE9}, le32(rel)...), nil
case "CALL":
return append([]byte{0xE8}, le32(rel)...), nil
default:
cc, _ := condCode(mnem)
return append([]byte{0x0F, 0x80 + byte(cc)}, le32(rel)...), nil
}
}
// labelName extracts a local-label name from a jump operand.
func labelName(op *ast.Operand) (string, bool) {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
return "", false
}
// spReg is the hardware stack pointer used to realise FP/SP pseudo-operands.
var spReg = Reg{idx: 4, size: 8}
// operandFromAST converts a parsed operand into an encoder Operand, applying
// the frame translation to FP/SP pseudo-register operands.
func operandFromAST(op *ast.Operand, size int, fi frameInfo) (Operand, error) {
switch op.Kind {
case ast.OpImmediate:
if op.Imm.HasVal {
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return Imm(v), nil
}
return nil, fmt.Errorf("non-integer immediate not supported")
case ast.OpAddr:
a := op.Addr
// FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the
// symbol, not the address displacement.
if a.Sym != nil && a.Sym.Pseudo == "FP" {
off := a.Sym.Offset + fi.fpAdjust
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
}
// SP-relative local: x-N(SP) → (spAdjust + offset)(SP).
if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" {
off := fi.spAdjust + a.Sym.Offset
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
}
// SB (global symbol) needs a relocation — not yet supported.
if a.Sym != nil && a.Sym.Pseudo == "SB" {
return nil, fmt.Errorf("SB (global symbol) operands need relocation support (pending)")
}
// Memory with a real base register: (base), off(base), (base)(index*scale).
if a.Base != "" {
base, ok := ParseReg(a.Base)
if !ok {
return nil, fmt.Errorf("unknown base register %q", a.Base)
}
m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size}
if a.Index != "" {
idx, ok := ParseReg(a.Index)
if !ok {
return nil, fmt.Errorf("unknown index register %q", a.Index)
}
m.Index = idx
m.Scale = a.Scale
m.HasIndex = true
}
return m, nil
}
// Bare register.
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
if r, ok := ParseReg(a.Sym.Name); ok {
return r, nil
}
}
return nil, fmt.Errorf("operand form not yet supported")
}
return nil, fmt.Errorf("unsupported operand")
}
+201
View File
@@ -0,0 +1,201 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"strings"
"testing"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// firstText parses src and returns its first TEXT function.
func firstText(t *testing.T, src string) *ast.Text {
t.Helper()
f, errs := parser.Parse("f_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
for _, d := range f.Decls {
if txt, ok := d.(*ast.Text); ok {
return txt
}
}
t.Fatal("no TEXT function found")
return nil
}
// disasm decodes a machine-code blob into Intel-syntax instruction strings.
func disasm(t *testing.T, code []byte) []string {
t.Helper()
var out []string
for len(code) > 0 {
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Fatalf("decode %x: %v", code, err)
}
out = append(out, x86asm.IntelSyntax(inst, 0, nil))
code = code[inst.Len:]
}
return out
}
func hexBytes(b []byte) string {
var sb strings.Builder
for _, x := range b {
sb.WriteString(" ")
const hexdig = "0123456789abcdef"
sb.WriteByte(hexdig[x>>4])
sb.WriteByte(hexdig[x&0xf])
}
return strings.TrimSpace(sb.String())
}
func TestAssembleLoop(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
XORQ AX, AX
loop:
ADDQ $1, AX
CMPQ $10, AX
JLT loop
RET
`)
code, labels, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
if _, ok := labels["loop"]; !ok {
t.Fatalf("label 'loop' not recorded: %v", labels)
}
got := strings.Join(disasm(t, code), "\n")
want := strings.Join([]string{
"xor rax, rax",
"add rax, 0x1",
"cmp rax, 0xa",
"jl 0x0",
"ret",
}, "\n")
gotLines := strings.Split(got, "\n")
wantLines := strings.Split(want, "\n")
if len(gotLines) != len(wantLines) {
t.Fatalf("instruction count mismatch:\n got:\n%s\n want:\n%s", got, want)
}
for i := range wantLines {
if strings.HasPrefix(wantLines[i], "jl") {
if !strings.HasPrefix(gotLines[i], "jl") {
t.Errorf("line %d: got %q, want a jl", i, gotLines[i])
}
continue
}
if gotLines[i] != wantLines[i] {
t.Errorf("line %d: got %q, want %q", i, gotLines[i], wantLines[i])
}
}
}
func TestAssembleMemory(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·g(SB), NOSPLIT, $0
MOVQ (AX), BX
MOVQ 8(AX), CX
LEAQ (AX)(BX*4), DX
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
got := strings.Join(disasm(t, code), "\n")
want := strings.Join([]string{
"mov rbx, qword ptr [rax]",
"mov rcx, qword ptr [rax+0x8]",
"lea rdx, ptr [rax+4*rbx]",
"ret",
}, "\n")
if got != want {
t.Errorf("assemble memory:\n got:\n%s\n want:\n%s", got, want)
}
}
// TestAssembleFP verifies the FP pseudo-register translation for a NOSPLIT $0
// function against the exact bytes the Go assembler produces (verified via
// `go tool objdump`): x+N(FP) maps to (N+8)(SP).
func TestAssembleFP(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·loadarg(SB), NOSPLIT, $0-24
MOVQ p+0(FP), AX
MOVQ n+8(FP), CX
ADDQ CX, AX
MOVQ AX, ret+16(FP)
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// From `go tool objdump` of the Go-assembled function:
// MOVQ 0x8(SP), AX 488b442408
// MOVQ 0x10(SP), CX 488b4c2410
// ADDQ CX, AX 4801c8
// MOVQ AX, 0x18(SP) 4889442418
// RET c3
want := []byte{
0x48, 0x8b, 0x44, 0x24, 0x08,
0x48, 0x8b, 0x4c, 0x24, 0x10,
0x48, 0x01, 0xc8,
0x48, 0x89, 0x44, 0x24, 0x18,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
// the bytes the Go assembler produces.
func TestAssembleFrame(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·withframe(SB), NOSPLIT, $16-16
MOVQ a+0(FP), AX
MOVQ b+8(FP), CX
ADDQ CX, AX
MOVQ AX, ret+16(FP)
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// From `go tool objdump`:
// PUSHQ BP 55
// MOVQ SP, BP 4889e5
// SUBQ $0x10, SP 4883ec10
// MOVQ 0x20(SP), AX 488b442420 (0 + 16 + 16)
// MOVQ 0x28(SP), CX 488b4c2428 (8 + 16 + 16)
// ADDQ CX, AX 4801c8
// MOVQ AX, 0x30(SP) 4889442430 (16 + 16 + 16)
// ADDQ $0x10, SP 4883c410
// POPQ BP 5d
// RET c3
want := []byte{
0x55, 0x48, 0x89, 0xe5, 0x48, 0x83, 0xec, 0x10,
0x48, 0x8b, 0x44, 0x24, 0x20,
0x48, 0x8b, 0x4c, 0x24, 0x28,
0x48, 0x01, 0xc8,
0x48, 0x89, 0x44, 0x24, 0x30,
0x48, 0x83, 0xc4, 0x10, 0x5d, 0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("frame translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
+288
View File
@@ -0,0 +1,288 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"strings"
)
// Encode encodes one Plan 9 instruction (mnemonic plus operands, in source
// order) into x86-64 machine code.
func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
e := &enc{}
if err := e.encode(mnemonic, ops); err != nil {
return nil, err
}
return e.out, nil
}
type enc struct {
out []byte
}
func (e *enc) encode(mnem string, ops []Operand) error {
upper := strings.ToUpper(mnem)
// Fixed-name instructions (no size suffix).
switch {
case upper == "RET":
return e.encodeRet()
case upper == "NOP":
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
case upper == "CALL":
return e.encodeJmpRel(ops, []byte{0xE8})
case upper == "JMP":
return e.encodeJmpRel(ops, []byte{0xE9})
}
if cc, ok := condCode(upper); ok {
return e.encodeJcc(cc, ops)
}
// VEX (AVX/AVX2) instructions: the trailing B/W/L/Q/D is part of the
// mnemonic, not a size suffix, so dispatch before splitSize.
if isVex(upper) {
return e.encodeVex(upper, ops)
}
base, size := splitSize(upper)
if size == 0 {
size = 8 // default operand size in 64-bit mode (e.g. PUSHQ)
}
switch base {
case "MOV":
return e.encodeMov(ops, size)
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
return e.encodeALU(aluOp[base], ops, size)
case "TEST":
return e.encodeTest(ops, size)
case "LEA":
return e.encodeLea(ops, size)
case "INC", "DEC", "NEG", "NOT":
return e.encodeUnary(unaryOp[base], ops, size)
case "SHL", "SHR", "SAR":
return e.encodeShift(shiftOp[base], ops, size)
case "IMUL":
return e.encodeImul(ops, size)
case "PUSH":
return e.encodePushPop(ops, true)
case "POP":
return e.encodePushPop(ops, false)
}
return fmt.Errorf("unsupported instruction %q", mnem)
}
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
func splitSize(upper string) (base string, size int) {
if upper == "" {
return upper, 0
}
switch upper[len(upper)-1] {
case 'B':
return upper[:len(upper)-1], 1
case 'W':
return upper[:len(upper)-1], 2
case 'L':
return upper[:len(upper)-1], 4
case 'Q':
return upper[:len(upper)-1], 8
}
return upper, 0
}
// --- instruction components -------------------------------------------------
type instr struct {
opSize16 bool
rexW bool
rexR bool
rexX bool
rexB bool
rexForced bool // REX needed even with all bits zero (8-bit low registers)
opcode []byte
modrm int // -1 if absent
sib int // -1 if absent
disp []byte
imm []byte
}
func (e *enc) emit(i *instr) error {
if i.opSize16 {
e.out = append(e.out, 0x66)
}
rex := byte(0)
if i.rexW {
rex |= 0x08
}
if i.rexR {
rex |= 0x04
}
if i.rexX {
rex |= 0x02
}
if i.rexB {
rex |= 0x01
}
if rex != 0 || i.rexForced {
e.out = append(e.out, 0x40|rex)
}
e.out = append(e.out, i.opcode...)
if i.modrm >= 0 {
e.out = append(e.out, byte(i.modrm))
}
if i.sib >= 0 {
e.out = append(e.out, byte(i.sib))
}
e.out = append(e.out, i.disp...)
e.out = append(e.out, i.imm...)
return nil
}
// newInstr starts an instruction with a size-derived REX.W and 0x66 prefix.
func newInstr(opSize int, opcode []byte) *instr {
return &instr{
opSize16: opSize == 2,
rexW: opSize == 8,
opcode: opcode,
modrm: -1,
sib: -1,
}
}
// --- ModR/M, SIB, displacement ----------------------------------------------
// setRM fills in the ModR/M (and SIB, displacement, REX bits) for an
// instruction whose reg field holds a real register `reg` and whose r/m field
// holds `rm`.
func setRM(i *instr, reg Reg, rm Operand, opSize int) error {
return setRMReg(i, reg.idx&7, reg.idx >= 8, reg.needsREX(opSize), rm, opSize)
}
// setRMDigit fills in the ModR/M for an instruction whose reg field is an
// opcode /digit extension (0–7), which carries none of the register REX rules.
func setRMDigit(i *instr, digit int, rm Operand, opSize int) error {
return setRMReg(i, digit, false, false, rm, opSize)
}
func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize int) error {
i.rexR = rexR
if regForced {
i.rexForced = true
}
switch r := rm.(type) {
case Reg:
i.rexB = r.idx >= 8
if r.needsREX(opSize) {
i.rexForced = true
}
i.modrm = 0xC0 | regField<<3 | (r.idx & 7)
return nil
case Mem:
return setMem(i, regField, r)
default:
return fmt.Errorf("invalid r/m operand %T", rm)
}
}
func setMem(i *instr, regField int, m Mem) error {
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
if err != nil {
return err
}
i.modrm = modrm
i.sib = sib
i.disp = disp
i.rexX = xBit == 1
i.rexB = bBit == 1
return nil
}
// memComponents computes the ModR/M byte (with the given reg field), the SIB
// byte (-1 if none), the displacement bytes, and the high index/base bits, for
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
sib = -1
// RIP-relative: neither base nor index.
if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
}
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
var mod int
switch {
case !m.HasBase:
mod = 0
disp = le32(m.Disp)
case m.Base.idx&7 == 5 && m.Disp == 0:
mod = 1
disp = []byte{0}
case m.Disp == 0:
mod = 0
case fits8(m.Disp):
mod = 1
disp = []byte{byte(int8(m.Disp))}
default:
mod = 2
disp = le32(m.Disp)
}
if needSIB {
idxField := 4 // 100 = no index
if m.HasIndex {
idxField = m.Index.idx & 7
if m.Index.idx >= 8 {
xBit = 1
}
}
baseField := 5 // 101 = no base (with mod=00 → disp32)
if m.HasBase {
baseField = m.Base.idx & 7
if m.Base.idx >= 8 {
bBit = 1
}
}
return mod<<6 | regField<<3 | 0x04, scaleBits(m.Scale)<<6 | idxField<<3 | baseField, disp, xBit, bBit, nil
}
if m.Base.idx >= 8 {
bBit = 1
}
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 0, bBit, nil
}
func scaleBits(scale int) int {
switch scale {
case 2:
return 1
case 4:
return 2
case 8:
return 3
default:
return 0 // scale 1 (or unset)
}
}
func fits8(v int64) bool { return v >= -128 && v <= 127 }
func le32(v int64) []byte {
u := uint32(v)
return []byte{byte(u), byte(u >> 8), byte(u >> 16), byte(u >> 24)}
}
func le16(v int64) []byte {
u := uint16(v)
return []byte{byte(u), byte(u >> 8)}
}
func le64(v int64) []byte {
u := uint64(v)
b := make([]byte, 8)
for i := 0; i < 8; i++ {
b[i] = byte(u >> (8 * i))
}
return b
}
+132
View File
@@ -0,0 +1,132 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"golang.org/x/arch/x86/x86asm"
)
// decode encodes an instruction and decodes it back, returning the decoded
// instruction and its Intel-syntax rendering.
func decode(t *testing.T, mnemonic string, ops ...Operand) (x86asm.Inst, string) {
t.Helper()
code, err := Encode(mnemonic, ops...)
if err != nil {
t.Fatalf("Encode(%s): %v", mnemonic, err)
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Fatalf("Decode(%x) of %s: %v", code, mnemonic, err)
}
if inst.Len != len(code) {
t.Fatalf("Decode consumed %d of %d bytes for %s (%x)", inst.Len, len(code), mnemonic, code)
}
return inst, x86asm.IntelSyntax(inst, 0, nil)
}
// checkSyntax asserts an instruction encodes and decodes to the expected
// Intel-syntax string.
func checkSyntax(t *testing.T, want, mnemonic string, ops ...Operand) {
t.Helper()
_, got := decode(t, mnemonic, ops...)
if got != want {
t.Errorf("%s: got %q, want %q", mnemonic, got, want)
}
}
// checkOp asserts the decoded opcode (used for relative jumps, whose rendered
// target depends on the program counter).
func checkOp(t *testing.T, want x86asm.Op, mnemonic string, ops ...Operand) {
t.Helper()
inst, _ := decode(t, mnemonic, ops...)
if inst.Op != want {
t.Errorf("%s: got op %v, want %v", mnemonic, inst.Op, want)
}
}
func TestMov(t *testing.T) {
checkSyntax(t, "mov rbx, rax", "MOVQ", AX, BX)
checkSyntax(t, "mov ebx, eax", "MOVL", AX, BX)
checkSyntax(t, "mov bl, al", "MOVB", AL, BL)
checkSyntax(t, "mov rax, rbx", "MOVQ", BX, AX)
checkSyntax(t, "mov rbx, qword ptr [rax]", "MOVQ", Ptr(AX, 0, 8), BX)
checkSyntax(t, "mov qword ptr [rbx], rax", "MOVQ", AX, Ptr(BX, 0, 8))
checkSyntax(t, "mov rbx, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), BX)
checkSyntax(t, "mov rbx, qword ptr [rsi+4*rbx]", "MOVQ", Idx(SI, BX, 4, 0, 8), BX)
checkSyntax(t, "mov rax, 0x5", "MOVQ", Imm(5), AX)
checkSyntax(t, "mov r8, 0x5", "MOVQ", Imm(5), Reg{idx: 8, size: 8})
checkSyntax(t, "mov qword ptr [rax], 0x5", "MOVQ", Imm(5), Ptr(AX, 0, 8))
checkSyntax(t, "mov r12, r13", "MOVQ", Reg{idx: 13, size: 8}, Reg{idx: 12, size: 8})
}
func TestALU(t *testing.T) {
checkSyntax(t, "add rbx, rax", "ADDQ", AX, BX)
checkSyntax(t, "add rax, 0x1", "ADDQ", Imm(1), AX)
checkSyntax(t, "add rax, 0x12c", "ADDQ", Imm(300), AX)
checkSyntax(t, "sub rdx, rcx", "SUBQ", CX, DX)
checkSyntax(t, "and rbx, 0x7", "ANDQ", Imm(7), BX)
checkSyntax(t, "or rcx, rbx", "ORQ", BX, CX)
checkSyntax(t, "xor rax, rax", "XORQ", AX, AX)
checkSyntax(t, "cmp r10, rsi", "CMPQ", SI, Reg{idx: 10, size: 8})
checkSyntax(t, "add rbx, qword ptr [rax]", "ADDQ", Ptr(AX, 0, 8), BX)
checkSyntax(t, "add qword ptr [rax], rbx", "ADDQ", BX, Ptr(AX, 0, 8))
checkSyntax(t, "cmp rbx, -0x20", "CMPQ", Imm(-32), BX)
}
func TestLea(t *testing.T) {
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
checkSyntax(t, "lea rax, ptr [rbx+0x8]", "LEAQ", Ptr(BX, 0x8, 8), AX)
}
func TestTest(t *testing.T) {
checkSyntax(t, "test rax, rax", "TESTQ", AX, AX)
checkSyntax(t, "test rbx, 0x7", "TESTQ", Imm(7), BX)
}
func TestPushPop(t *testing.T) {
checkSyntax(t, "push rbx", "PUSHQ", BX)
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
}
func TestUnary(t *testing.T) {
checkSyntax(t, "inc rax", "INCQ", AX)
checkSyntax(t, "dec rbx", "DECQ", BX)
checkSyntax(t, "neg rcx", "NEGQ", CX)
checkSyntax(t, "not rdx", "NOTQ", DX)
}
func TestShift(t *testing.T) {
checkSyntax(t, "shl rdx, 0x2", "SHLQ", Imm(2), DX)
checkSyntax(t, "shl rdx, cl", "SHLQ", CL, DX)
checkSyntax(t, "shl rdx, 0x1", "SHLQ", Imm(1), DX)
checkSyntax(t, "sar rcx, 0x1f", "SARQ", Imm(31), CX)
}
func TestImul(t *testing.T) {
checkSyntax(t, "imul rdx, rcx", "IMULQ", CX, DX)
checkSyntax(t, "imul edx, edx, 0x3", "IMULL", Imm(3), DX, DX)
checkSyntax(t, "imul rdx, rcx, 0x100", "IMULQ", Imm(256), CX, DX)
}
func TestControl(t *testing.T) {
checkSyntax(t, "ret", "RET")
checkSyntax(t, "nop", "NOP")
checkOp(t, x86asm.JMP, "JMP", Imm(0))
checkOp(t, x86asm.CALL, "CALL", Imm(0))
checkOp(t, x86asm.JGE, "JGE", Imm(0))
checkOp(t, x86asm.JNE, "JNE", Imm(0))
checkOp(t, x86asm.JBE, "JLS", Imm(0))
}
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
// the encoder handles a realistic instruction sequence.
func TestGoFlacScalarTail(t *testing.T) {
// MOVQ swin_base+0(FP), SI — modelled as MOVQ disp(reg), reg.
checkSyntax(t, "mov rsi, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), SI)
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
checkSyntax(t, "and r10, -0x8", "ANDQ", Imm(-8), Reg{idx: 10, size: 8})
}
+488
View File
@@ -0,0 +1,488 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "fmt"
// aluOp maps an arithmetic/logic mnemonic to its base "r/m, r" opcode (for
// 16/32/64-bit; the 8-bit form is one less) and its /digit for the immediate
// forms (0x80/0x81/0x83).
var aluOp = map[string]struct {
rr byte
digit int
}{
"ADD": {0x01, 0},
"OR": {0x09, 1},
"AND": {0x21, 4},
"SUB": {0x29, 5},
"XOR": {0x31, 6},
"CMP": {0x39, 7},
}
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
// the 0xFE/0xFF group (the short 0x40–0x4F forms are REX prefixes in 64-bit
// mode); NEG/NOT use the 0xF6/0xF7 group.
var unaryOp = map[string]struct {
digit int
op byte
}{
"INC": {0, 0xFF},
"DEC": {1, 0xFF},
"NOT": {2, 0xF7},
"NEG": {3, 0xF7},
}
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0–0xD3 group.
var shiftOp = map[string]int{
"SHL": 4,
"SHR": 5,
"SAR": 7,
}
// --- MOV --------------------------------------------------------------------
func (e *enc) encodeMov(ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("MOV expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, dstIsReg := dst.(Reg)
switch src := src.(type) {
case Reg:
if dstIsReg {
// MOV r, r/m: 0x8A/0x8B, reg=dst, rm=src.
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
}
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst(mem).
i := newInstr(size, []byte{movRM(size)})
if err := setRM(i, src, dst, size); err != nil {
return err
}
return e.emit(i)
case Mem:
if !dstIsReg {
return fmt.Errorf("MOV: two memory operands")
}
// MOV r, r/m: reg=dst, rm=src(mem).
i := newInstr(size, []byte{movRR(size)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
case Imm:
if dstIsReg {
// MOV r, imm: 0xB0+reg (8-bit) / 0xB8+reg (16/32/64, imm64 for Q).
opBase := byte(0xB8)
if size == 1 {
opBase = 0xB0
}
i := newInstr(size, []byte{opBase + byte(dstReg.idx&7)})
i.rexB = dstReg.idx >= 8
if dstReg.needsREX(size) {
i.rexForced = true
}
i.imm = immediate(int64(src), size, true)
return e.emit(i)
}
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
op := byte(0xC7)
if size == 1 {
op = 0xC6
}
i := newInstr(size, []byte{op})
if err := setRMDigit(i, 0, dst, size); err != nil {
return err
}
i.imm = immediate(int64(src), size, false)
return e.emit(i)
}
return fmt.Errorf("MOV: invalid operands")
}
func movRR(size int) byte { // MOV r, r/m
if size == 1 {
return 0x8A
}
return 0x8B
}
func movRM(size int) byte { // MOV r/m, r
if size == 1 {
return 0x88
}
return 0x89
}
// --- ALU (ADD/OR/AND/SUB/XOR/CMP) -------------------------------------------
func (e *enc) encodeALU(op struct {
rr byte
digit int
}, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("ALU instruction expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
if imm, ok := src.(Imm); ok {
return e.encodeALUImm(op.digit, dst, int64(imm), size)
}
dstReg, dstIsReg := dst.(Reg)
srcReg, srcIsReg := src.(Reg)
switch {
case srcIsReg:
// OP r/m, r: reg=src, rm=dst (dst is a register or memory). This is the
// form the Go assembler prefers when the source is a register.
opc := op.rr
if size == 1 {
opc = op.rr - 1
}
i := newInstr(size, []byte{opc})
if err := setRM(i, srcReg, dst, size); err != nil {
return err
}
return e.emit(i)
case dstIsReg:
// OP r, r/m: reg=dst, rm=src(memory).
opc := op.rr + 2
if size == 1 {
opc = op.rr + 1
}
i := newInstr(size, []byte{opc})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("two memory operands")
}
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if size == 1 {
i := newInstr(1, []byte{0x80})
if err := setRMDigit(i, digit, dst, 1); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
if fits8(imm) {
// 0x83 /digit, sign-extended imm8.
i := newInstr(size, []byte{0x83})
if err := setRMDigit(i, digit, dst, size); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
// 0x81 /digit, imm16/imm32.
i := newInstr(size, []byte{0x81})
if err := setRMDigit(i, digit, dst, size); err != nil {
return err
}
i.imm = immediate(imm, size, false)
return e.emit(i)
}
// --- TEST -------------------------------------------------------------------
func (e *enc) encodeTest(ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("TEST expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
if imm, ok := src.(Imm); ok {
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0.
op := byte(0xF7)
if size == 1 {
op = 0xF6
}
i := newInstr(size, []byte{op})
if err := setRMDigit(i, 0, dst, size); err != nil {
return err
}
i.imm = immediate(int64(imm), size, false)
return e.emit(i)
}
srcReg, ok := src.(Reg)
if !ok {
return fmt.Errorf("TEST: source must be a register or immediate")
}
// TEST r/m, r: 0x84 (8-bit) / 0x85.
op := byte(0x85)
if size == 1 {
op = 0x84
}
i := newInstr(size, []byte{op})
if err := setRM(i, srcReg, dst, size); err != nil {
return err
}
return e.emit(i)
}
// --- LEA --------------------------------------------------------------------
func (e *enc) encodeLea(ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("LEA expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1] // LEAQ addr, reg
dstReg, ok := dst.(Reg)
if !ok {
return fmt.Errorf("LEA: destination must be a register")
}
mem, ok := src.(Mem)
if !ok {
return fmt.Errorf("LEA: source must be a memory operand")
}
i := newInstr(size, []byte{0x8D})
if err := setRM(i, dstReg, mem, size); err != nil {
return err
}
return e.emit(i)
}
// --- INC/DEC/NEG/NOT --------------------------------------------------------
func (e *enc) encodeUnary(op struct {
digit int
op byte
}, ops []Operand, size int) error {
if len(ops) != 1 {
return fmt.Errorf("unary instruction expects 1 operand, got %d", len(ops))
}
base := op.op
if size == 1 {
base-- // 0xFF→0xFE, 0xF7→0xF6
}
i := newInstr(size, []byte{base})
if err := setRMDigit(i, op.digit, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// --- SHL/SHR/SAR ------------------------------------------------------------
func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("shift expects 2 operands, got %d", len(ops))
}
count, dst := ops[0], ops[1]
// Count is $1, %CL, or an imm8.
if reg, ok := count.(Reg); ok && reg.idx == 1 && reg.size <= 1 {
// CL: 0xD2 (8-bit) / 0xD3.
op := byte(0xD3)
if size == 1 {
op = 0xD2
}
i := newInstr(size, []byte{op})
if err := setRMDigit(i, digit, dst, size); err != nil {
return err
}
return e.emit(i)
}
imm, ok := count.(Imm)
if !ok {
return fmt.Errorf("shift count must be $1, CL or an immediate")
}
if imm == 1 {
// 0xD0 (8-bit) / 0xD1.
op := byte(0xD1)
if size == 1 {
op = 0xD0
}
i := newInstr(size, []byte{op})
if err := setRMDigit(i, digit, dst, size); err != nil {
return err
}
return e.emit(i)
}
// 0xC0 (8-bit) / 0xC1, imm8.
op := byte(0xC1)
if size == 1 {
op = 0xC0
}
i := newInstr(size, []byte{op})
if err := setRMDigit(i, digit, dst, size); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
// --- IMUL -------------------------------------------------------------------
func (e *enc) encodeImul(ops []Operand, size int) error {
switch len(ops) {
case 2:
// IMUL r, r/m: 0x0F 0xAF.
dstReg, ok := ops[1].(Reg)
if !ok {
return fmt.Errorf("IMUL: destination must be a register")
}
i := newInstr(size, []byte{0x0F, 0xAF})
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
case 3:
// IMUL r, r/m, imm: 0x6B (imm8) / 0x69 (imm16/32).
dstReg, ok := ops[2].(Reg)
if !ok {
return fmt.Errorf("IMUL: destination must be a register")
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("IMUL: immediate operand expected first")
}
// Plan 9 order: IMUL $imm, src, dst.
if fits8(int64(imm)) {
i := newInstr(size, []byte{0x6B})
if err := setRM(i, dstReg, ops[1], size); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
i := newInstr(size, []byte{0x69})
if err := setRM(i, dstReg, ops[1], size); err != nil {
return err
}
i.imm = immediate(int64(imm), size, false)
return e.emit(i)
}
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
}
// --- PUSH / POP -------------------------------------------------------------
func (e *enc) encodePushPop(ops []Operand, push bool) error {
if len(ops) != 1 {
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
}
switch op := ops[0].(type) {
case Reg:
base := byte(0x50) // PUSH r; POP is 0x58
if !push {
base = 0x58
}
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
i := &instr{opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
i.rexB = op.idx >= 8
return e.emit(i)
case Mem:
opc := byte(0xFF) // PUSH r/m: /6
digit := 6
if !push {
opc = 0x8F // POP r/m: /0
digit = 0
}
i := &instr{opcode: []byte{opc}, modrm: -1, sib: -1}
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
return err
}
return e.emit(i)
case Imm:
if !push {
return fmt.Errorf("POP does not take an immediate")
}
if fits8(int64(op)) {
i := &instr{opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
return e.emit(i)
}
i := &instr{opSize16: false, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: le32(int64(op))}
return e.emit(i)
}
return fmt.Errorf("PUSH/POP: invalid operand")
}
// --- RET / JMP / CALL / Jcc -------------------------------------------------
func (e *enc) encodeRet() error {
return e.emit(&instr{opcode: []byte{0xC3}, modrm: -1, sib: -1})
}
// encodeJmpRel encodes JMP/CALL with a relative displacement (the operand is an
// Imm holding the already-computed rel32 offset).
func (e *enc) encodeJmpRel(ops []Operand, opcode []byte) error {
if len(ops) != 1 {
return fmt.Errorf("JMP/CALL expects 1 operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("JMP/CALL: relative offset must be an immediate (labels are resolved by the assembler)")
}
return e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, imm: le32(int64(imm))})
}
// condCode maps a Plan 9 conditional-jump mnemonic to its x86 condition code.
func condCode(upper string) (int, bool) {
if len(upper) < 2 || upper[0] != 'J' || upper == "JMP" {
return 0, false
}
cc, ok := jccMap[upper[1:]]
return cc, ok
}
var jccMap = map[string]int{
"O": 0x0, "NO": 0x1, "OS": 0x0, "OC": 0x1,
"B": 0x2, "C": 0x2, "NAE": 0x2, "CS": 0x2,
"NB": 0x3, "NC": 0x3, "AE": 0x3, "CC": 0x3,
"E": 0x4, "Z": 0x4, "EQ": 0x4,
"NE": 0x5, "NZ": 0x5,
"BE": 0x6, "NA": 0x6, "LS": 0x6,
"NBE": 0x7, "A": 0x7, "HI": 0x7,
"S": 0x8, "MI": 0x8,
"NS": 0x9, "PL": 0x9,
"P": 0xA, "PE": 0xA, "PS": 0xA,
"NP": 0xB, "PO": 0xB, "PC": 0xB,
"L": 0xC, "NGE": 0xC, "LT": 0xC,
"NL": 0xD, "GE": 0xD,
"LE": 0xE, "NG": 0xE,
"NLE": 0xF, "G": 0xF, "GT": 0xF,
}
func (e *enc) encodeJcc(cc int, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("conditional jump expects 1 operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("conditional jump: relative offset must be an immediate")
}
if fits8(int64(imm)) {
// Short form: 0x70+cc, rel8.
return e.emit(&instr{opcode: []byte{0x70 + byte(cc)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}})
}
// Near form: 0x0F 0x80+cc, rel32.
return e.emit(&instr{opcode: []byte{0x0F, 0x80 + byte(cc)}, modrm: -1, sib: -1, imm: le32(int64(imm))})
}
// immediate encodes an immediate of the given operand size. full64 selects the
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
// sign-extended immediate is used for 64-bit operands.
func immediate(v int64, size int, full64 bool) []byte {
switch size {
case 1:
return []byte{byte(int8(v))}
case 2:
return le16(v)
case 4:
return le32(v)
default: // 8
if full64 {
return le64(v)
}
return le32(v) // sign-extended imm32
}
}
+43
View File
@@ -0,0 +1,43 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// Operand is an instruction operand: a Reg, a Mem reference or an Imm value.
type Operand interface {
isOperand()
}
// Imm is an immediate value. Its encoded width is chosen by the instruction
// (sign-extended imm8 where possible, otherwise imm32, imm64 for MOV).
type Imm int64
func (Imm) isOperand() {}
// Mem is a memory operand of the form disp(base)(index*scale).
type Mem struct {
Base Reg
Index Reg
Scale int // 1, 2, 4 or 8; 0 means no index
Disp int64
Size int // operand width in bytes
HasBase bool
HasIndex bool
}
func (Mem) isOperand() {}
// Ptr builds a plain displaced memory operand (base)+disp of the given size.
func Ptr(base Reg, disp int64, size int) Mem {
return Mem{Base: base, Disp: disp, Size: size, HasBase: true}
}
// Idx builds an indexed memory operand disp(base)(index*scale).
func Idx(base, index Reg, scale int, disp int64, size int) Mem {
return Mem{Base: base, Index: index, Scale: scale, Disp: disp, Size: size, HasBase: true, HasIndex: true}
}
// Rip builds a RIP-relative memory operand (RIP)+disp.
func Rip(disp int64, size int) Mem {
return Mem{Disp: disp, Size: size}
}
+169
View File
@@ -0,0 +1,169 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package asm is a standalone assembler: it encodes Plan 9 assembly
// instructions into machine code without the Go toolchain. Phase 2 begins
// with an amd64 (x86-64) scalar instruction encoder; the encoding is validated
// by round-tripping through golang.org/x/arch's decoder in the tests.
package asm
import "strings"
// Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …)
// are size-agnostic — the instruction suffix (MOVQ vs MOVL) fixes the width —
// so the encoder keys off the register's index and lets the mnemonic supply the
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
}
// Index returns the register number (0–15).
func (r Reg) Index() int { return r.idx }
// Size returns the width in bytes implied by the register's name.
func (r Reg) Size() int { return r.size }
func (r Reg) isOperand() {}
// needsREX reports whether this register forces a REX prefix at the given
// operand size: the extended registers R8–R15 always do, and at byte size the
// low registers SPL/BPL/SIL/DIL (indices 4–7, not high) do as well.
func (r Reg) needsREX(opSize int) bool {
if r.idx >= 8 {
return true
}
return opSize == 1 && r.idx >= 4 && !r.high
}
// Register constants (the size is the width the name implies).
var (
AL = Reg{0, 1, false}
CL = Reg{1, 1, false}
DL = Reg{2, 1, false}
BL = Reg{3, 1, false}
AH = Reg{4, 1, true}
CH = Reg{5, 1, true}
DH = Reg{6, 1, true}
BH = Reg{7, 1, true}
SPL = Reg{4, 1, false}
BPL = Reg{5, 1, false}
SIL = Reg{6, 1, false}
DIL = Reg{7, 1, false}
AX = Reg{0, 2, false}
CX = Reg{1, 2, false}
DX = Reg{2, 2, false}
BX = Reg{3, 2, false}
SP = Reg{4, 2, false}
BP = Reg{5, 2, false}
SI = Reg{6, 2, false}
DI = Reg{7, 2, false}
EAX = Reg{0, 4, false}
ECX = Reg{1, 4, false}
EDX = Reg{2, 4, false}
EBX = Reg{3, 4, false}
ESP = Reg{4, 4, false}
EBP = Reg{5, 4, false}
ESI = Reg{6, 4, false}
EDI = Reg{7, 4, false}
RAX = Reg{0, 8, false}
RCX = Reg{1, 8, false}
RDX = Reg{2, 8, false}
RBX = Reg{3, 8, false}
RSP = Reg{4, 8, false}
RBP = Reg{5, 8, false}
RSI = Reg{6, 8, false}
RDI = Reg{7, 8, false}
)
// regByName maps an assembly register name (case-insensitive) to a Reg.
var regByName = buildRegByName()
func buildRegByName() map[string]Reg {
m := map[string]Reg{}
// 64-bit: RAX..RDI, R8..R15.
r64 := []string{"RAX", "RCX", "RDX", "RBX", "RSP", "RBP", "RSI", "RDI"}
for i, n := range r64 {
m[n] = Reg{i, 8, false}
}
for i := 8; i <= 15; i++ {
m["R"+itoa(i)] = Reg{i, 8, false}
}
// 32-bit: EAX..EDI, R8D..R15D.
e32 := []string{"EAX", "ECX", "EDX", "EBX", "ESP", "EBP", "ESI", "EDI"}
for i, n := range e32 {
m[n] = Reg{i, 4, false}
}
for i := 8; i <= 15; i++ {
m["R"+itoa(i)+"D"] = Reg{i, 4, false}
}
// 16-bit: AX..DI, R8W..R15W.
w16 := []string{"AX", "CX", "DX", "BX", "SP", "BP", "SI", "DI"}
for i, n := range w16 {
m[n] = Reg{i, 2, false}
}
for i := 8; i <= 15; i++ {
m["R"+itoa(i)+"W"] = Reg{i, 2, false}
}
// 8-bit: AL..BH, SPL..DIL, R8B..R15B.
for n, r := range map[string]Reg{
"AL": AL, "CL": CL, "DL": DL, "BL": BL,
"AH": AH, "CH": CH, "DH": DH, "BH": BH,
"SPL": SPL, "BPL": BPL, "SIL": SIL, "DIL": DIL,
} {
m[n] = r
}
for i := 8; i <= 15; i++ {
m["R"+itoa(i)+"B"] = Reg{i, 1, false}
}
// Vector: X0..X15 (128-bit, encoded size 16), Y0..Y15 (256-bit, size 32).
// Z (512-bit) and K (mask) registers arrive with EVEX/AVX-512 support.
for i := 0; i <= 15; i++ {
m["X"+itoa(i)] = Reg{i, 16, false}
m["Y"+itoa(i)] = Reg{i, 32, false}
}
return m
}
// isVec reports whether r is an XMM/YMM vector register.
func (r Reg) isVec() bool { return r.size == 16 || r.size == 32 }
// vecLenBit returns the VEX.L bit for a vector register (X=0/128-bit,
// Y=1/256-bit).
func (r Reg) vecLenBit() int {
if r.size == 32 {
return 1
}
return 0
}
// ParseReg resolves an assembly register name to a Reg.
func ParseReg(name string) (Reg, bool) {
r, ok := regByName[strings.ToUpper(name)]
return r, ok
}
func itoa(n int) string {
if n == 0 {
return "0"
}
var buf [3]byte
i := len(buf)
for n > 0 {
i--
buf[i] = byte('0' + n%10)
n /= 10
}
return string(buf[i:])
}
+234
View File
@@ -0,0 +1,234 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "fmt"
// This file implements VEX (AVX/AVX2) instruction encoding. EVEX (AVX-512)
// support is a later increment.
// vexForm selects how an instruction's operands map onto the VEX.vvvv,
// ModRM.reg and ModRM.rm fields.
type vexForm int
const (
// vexNDS3 is the three-operand form `OP src2, src1, dst` (Plan 9 order):
// ModRM.reg = dst (op2), VEX.vvvv = src1 (op1), ModRM.rm = src2 (op0).
vexNDS3 vexForm = iota
// vexRM is the two-operand form `OP src, dst` with no vvvv source:
// ModRM.reg = dst (op1), ModRM.rm = src (op0), VEX.vvvv = 1111 (unused).
vexRM
// vexShiftImm is the immediate-shift form `OP $imm, src, dst`: ModRM.reg =
// /digit, ModRM.rm = src (op1), VEX.vvvv = dst (op2), imm8 = op0.
vexShiftImm
)
// vexSpec describes one VEX instruction's encoding parameters.
type vexSpec struct {
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
opcode byte
w int // VEX.W (0 for WIG)
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
opdigit int // ModRM.reg /digit, or -1 when reg is a register
form vexForm
}
// vexTable maps an upper-case mnemonic to its VEX encoding. It covers the
// AVX2 instructions used by the go-flac kernels in the three-operand NDS form;
// it is extended incrementally.
var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG — integer arithmetic / logic / compare.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
"VPSUBQ": {1, 0xFB, 0, 1, -1, vexNDS3},
"VPXOR": {1, 0xEF, 0, 1, -1, vexNDS3},
"VPOR": {1, 0xEB, 0, 1, -1, vexNDS3},
"VPAND": {1, 0xDB, 0, 1, -1, vexNDS3},
"VPANDN": {1, 0xDF, 0, 1, -1, vexNDS3},
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3},
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3},
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG.
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
"VPMULDQ": {2, 0x28, 0, 1, -1, vexNDS3},
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG — sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
// VEX.128/256.66.0F.WIG — immediate shifts (opdigit selects the shift).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
}
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
func isVex(mnemUpper string) bool {
_, ok := vexTable[mnemUpper]
return ok
}
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
spec := vexTable[mnemUpper]
switch spec.form {
case vexNDS3:
return e.encodeVexNDS3(spec, ops)
case vexRM:
return e.encodeVexRM(spec, ops)
case vexShiftImm:
return e.encodeVexShiftImm(spec, ops)
}
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
}
// encodeVexNDS3 encodes the three-operand NDS form: OP src2, src1, dst.
func (e *enc) encodeVexNDS3(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("VEX destination must be a vector register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a vector register")
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
vvvvBar := 15 - (vvvvReg.idx & 15)
return e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2)
}
// encodeVexRM encodes the two-operand form: OP src, dst (no vvvv source).
// ModRM.reg = dst, ModRM.rm = src; the vector length comes from whichever
// operand is a vector register (the destination for extends/broadcasts, the
// source for the move-mask instructions whose destination is a GPR).
func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("VEX two-operand instruction expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok {
return fmt.Errorf("VEX destination must be a register")
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
// Vector length: from the destination if it is a vector, otherwise from the
// source (move-mask instructions have a GPR destination and a vector source).
l := 0
if dstReg.isVec() {
l = dstReg.vecLenBit()
} else if srcReg, ok := src.(Reg); ok && srcReg.isVec() {
l = srcReg.vecLenBit()
}
return e.emitVexFields(spec, l, regField, rBit, 0, src) // vvvv unused → vvvvBar=0
}
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
// shift kind in the ModRM.reg /digit.
func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift count must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("shift source must be a vector register")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("shift destination must be a vector register")
}
vvvvBar := 15 - (dstReg.idx & 15)
l := dstReg.vecLenBit()
rmField := srcReg.idx & 7
bBit := 0
if srcReg.idx >= 8 {
bBit = 1
}
modrm := 0xC0 | spec.opdigit<<3 | rmField
if spec.mapSel == 1 && bBit == 0 && spec.w == 0 {
e.out = append(e.out, 0xC5, byte(1<<7|vvvvBar<<3|l<<2|spec.pp))
} else {
e.out = append(e.out, 0xC4,
byte(1<<7|1<<6|(1-bBit)<<5|spec.mapSel),
byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp))
}
e.out = append(e.out, spec.opcode, byte(modrm), byte(int8(immVal)))
return nil
}
// emitVexFields emits the VEX prefix, opcode, ModR/M, SIB and displacement for
// the given precomputed fields. It is shared by the NDS and RM forms.
func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error {
var modrm, sib int
var disp []byte
var xBit, bBit int
switch r := rm.(type) {
case Reg:
modrm = 0xC0 | regField<<3 | (r.idx & 7)
sib = -1
if r.idx >= 8 {
bBit = 1
}
case Mem:
var err error
modrm, sib, disp, xBit, bBit, err = memComponents(regField, r)
if err != nil {
return err
}
default:
return fmt.Errorf("invalid VEX r/m operand")
}
if spec.mapSel == 1 && xBit == 0 && bBit == 0 && spec.w == 0 {
e.out = append(e.out, 0xC5, byte((1-rBit)<<7|vvvvBar<<3|l<<2|spec.pp))
} else {
e.out = append(e.out, 0xC4,
byte((1-rBit)<<7|(1-xBit)<<6|(1-bBit)<<5|spec.mapSel),
byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp))
}
e.out = append(e.out, spec.opcode, byte(modrm))
if sib >= 0 {
e.out = append(e.out, byte(sib))
}
e.out = append(e.out, disp...)
return nil
}
+142
View File
@@ -0,0 +1,142 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"golang.org/x/arch/x86/x86asm"
)
func vreg(t *testing.T, name string) Reg {
t.Helper()
r, ok := ParseReg(name)
if !ok {
t.Fatalf("unknown register %s", name)
}
return r
}
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
// instruction and verifies it round-trips through the x86 decoder to the same
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
func TestVexNDS3(t *testing.T) {
for mnem, spec := range vexTable {
if spec.form != vexNDS3 {
continue
}
code, err := Encode(mnem, vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2"))
if err != nil {
t.Errorf("%s: Encode: %v", mnem, err)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(% x): %v", mnem, code, err)
continue
}
if inst.Op.String() != mnem {
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code)
}
}
}
// TestVexGoFlac checks a representative go-flac instruction sequence encodes
// and decodes as expected.
func TestVexGoFlac(t *testing.T) {
// VPADDD Y5, Y8, Y8 → vpaddd ymm8, ymm8, ymm5.
code, err := Encode("VPADDD", vreg(t, "Y5"), vreg(t, "Y8"), vreg(t, "Y8"))
if err != nil {
t.Fatalf("Encode: %v", err)
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Fatalf("Decode(% x): %v", code, err)
}
if inst.Op != x86asm.VPADDD {
t.Fatalf("decoded %s, want VPADDD", inst.Op)
}
}
// TestVexXMM checks the 128-bit (XMM) form selects VEX.L=0.
func TestVexXMM(t *testing.T) {
code, err := Encode("VPXOR", vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7"))
if err != nil {
t.Fatalf("Encode: %v", err)
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Fatalf("Decode(% x): %v", code, err)
}
if inst.Op != x86asm.VPXOR {
t.Fatalf("decoded %s, want VPXOR", inst.Op)
}
// vpxor xmm7, xmm7, xmm7 → C5 C9 EF FF (2-byte VEX, L=0).
if code[0] != 0xC5 {
t.Errorf("expected 2-byte VEX (C5), got % x", code)
}
}
// TestVexRM validates the two-operand (reg=dst, rm=src, no vvvv) forms by
// round-tripping through the decoder.
func TestVexRM(t *testing.T) {
cases := []struct {
mnem string
ops []Operand
want x86asm.Op
}{
{"VPMOVSXWD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y0")}, x86asm.VPMOVSXWD},
{"VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, x86asm.VPMOVSXDQ},
{"VPMOVZXDQ", []Operand{vreg(t, "X4"), vreg(t, "Y4")}, x86asm.VPMOVZXDQ},
{"VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, x86asm.VPBROADCASTD},
{"VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, x86asm.VPMOVMSKB},
{"VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, x86asm.VMOVMSKPS},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.mnem, err)
continue
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Errorf("%s: Decode(% x): %v", c.mnem, code, err)
continue
}
if inst.Op != c.want {
t.Errorf("%s: decoded as %s (% x)", c.mnem, inst.Op, code)
}
}
}
// TestVexShiftImm validates the immediate-shift form, checking the destination
// (VEX.vvvv) and source (ModRM.rm) land in the right places.
func TestVexShiftImm(t *testing.T) {
// VPSLLD $1, Y3, Y4 → vpslld ymm4, ymm3, 1.
code, err := Encode("VPSLLD", Imm(1), vreg(t, "Y3"), vreg(t, "Y4"))
if err != nil {
t.Fatalf("Encode: %v", err)
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
t.Fatalf("Decode(% x): %v", code, err)
}
if inst.Op != x86asm.VPSLLD {
t.Fatalf("decoded %s, want VPSLLD (% x)", inst.Op, code)
}
// Intel order: dst, src, imm → "vpslld ymm4, ymm3, 0x1".
if got := x86asm.IntelSyntax(inst, 0, nil); got != "vpslld ymm4, ymm3, 0x1" {
t.Errorf("VPSLLD syntax = %q, want \"vpslld ymm4, ymm3, 0x1\" (% x)", got, code)
}
// VPSRAD $31, Y3, Y3 → vpsrad ymm3, ymm3, 31.
code, err = Encode("VPSRAD", Imm(31), vreg(t, "Y3"), vreg(t, "Y3"))
if err != nil {
t.Fatalf("Encode VPSRAD: %v", err)
}
inst, err = x86asm.Decode(code, 64)
if err != nil || inst.Op != x86asm.VPSRAD {
t.Fatalf("VPSRAD decoded %v (err %v), want VPSRAD", inst.Op, err)
}
}
+162
View File
@@ -0,0 +1,162 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package ast defines the abstract syntax tree of a GAsm source file. The
// tree is produced by the parser and consumed by the formatter, linter and
// language server. Operand classification that depends on the target
// architecture (is this bare name a register or a label?) is deliberately left
// to the arch package; the AST records syntax only.
package ast
import "sourcedock.dev/petrbalvin/gasm-devkit/token"
// File is the parsed representation of one .s source file.
type File struct {
Path string
Decls []Decl
Orphans []Stmt // labels/instructions seen before any TEXT directive
// Macros holds the names introduced by #define directives in this file.
// The linter uses it to avoid flagging macro invocations as unknown
// instructions (macro expansion itself is out of scope — see the docs).
Macros map[string]bool
}
// Decl is a top-level declaration.
type Decl interface {
declNode()
// Pos returns the position of the declaration's first token.
Pos() token.Position
}
// Stmt is a statement inside a TEXT body.
type Stmt interface {
stmtNode()
Pos() token.Position
}
// Include is a #include "header" line.
type Include struct {
Hash token.Token
Name token.Token // the directive name, usually "include"
Header token.Token // the string literal, quotes included
}
func (*Include) declNode() {}
func (d *Include) Pos() token.Position { return d.Hash.Pos }
// Preproc is any other preprocessor line (#define, #undef, …) captured loosely.
type Preproc struct {
Hash token.Token
Raw string // verbatim text after the '#'
}
func (*Preproc) declNode() {}
func (d *Preproc) Pos() token.Position { return d.Hash.Pos }
// Text is a TEXT function definition and its body.
type Text struct {
Keyword token.Token // the TEXT token
Name *Symbol // ·funcName(SB)
Flags []string // NOSPLIT, DUPOK, …
Frame *Operand // $0
Args *Operand // the -65 part; nil when absent
Body []Stmt
Doc string // preceding comment block (typically the Go signature)
}
func (*Text) declNode() {}
func (d *Text) Pos() token.Position { return d.Keyword.Pos }
// Globl is a GLOBL symbol declaration.
type Globl struct {
Keyword token.Token
Name *Symbol
Flags []string
Size *Operand
}
func (*Globl) declNode() {}
func (d *Globl) Pos() token.Position { return d.Keyword.Pos }
// Data is a DATA symbol initialiser.
type Data struct {
Keyword token.Token
Name *Symbol // includes any +offset
Width int // the /width suffix; 0 when absent
Value *Operand
}
func (*Data) declNode() {}
func (d *Data) Pos() token.Position { return d.Keyword.Pos }
// Label is a local label definition such as vec1:.
type Label struct {
Name token.Token
Colon token.Token
}
func (*Label) stmtNode() {}
func (s *Label) Pos() token.Position { return s.Name.Pos }
// Instr is a single machine instruction with zero or more operands.
type Instr struct {
Mnemonic token.Token
Operands []*Operand
Comment string // trailing comment text, without the leading //
}
func (*Instr) stmtNode() {}
func (s *Instr) Pos() token.Position { return s.Mnemonic.Pos }
// Symbol is a symbol reference: ·name(SB), name<>(SB), name+8(FP), …
type Symbol struct {
Raw string // verbatim text
Pkg string // package prefix before the middle dot ("" = current package)
Name string // identifier without the middle dot or <>
Static bool // the <> marker is present
Pseudo string // FP, SP, SB or PC ("" for a bare name)
Offset int64
HasOff bool
Pos token.Position
}
// OpKind classifies an operand syntactically.
type OpKind int
// Operand kinds.
const (
OpInvalid OpKind = iota
OpImmediate // $value
OpAddr // register, memory reference, symbol or label
)
// Operand is one instruction operand.
type Operand struct {
Raw string
Kind OpKind
Imm Immediate
Addr Address
Pos token.Position
}
// Immediate is a $ value.
type Immediate struct {
Neg bool
Val int64
HasVal bool // a simple integer immediate was parsed
Float string // non-empty for a floating-point immediate
Str string // non-empty for a string/rune immediate
Sym *Symbol // non-nil for $sym(…)
}
// Address is a non-immediate operand: a register, a memory reference, a symbol
// reference or a label. Fields are populated best-effort from the syntax.
type Address struct {
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
Base string // base register, from (base)
Index string // index register, from (index*scale)
Scale int // index scale; 0 when absent
Offset int64 // leading displacement, from off(base)
HasOff bool // a leading displacement is present
Shift string // verbatim arm64 shift suffix, e.g. "<<2"
}
+65
View File
@@ -0,0 +1,65 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package ast
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} }
func TestDeclPositions(t *testing.T) {
hash := token.Token{Kind: token.Hash, Pos: pos(1, 1)}
inc := &Include{Hash: hash}
if inc.Pos() != hash.Pos {
t.Errorf("Include.Pos() = %v, want %v", inc.Pos(), hash.Pos)
}
pre := &Preproc{Hash: hash}
if pre.Pos() != hash.Pos {
t.Errorf("Preproc.Pos() = %v", pre.Pos())
}
kw := token.Token{Kind: token.Ident, Text: "TEXT", Pos: pos(5, 1)}
text := &Text{Keyword: kw}
if text.Pos() != kw.Pos {
t.Errorf("Text.Pos() = %v", text.Pos())
}
gkw := token.Token{Kind: token.Ident, Text: "GLOBL", Pos: pos(7, 1)}
if (&Globl{Keyword: gkw}).Pos() != gkw.Pos {
t.Error("Globl.Pos()")
}
dkw := token.Token{Kind: token.Ident, Text: "DATA", Pos: pos(8, 1)}
if (&Data{Keyword: dkw}).Pos() != dkw.Pos {
t.Error("Data.Pos()")
}
}
func TestStmtPositions(t *testing.T) {
name := token.Token{Kind: token.Ident, Text: "loop", Pos: pos(3, 1)}
if (&Label{Name: name}).Pos() != name.Pos {
t.Error("Label.Pos()")
}
mnem := token.Token{Kind: token.Ident, Text: "RET", Pos: pos(4, 2)}
if (&Instr{Mnemonic: mnem}).Pos() != mnem.Pos {
t.Error("Instr.Pos()")
}
}
// TestInterfaces confirms the node types satisfy their interfaces, so callers
// can range over Decls and Stmts.
func TestInterfaces(t *testing.T) {
var decls []Decl = []Decl{&Include{}, &Preproc{}, &Text{}, &Globl{}, &Data{}}
if len(decls) != 5 {
t.Fatal("decl interface set")
}
var stmts []Stmt = []Stmt{&Label{}, &Instr{}}
if len(stmts) != 2 {
t.Fatal("stmt interface set")
}
}
+280
View File
@@ -0,0 +1,280 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Command gasm is the developer frontend for GAsm — Go's Plan 9 assembler.
// It bundles a token dumper, a parser, a formatter, a linter and a language
// server into one binary. Every subcommand works headlessly so it can be
// driven from scripts and CI as well as from an editor.
package main
import (
"flag"
"fmt"
"io"
"os"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/format"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// version is the release version, stamped at build time via
// -ldflags "-X main.version=…" (defaulting to the current release).
var version = "0.1.0"
func main() {
if len(os.Args) < 2 {
usage(os.Stderr)
os.Exit(2)
}
switch os.Args[1] {
case "tokens":
os.Exit(cmdTokens(os.Args[2:]))
case "parse":
os.Exit(cmdParse(os.Args[2:]))
case "fmt":
os.Exit(cmdFmt(os.Args[2:]))
case "lint":
os.Exit(cmdLint(os.Args[2:]))
case "asm":
os.Exit(cmdAsm(os.Args[2:]))
case "lsp":
os.Exit(cmdLSP(os.Args[2:]))
case "version", "--version", "-V":
fmt.Printf("gasm %s\n", version)
case "help", "-h", "--help":
usage(os.Stdout)
default:
fmt.Fprintf(os.Stderr, "gasm: unknown command %q\n\n", os.Args[1])
usage(os.Stderr)
os.Exit(2)
}
}
func usage(w io.Writer) {
fmt.Fprintf(w, `gasm %s — developer tooling for Go's Plan 9 assembler
Usage:
gasm tokens <file> print the lexical token stream
gasm parse <file> parse and report syntax errors
gasm fmt [-w] <file...> canonicalise formatting (-w writes in place)
gasm lint <file...> run static checks
gasm asm [-o out.bin] <file> assemble to machine code (amd64, Phase 2)
gasm lsp run the language server over stdio
gasm version print the version
`, version)
}
// readSource returns the contents of path, or stdin when path is "-".
func readSource(path string) (string, error) {
if path == "-" {
b, err := io.ReadAll(os.Stdin)
return string(b), err
}
b, err := os.ReadFile(path)
return string(b), err
}
func cmdTokens(args []string) int {
fs := flag.NewFlagSet("tokens", flag.ExitOnError)
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm tokens <file>")
return 2
}
src, err := readSource(fs.Arg(0))
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
return 1
}
for _, tok := range lexer.Tokenize(src) {
fmt.Printf("%s\t%s\t%q\n", tok.Pos, tok.Kind, tok.Text)
}
return 0
}
func cmdParse(args []string) int {
fs := flag.NewFlagSet("parse", flag.ExitOnError)
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm parse <file>")
return 2
}
path := fs.Arg(0)
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
return 1
}
file, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
funcs := 0
for _, d := range file.Decls {
if _, ok := d.(*ast.Text); ok {
funcs++
}
}
fmt.Printf("%s: OK — %d declarations, %d functions\n", path, len(file.Decls), funcs)
return 0
}
func cmdFmt(args []string) int {
fs := flag.NewFlagSet("fmt", flag.ExitOnError)
write := fs.Bool("w", false, "write result to the source file")
fs.Parse(args)
if fs.NArg() == 0 {
fmt.Fprintln(os.Stderr, "usage: gasm fmt [-w] <file...>")
return 2
}
rc := 0
for _, path := range fs.Args() {
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
rc = 1
continue
}
out := format.Source(path, src)
if *write {
if out != src {
if err := os.WriteFile(path, []byte(out), 0o644); err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
rc = 1
}
}
continue
}
fmt.Print(out)
}
return rc
}
func cmdLint(args []string) int {
fs := flag.NewFlagSet("lint", flag.ExitOnError)
disable := fs.String("disable", "", "comma-separated rule codes to disable")
fs.Parse(args)
if fs.NArg() == 0 {
fmt.Fprintln(os.Stderr, "usage: gasm lint <file...>")
return 2
}
disabled := map[string]bool{}
for _, code := range strings.Split(*disable, ",") {
if code = strings.TrimSpace(code); code != "" {
disabled[code] = true
}
}
hadError := false
for _, path := range fs.Args() {
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
hadError = true
continue
}
file, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
hadError = true
}
diags := lint.File(file, lint.Config{Arch: arch.FromFilename(path), Disable: disabled})
for _, d := range diags {
fmt.Printf("%s:%d:%d: %s: %s [%s]\n", path, d.Pos.Line, d.Pos.Column, d.Severity, d.Message, d.Code)
if d.Severity == lint.Error {
hadError = true
}
}
}
if hadError {
return 1
}
return 0
}
func cmdLSP(args []string) int {
fs := flag.NewFlagSet("lsp", flag.ExitOnError)
fs.Parse(args)
srv := lsp.New(os.Stdin, os.Stdout)
if err := srv.Run(); err != nil {
fmt.Fprintln(os.Stderr, "gasm lsp:", err)
return 1
}
return 0
}
func cmdAsm(args []string) int {
fs := flag.NewFlagSet("asm", flag.ExitOnError)
out := fs.String("o", "", "write the concatenated machine code to this file")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm asm [-o out.bin] <file>")
return 2
}
path := fs.Arg(0)
if arch.FromFilename(path) != arch.AMD64 {
fmt.Fprintln(os.Stderr, "gasm asm: only amd64 is supported in this Phase 2 increment")
return 1
}
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm:", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
var all []byte
functions := 0
for _, d := range f.Decls {
txt, ok := d.(*ast.Text)
if !ok {
continue
}
code, _, err := asm.Assemble(txt)
if err != nil {
fmt.Fprintf(os.Stderr, "%s: %s: %v\n", path, txt.Name.Name, err)
return 1
}
functions++
fmt.Printf("%s: %d bytes\n", txt.Name.Name, len(code))
for i := 0; i < len(code); i += 16 {
end := i + 16
if end > len(code) {
end = len(code)
}
fmt.Printf(" %04x:", i)
for _, b := range code[i:end] {
fmt.Printf(" %02x", b)
}
fmt.Println()
}
all = append(all, code...)
}
if functions == 0 {
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found")
return 1
}
if *out != "" {
if err := os.WriteFile(*out, all, 0o644); err != nil {
fmt.Fprintln(os.Stderr, "gasm asm:", err)
return 1
}
fmt.Printf("wrote %d bytes to %s\n", len(all), *out)
}
return 0
}
+174
View File
@@ -0,0 +1,174 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"bytes"
"io"
"os"
"path/filepath"
"strings"
"testing"
)
const clean = "#include \"textflag.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\tMOVQ AX, BX\n" +
"loop:\n" +
"\tJMP loop\n" +
"\tRET\n"
const buggy = "#include \"textflag.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\tBOGUS AX, BX\n" +
"\tJMP nowhere\n" +
"\tRET\n"
func writeTemp(t *testing.T, name, content string) string {
t.Helper()
path := filepath.Join(t.TempDir(), name)
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
t.Fatal(err)
}
return path
}
// capture runs fn with stdout and stderr redirected and returns both plus the
// exit code fn produced.
func capture(fn func() int) (stdout, stderr string, code int) {
oldOut, oldErr := os.Stdout, os.Stderr
rOut, wOut, _ := os.Pipe()
rErr, wErr, _ := os.Pipe()
os.Stdout, os.Stderr = wOut, wErr
code = fn()
wOut.Close()
wErr.Close()
os.Stdout, os.Stderr = oldOut, oldErr
ob, _ := io.ReadAll(rOut)
eb, _ := io.ReadAll(rErr)
return string(ob), string(eb), code
}
func TestCmdTokens(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
out, _, code := capture(func() int { return cmdTokens([]string{path}) })
if code != 0 {
t.Fatalf("code = %d", code)
}
if !strings.Contains(out, "IDENT") || !strings.Contains(out, "TEXT") {
t.Errorf("token dump missing expected tokens:\n%s", out)
}
}
func TestCmdParseOK(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
out, _, code := capture(func() int { return cmdParse([]string{path}) })
if code != 0 {
t.Fatalf("code = %d", code)
}
if !strings.Contains(out, "OK") || !strings.Contains(out, "1 functions") {
t.Errorf("parse output = %q", out)
}
}
func TestCmdParseError(t *testing.T) {
path := writeTemp(t, "bad_amd64.s", "TEXT ·f(SB), NOSPLIT, $0\n) (\n\tRET\n")
_, errOut, code := capture(func() int { return cmdParse([]string{path}) })
if code != 1 {
t.Fatalf("code = %d, want 1", code)
}
if errOut == "" {
t.Error("expected a parse error on stderr")
}
}
func TestCmdParseMissingFile(t *testing.T) {
_, _, code := capture(func() int { return cmdParse([]string{"/nonexistent/file.s"}) })
if code != 1 {
t.Fatalf("code = %d, want 1", code)
}
}
func TestCmdLintClean(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
_, _, code := capture(func() int { return cmdLint([]string{path}) })
if code != 0 {
t.Fatalf("clean file should lint with code 0, got %d", code)
}
}
func TestCmdLintErrors(t *testing.T) {
path := writeTemp(t, "f_amd64.s", buggy)
out, _, code := capture(func() int { return cmdLint([]string{path}) })
if code != 1 {
t.Fatalf("code = %d, want 1", code)
}
if !strings.Contains(out, "unknown-instruction") || !strings.Contains(out, "undefined-label") {
t.Errorf("lint output missing expected codes:\n%s", out)
}
}
func TestCmdLintDisable(t *testing.T) {
path := writeTemp(t, "f_amd64.s", buggy)
args := []string{"-disable", "unknown-instruction,undefined-label", path}
_, _, code := capture(func() int { return cmdLint(args) })
if code != 0 {
t.Fatalf("disabling both rules should yield code 0, got %d", code)
}
}
func TestCmdFmtStdout(t *testing.T) {
path := writeTemp(t, "f_amd64.s", "TEXT ·f(SB),NOSPLIT,$0\nMOVQ AX,BX\nRET\n")
out, _, code := capture(func() int { return cmdFmt([]string{path}) })
if code != 0 {
t.Fatalf("code = %d", code)
}
if !strings.Contains(out, "TEXT ·f(SB), NOSPLIT, $0") || !strings.Contains(out, "\tMOVQ AX, BX") {
t.Errorf("formatted output unexpected:\n%s", out)
}
}
func TestCmdFmtWrite(t *testing.T) {
path := writeTemp(t, "f_amd64.s", "TEXT ·f(SB),NOSPLIT,$0\nMOVQ AX,BX\nRET\n")
_, _, code := capture(func() int { return cmdFmt([]string{"-w", path}) })
if code != 0 {
t.Fatalf("code = %d", code)
}
b, _ := os.ReadFile(path)
if !strings.Contains(string(b), "\tMOVQ AX, BX") {
t.Errorf("file not rewritten:\n%s", b)
}
// Idempotent: a second -w pass leaves the file unchanged.
_, _, _ = capture(func() int { return cmdFmt([]string{"-w", path}) })
b2, _ := os.ReadFile(path)
if string(b) != string(b2) {
t.Error("fmt -w is not idempotent")
}
}
func TestUsage(t *testing.T) {
var b bytes.Buffer
usage(&b)
if !strings.Contains(b.String(), "gasm") {
t.Errorf("usage text unexpected:\n%s", b.String())
}
}
func TestCmdArgErrors(t *testing.T) {
// Missing file arguments produce a usage error (code 2).
if _, _, code := capture(func() int { return cmdFmt(nil) }); code != 2 {
t.Errorf("cmdFmt() code = %d, want 2", code)
}
if _, _, code := capture(func() int { return cmdLint(nil) }); code != 2 {
t.Errorf("cmdLint() code = %d, want 2", code)
}
if _, _, code := capture(func() int { return cmdTokens(nil) }); code != 2 {
t.Errorf("cmdTokens() code = %d, want 2", code)
}
if _, _, code := capture(func() int { return cmdParse(nil) }); code != 2 {
t.Errorf("cmdParse() code = %d, want 2", code)
}
}
+212
View File
@@ -0,0 +1,212 @@
# Architecture
How gasm-devkit is put together and why.
## Design goals
1. **A real AST, not a grammar hack.** The linter, analyser, assembler and
language server all need to *reason* about assembly — not just colour it.
So the centre of the toolkit is a hand-written lexer and a parser that
produce a typed AST with source positions on every node.
2. **Architecture as data, not code.** Per-architecture differences (amd64,
arm64, riscv64, loong64) live in register and instruction *tables* (`arch`),
never in `if arch == …` branches scattered through the logic. The
instruction tables are generated from the Go toolchain's own assembler
source (`just gen`), so adding or refreshing an architecture is a data
operation, not a coding one.
3. **Open integration surface.** Everything the toolkit can do is reachable
through two vendor-neutral interfaces: a CLI and an LSP server. No editor
owns the toolkit; the toolkit is offered to editors on standard terms.
## Pipeline
```mermaid
graph TD
SRC["source .s"] --> LEX["lexer<br/>token stream"]
LEX --> PAR["parser<br/>AST + diagnostics"]
LEX --> FMT["format<br/>re-space tokens"]
PAR --> LINT["lint<br/>static checks"]
PAR --> LSP["lsp server"]
LEX --> LSP
ARCH["arch tables<br/>amd64 / arm64 / riscv64 / loong64"] --> LINT
ARCH --> LSP
LINT --> LSP
FMT --> CLI["gasm CLI"]
LINT --> CLI
PAR --> CLI
LEX --> CLI
LSP --> EDITOR["any LSP editor"]
```
The lexer is the shared foundation: the parser builds the AST from it, the
formatter re-spaces its tokens directly, and the language server uses it for
semantic highlighting.
## Components
### `token` and `lexer`
The scanner is hand-written and permissive: it never panics and maps anything
it cannot classify to an `Illegal` token, so every downstream tool still works
on malformed input. Newlines are significant tokens, because Plan 9 assembly
is line-oriented and the parser relies on line structure.
The middle dot (`·`, U+00B7) is treated as an identifier character so that
`·funcName(SB)` lexes as one symbol. Multi-character operators (`<<`, `>>`,
`->`) are recognised so arm64 shift operands scan correctly. A backslash
immediately before a newline is a C-preprocessor line continuation (used by
`#define` macros in the runtime `.s` files); the lexer splices the lines
together so a multi-line macro becomes one logical line the parser treats as an
opaque preprocessor directive.
### `ast` and `parser`
The parser is **line-oriented**, matching how the Plan 9 assembler reads a
file: it groups tokens into lines, classifies each line (directive, label,
instruction, comment, preprocessor) and dispatches. A malformed line is
reported and skipped; it never aborts the file.
Operands are parsed into a faithful, flat representation. The amd64
addressing modes — `reg`, `$imm`, `(base)`, `off(base)`, `(base)(index*scale)`,
`name+off(FP)`, `name<>(SB)` — are all captured structurally, and the original
token text is retained for fidelity.
A deliberate boundary: the AST records **syntax only**. Whether a bare
identifier is a register or a label is an *architecture* question, so it is
left to `arch` and resolved in the lint/lsp layers. This keeps the parser
arch-agnostic and its output deterministic.
### `arch`
Register files are generated programmatically (the regular `R8`–`R15`,
`X0`–`X15`, `Y0`–`Y15`, `Z0`–`Z31`, `K0`–`K7` ranges) plus the irregularly
named registers listed explicitly. Instruction names are **generated from the
Go toolchain's own assembler source** (`cmd/internal/obj/<arch>/anames.go`,
plus the common opcodes and the per-architecture front-end aliases such as the
arm64 `B`/`BL` branches and the `.P`/`.W` load-store addressing suffixes) by
`just gen`, so the tables always match what the real assembler accepts. Each
mnemonic maps to a summary and an optional operand-count range; counts are
recorded only where unambiguous (`-1` disables the operand-count lint for that
instruction) so the linter stays silent rather than guess. For architectures
with highly variable operand forms (arm64, riscv64, loong64) only a few
fixed-arity instructions (`RET`, `NOP`, `JMP`, `CALL`) carry counts at all.
### `lint`
Rules are conservative by design — silence beats a false positive. The rules
are `unknown-instruction`, `operand-count`, `undefined-label`,
`duplicate-label`, `missing-ret`, `missing-textflag-include`, `abi-argsize` and
`unreachable-code`. Every diagnostic carries a stable code so callers can
disable rules individually, and arch-specific rules switch off entirely when
the target architecture cannot be inferred from the file name.
Two things keep the rules honest on real-world code:
- **Pseudo-ops and macros are not instructions.** `unknown-instruction` knows
the assembler pseudo-ops (`BYTE`, `WORD`, `FUNCDATA`, `PCDATA`, …) and
recognises macro invocations — an in-file `#define` name, or any identifier
containing an underscore (no Plan 9 mnemonic ever does).
- **Macro-heavy files get the label/RET heuristics turned off.** Without a
preprocessor, labels a macro defines are invisible, so `undefined-label` and
`missing-ret` are suppressed for files that use macros (an in-file `#define`
or a `#include` of anything other than `textflag.h`). `missing-ret` also
treats a trailing unconditional jump and `UNDEF` as valid terminators.
The result is validated by `TestGoRuntimeCorpus`, which parses and lints every
`src/runtime/*.s` file the toolchain ships for all four architectures and
asserts zero parse errors and zero error-severity diagnostics.
Two deeper analyses sit on top of the AST:
- **`abi-argsize`.** Hand-written kernels document their signature in a
`// func …` comment above the `TEXT`. The linter parses that signature with
the standard library's Go parser, lays out the parameters and results under
Go's ABI0 stack rules (results begin on a word boundary after the
parameters), and checks the total against the argument size declared in the
`TEXT` directive. It only runs for stack-argument functions (a non-zero
declared arg area that is actually addressed through `FP`), and aborts
silently on a type whose size it cannot determine — so it never guesses.
- **`unreachable-code`.** Code after a `RET` and before the next label is
dead. The check is suppressed for any function whose reachability cannot be
decided statically: those using PC-relative jumps (`JMP 2(PC)`),
register-indirect branches (`JALR`/`JR`/`JIRL`/`BR`/`BLR`), or living in a
file with `#ifdef` conditionals. `UNDEF` is deliberately not a terminator —
code after it is occasionally intentional metadata.
- **`register-clobber` (register liveness).** The linter builds the function's
control-flow graph (basic blocks split at labels and after branches, with
fall-through and jump-target edges), computes a conservative per-instruction
register def/use, and runs the standard backward liveness iteration to a fixed
point. On top of that it flags a **callee-saved register that is written but
never saved and restored** — the per-architecture callee-saved set is amd64
`BX/BP/R12–R15`, arm64 `R19–R30`, riscv64 `X1/X8/X9/X18–X27`, loong64
`R1/R22–R31`. This is an *audit*: the runtime's own assembly clobbers these
registers freely (it controls both sides of the call), so the rule is
advisory there, but in hand-written kernels called from ordinary Go code a
clobber is a genuine ABI violation. It runs only on macro-free files, where
no opaque macro can perform the save/restore.
- **`funcdata-pcdata`.** `FUNCDATA $idx, sym(SB)` and `PCDATA $idx, $val` are
checked for well-formed operands (arity, immediate index and value, symbol
reference) and a literal index is range-checked; a named index constant such
as `$PCDATA_StackMapIndex` is accepted without a range check.
### `format`
The formatter works on the **token stream, not the AST**, so it preserves
every line — comments and blanks included. It only normalises indentation,
operand spacing and per-function mnemonic alignment. It is idempotent and its
output always round-trips through the parser.
### `lsp`
The server speaks JSON-RPC 2.0 with `Content-Length` framing over any
`io.Reader`/`io.Writer` (normally stdin/stdout). It maintains an in-memory
document store, republishes diagnostics on every change, and provides:
- **completion** — instructions, registers, pseudo-registers, textflag macros
and local labels;
- **hover** — instruction summaries and register descriptions from `arch`;
- **document symbols** — `TEXT` functions with their labels, plus `GLOBL`/`DATA`;
- **semantic tokens** — syntax highlighting delivered as LSP semantic tokens,
classified with the lexer plus `arch` (instructions, registers by class,
pseudo-registers, labels, immediates, comments, directives, textflag macros).
Semantic tokens are the key to editor-agnostic highlighting: the editor renders
them from the standard LSP legend, so no editor-specific grammar is needed.
### `asm`
The standalone assembler (Phase 2). Its core is an amd64 instruction encoder:
a REX/ModR-M/SIB/displacement/immediate engine plus the scalar instruction set,
with the Plan 9 operand order (source first) mapped onto the x86 encoding.
Every encoding is validated by decoding it again with `golang.org/x/arch` — the
one module dependency, used in tests only and never linked into the binary.
On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each
operand to an encoder operand, and lays the instructions out in two passes so
local labels resolve to fixed rel32 jump offsets. The `FP`/`SP` pseudo-
registers are translated onto the hardware stack pointer — `x+N(FP)` becomes
`(N+8)(SP)` for a zero-frame function and `(N+frame+16)(SP)` once a frame
pointer is set up, with the matching Go prologue/epilogue generated — so the
output is byte-identical to the Go assembler for these cases. SIMD is handled
SIMD is handled
by a VEX (AVX/AVX2) encoder — the two- and three-byte VEX prefixes with XMM/YMM
registers — across three operand forms (the three-operand NDS form, the
two-operand reg/rm form, and the immediate-shift form), together covering the
bulk of the integer SIMD set; each encoding is validated by round-trip
decoding. This increment covers register / memory / immediate / FP-frame
operands, local-label jumps and these VEX SIMD forms; the remaining SIMD forms
(shuffles, extract/insert, permute, moves), EVEX / AVX-512, `SB` (global
symbol) operands (relocations) and object-file emission are the rest of
Phase 2.
## Extension points
- **New architecture:** add an entry to the generator in `_gen`, run
`just gen`, and add a `buildXXX()` register file plus a case in `ForArch`.
- **New lint rule:** add a function in `lint` and a rule-code constant.
- **New LSP feature:** add a method case in `dispatch` and a handler.
The phases follow a dependency chain. Phase 1 (static analysis) builds only on
the AST; Phase 2 (the standalone assembler) emits object code; Phases 3
(dynamic analysis) and 4 (the debugger) both consume the execution substrate
that the assembler provides.
+79
View File
@@ -0,0 +1,79 @@
# Using gasm-devkit with Zed
This document is deliberately blunt, because the situation is a genuine
conflict between two of the project's own commitments, and papering over it
would be dishonest.
## The conflict
gasm-devkit is **pure Go, no C, no cgo, no JavaScript runtimes, no native
binaries, no vendor lock-in, no platform-specific IDE internals.**
Zed's extension model, as verified against Zed's own documentation, is:
- Extensions are written in **Rust** and compiled to **WebAssembly**
(`wasm32-wasip2`).
- Syntax highlighting is provided by **Tree-sitter** grammars, which are
**C** compiled to WebAssembly with the wasi-sdk, from a grammar written in a
**JavaScript** DSL.
- A *new* language cannot be registered through configuration alone. Defining
a language requires an extension, and every language extension must name a
Tree-sitter grammar. (Zed's `lsp` settings section configures
already-registered servers; it does not register an arbitrary external binary
for a brand-new language.)
There is therefore **no pure-Go path into Zed's extension host.** This is a
property of Zed, not of gasm-devkit: no language tooling author can feed Zed a
pure-Go highlighting grammar, because Zed's highlighting engine is Tree-sitter
and its plugin runtime is Rust/WASM.
## What gasm-devkit gives Zed regardless
The toolkit's integration surface is the **Language Server Protocol**, an open
standard. Through `gasm lsp` it provides, with zero editor-specific code:
- autocomplete (instructions, registers, pseudo-registers, labels),
- hover documentation,
- diagnostics (the linter, pushed as you type),
- document outline (functions and labels),
- **syntax highlighting, delivered as LSP semantic tokens.**
That last point matters: Zed can render highlighting entirely from LSP semantic
tokens (`"semantic_tokens": "full"` replaces Tree-sitter highlighting for a
language). So the highlighting *capability* exists in pure Go; what Zed needs
is merely to be told that `.s` files are a language served by `gasm lsp`.
## The honest options
1. **Use an editor that registers an external LSP by configuration.**
Neovim, Helix, VS Code and Sublime all let you associate `.s` with the
`gasm lsp` binary and use its semantic tokens — no Rust, no C, no lock-in.
This is the option that satisfies every stated constraint with no
exception.
2. **Treat a Zed adapter as one quarantined exception.** A minimal Zed
extension — a few lines of Rust that register the language and launch
`gasm lsp` — plus either a Tree-sitter grammar or `"full"` semantic tokens
for highlighting. Crucially, this adapter is the *editor's plugin format*;
it is sandboxed inside Zed and never linked into, compiled into, or shipped
with the Go toolkit. gasm-devkit itself stays pure Go. But producing it
uses the Rust/wasi-sdk/Tree-sitter toolchain, which the project constraints
forbid — so it must be a conscious, explicit decision, not a silent one.
The author's philosophy — digital sovereignty, no dependency on toolchains he
does not control — is the tie-breaker, and it is a value judgement rather than
a technical one. gasm-devkit is built so that **either** choice keeps the
toolkit itself clean: the pure-Go core and the LSP are the product; a Zed
adapter, if ever wanted, is a thin, separable leaf.
## Wiring the LSP (editor-agnostic)
Run the server and point an LSP client at it:
```sh
go run ./cmd/gasm lsp # or: go install ./cmd/gasm && gasm lsp
```
Associate the command with `*.s` (and `*_amd64.s` / `*_arm64.s`) in whichever
editor you use. The server infers the target architecture from the file-name
suffix and selects the amd64 or arm64 instruction tables accordingly.
+213
View File
@@ -0,0 +1,213 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package format implements a canonical formatter for GAsm source — the
// equivalent of gofmt for Plan 9 assembly. It works on the token stream
// rather than the AST so that every line (including comments and blanks) is
// preserved; it only normalises indentation, operand spacing and per-function
// mnemonic alignment. Formatting is idempotent.
package format
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Source returns the canonical formatting of src.
func Source(path, src string) string {
lines := splitLines(lexer.Tokenize(src))
// First pass: classify each line and record, for every instruction, the
// index of the TEXT function it belongs to, so that mnemonic widths can be
// aligned per function.
type info struct {
kind int
mnemLen int
funcID int
}
const (
kBlank = iota
kComment
kPreproc
kDirective
kLabel
kInstr
)
infos := make([]info, len(lines))
funcID := -1
maxWidth := map[int]int{} // funcID -> widest mnemonic
for i, line := range lines {
inf := info{kind: kBlank, funcID: funcID}
if len(line) > 0 {
switch {
case line[0].Kind == token.Comment:
inf.kind = kComment
case line[0].Kind == token.Hash:
inf.kind = kPreproc
case line[0].Kind == token.Ident && isDirective(line[0].Text):
inf.kind = kDirective
if line[0].Text == "TEXT" {
funcID++
inf.funcID = funcID
} else {
funcID = -1
inf.funcID = -1
}
case len(line) >= 2 && line[1].Kind == token.Colon:
inf.kind = kLabel
default:
inf.kind = kInstr
inf.funcID = funcID
inf.mnemLen = len(line[0].Text)
if funcID >= 0 && inf.mnemLen > maxWidth[funcID] {
maxWidth[funcID] = inf.mnemLen
}
}
}
infos[i] = inf
}
// Second pass: render.
var b strings.Builder
inBody := false
for i, line := range lines {
inf := infos[i]
var out string
switch inf.kind {
case kBlank:
out = ""
case kComment:
if inBody {
out = "\t" + line[0].Text
} else {
out = line[0].Text
}
case kPreproc:
out = renderPreproc(line)
case kDirective:
out = line[0].Text + " " + renderOps(line[1:])
inBody = line[0].Text == "TEXT"
case kLabel:
out = line[0].Text + ":"
// A label may share its line with an instruction; emit the
// instruction on the following line.
if rest := line[2:]; len(rest) > 0 {
out += "\n" + renderInstr(rest, maxWidth[inf.funcID])
}
case kInstr:
out = renderInstr(line, maxWidth[inf.funcID])
}
b.WriteString(strings.TrimRight(out, " \t"))
b.WriteByte('\n')
}
return b.String()
}
// renderInstr renders an instruction line: a tab, the mnemonic padded to the
// function's alignment width, then the re-spaced operands.
func renderInstr(line []token.Token, width int) string {
if len(line) == 0 {
return ""
}
mnem := line[0].Text
ops := renderOps(line[1:])
if ops == "" {
return "\t" + mnem
}
if width < len(mnem) {
width = len(mnem)
}
return "\t" + mnem + strings.Repeat(" ", width-len(mnem)) + " " + ops
}
// renderPreproc renders a preprocessor line such as #include "textflag.h".
func renderPreproc(line []token.Token) string {
// "#" directive [args]
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
line[2].Kind == token.String {
return "#include " + line[2].Text
}
parts := make([]string, 0, len(line)-1)
for _, t := range line[1:] {
parts = append(parts, t.Text)
}
return "#" + strings.Join(parts, " ")
}
// renderOps re-spaces a run of operand tokens into canonical form. It never
// invents or drops token text; it only chooses the whitespace between tokens.
func renderOps(toks []token.Token) string {
var b strings.Builder
for i, t := range toks {
if i > 0 && spaceBetween(toks[i-1], t) {
b.WriteByte(' ')
}
b.WriteString(t.Text)
}
return b.String()
}
// spaceBetween decides whether a single space separates prev and cur.
func spaceBetween(prev, cur token.Token) bool {
switch cur.Kind {
case token.RParen:
return false
case token.Comma:
return false
case token.Star, token.Plus, token.Minus, token.Slash:
return false
case token.LShift, token.RShift, token.Arrow, token.At:
return false
case token.LAngle, token.RAngle:
return false
case token.LParen:
// Attach '(' to a preceding name, number, ')' or '>'.
switch prev.Kind {
case token.Ident, token.Number, token.RParen, token.RAngle:
return false
default:
return true
}
}
switch prev.Kind {
case token.LParen, token.Star, token.Plus, token.Minus, token.Slash:
return false
case token.Dollar:
return false
case token.LShift, token.RShift, token.Arrow, token.At:
return false
case token.LAngle, token.RAngle:
return false
case token.Comma:
return true
}
return true
}
func isDirective(s string) bool {
return s == "TEXT" || s == "DATA" || s == "GLOBL"
}
// splitLines groups tokens into lines, dropping Newline and EOF tokens.
func splitLines(toks []token.Token) [][]token.Token {
var lines [][]token.Token
var cur []token.Token
for _, t := range toks {
if t.Kind == token.EOF {
break
}
if t.Kind == token.Newline {
lines = append(lines, cur)
cur = nil
continue
}
cur = append(cur, t)
}
if len(cur) > 0 {
lines = append(lines, cur)
}
return lines
}
+97
View File
@@ -0,0 +1,97 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package format
import (
"os"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
func TestGolden(t *testing.T) {
in := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"MOVQ swin_base+0(FP), SI\n" +
"LEAQ (SI)(BX*4), R9\n" +
"ANDQ $-8, R10\n" +
"VFMADD231PD Z14, Z12, Z10\n" +
"RET\n"
want := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\tMOVQ swin_base+0(FP), SI\n" +
"\tLEAQ (SI)(BX*4), R9\n" +
"\tANDQ $-8, R10\n" +
"\tVFMADD231PD Z14, Z12, Z10\n" +
"\tRET\n"
got := Source("f_amd64.s", in)
if got != want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
}
func TestOperandSpacing(t *testing.T) {
cases := map[string]string{
"4(SI)": "4(SI)",
"(SI)(BX*4)": "(SI)(BX*4)",
"$-8": "$-8",
"$0x80020100": "$0x80020100",
"swin_base+0(FP)": "swin_base+0(FP)",
"mask24<>(SB)": "mask24<>(SB)",
"·idx16+0(SB)/4": "·idx16+0(SB)/4",
}
for in, want := range cases {
toks := lexOperands(in)
if got := renderOps(toks); got != want {
t.Errorf("renderOps(%q) = %q, want %q", in, got, want)
}
}
}
// lexOperands lexes a single operand string and drops the EOF token.
func lexOperands(s string) []token.Token {
toks := lexer.Tokenize(s)
return toks[:len(toks)-1] // drop trailing EOF
}
func TestIdempotent(t *testing.T) {
src, err := os.ReadFile("../testdata/sample_amd64.s")
if err != nil {
t.Fatal(err)
}
once := Source("sample_amd64.s", string(src))
twice := Source("sample_amd64.s", once)
if once != twice {
t.Fatal("formatting is not idempotent on the fixture")
}
}
// TestRoundTrip checks that formatting produces source that still parses
// cleanly, on the fixture and on the real go-flac kernels when present.
func TestRoundTrip(t *testing.T) {
files := []string{"../testdata/sample_amd64.s"}
real, _ := filepath.Glob("../../go-libraries/go-*/*.s")
files = append(files, real...)
for _, path := range files {
src, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
formatted := Source(path, string(src))
if _, errs := parser.Parse(path, formatted); len(errs) > 0 {
t.Errorf("formatted %s no longer parses: %v", path, errs)
}
if strings.TrimSpace(formatted) == "" {
t.Errorf("formatted %s is empty", path)
}
}
}
+7
View File
@@ -0,0 +1,7 @@
module sourcedock.dev/petrbalvin/gasm-devkit
go 1.26
toolchain go1.26.5
require golang.org/x/arch v0.29.0
+2
View File
@@ -0,0 +1,2 @@
golang.org/x/arch v0.29.0 h1:8sSET5wB0+exBm0FGmOtdHMqjlRdV2DRD3/IV6OZgho=
golang.org/x/arch v0.29.0/go.mod h1:0X+GdSIP+kL5wPmpK7sdkEVTt2XoYP0cSjQSbZBwOi8=
+45
View File
@@ -0,0 +1,45 @@
# Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
# SPDX-License-Identifier: BSD-3-Clause
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
version := "0.1.0"
default:
@just --list
# Download module dependencies.
install:
go mod download
# Vet + gofmt check — zero errors, zero warnings.
build:
go vet ./...
@test -z "$(gofmt -l .)" || { echo "gofmt diff:"; gofmt -l .; exit 1; }
# Full test suite + race detector + 80 % coverage gate.
test:
go test -race -count=1 -coverprofile=coverage.out ./...
go tool cover -func=coverage.out | awk '/^total:/{gsub("%","",$3);if($3+0<80){print "coverage "$3"% < 80%";exit 1}print "coverage "$3"%"}'
# Format all Go sources.
fmt:
gofmt -w .
# Run the gasm CLI (pass args after --, e.g. `just run -- lint file.s`).
run *ARGS:
go run -ldflags "-X main.version={{version}}" ./cmd/gasm {{ARGS}}
# Install the gasm binary into $GOBIN (stamped with the release version).
install-bin:
go install -ldflags "-X main.version={{version}}" ./cmd/gasm
# Regenerate the architecture instruction tables from the Go toolchain source.
gen:
go run _gen/gen.go
gofmt -w arch/
# Remove build artefacts.
uninstall:
rm -f coverage.out gasm
find . -name '*.test' -delete
+395
View File
@@ -0,0 +1,395 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package lexer implements a hand-written scanner for Go's Plan 9 assembler
// (GAsm). It turns a source string into a flat token stream that the parser,
// formatter and language server all build on. The scanner is deliberately
// permissive: it never panics and maps anything it cannot classify to an
// Illegal token so that downstream tools can still operate on malformed input.
package lexer
import (
"strings"
"unicode"
"unicode/utf8"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// middleDot is the Plan 9 symbol separator (U+00B7), used in ·funcName(SB).
const middleDot = '\u00B7'
// Lexer scans a source string one token at a time.
type Lexer struct {
src []rune
off []int // off[i] is the byte offset of src[i]; off[len(src)] is len(bytes)
i int // index of the current rune
line int // one-based line of src[i]
col int // one-based rune column of src[i]
}
// New returns a Lexer over src.
func New(src string) *Lexer {
runes := []rune(src)
off := make([]int, len(runes)+1)
b := 0
for i, r := range runes {
off[i] = b
b += utf8.RuneLen(r)
}
off[len(runes)] = b
return &Lexer{src: runes, off: off, line: 1, col: 1}
}
// Tokenize scans src fully and returns every token up to and including the
// trailing EOF token.
func Tokenize(src string) []token.Token {
l := New(src)
var out []token.Token
for {
tok := l.Next()
out = append(out, tok)
if tok.Kind == token.EOF {
return out
}
}
}
// cur returns the current rune, or 0 at end of input.
func (l *Lexer) cur() rune {
if l.i >= len(l.src) {
return 0
}
return l.src[l.i]
}
// peek returns the rune k positions ahead, or 0 past the end.
func (l *Lexer) peek(k int) rune {
if l.i+k >= len(l.src) || l.i+k < 0 {
return 0
}
return l.src[l.i+k]
}
// pos snapshots the current source position.
func (l *Lexer) pos() token.Position {
return token.Position{Offset: l.off[l.i], Line: l.line, Column: l.col}
}
// advance consumes one rune, updating line and column bookkeeping.
func (l *Lexer) advance() {
if l.i >= len(l.src) {
return
}
if l.src[l.i] == '\n' {
l.line++
l.col = 1
} else {
l.col++
}
l.i++
}
// make builds a token of the given kind spanning [start, current position).
func (l *Lexer) make(kind token.Kind, start token.Position, text string) token.Token {
return token.Token{Kind: kind, Text: text, Pos: start, End: l.pos()}
}
// Next returns the next token, skipping spaces and tabs. Newlines are
// significant and returned as Newline tokens so the parser can treat the
// stream line by line.
func (l *Lexer) Next() token.Token {
for {
// Skip horizontal whitespace. A backslash immediately before a newline
// is a C-preprocessor line continuation (used by #define macros in the
// runtime .s files): splice the lines together by consuming both, so
// the whole macro becomes one logical line that the parser treats as an
// opaque preprocessor directive.
for {
c := l.cur()
if c == ' ' || c == '\t' || c == '\r' {
l.advance()
continue
}
if c == '\\' && (l.peek(1) == '\n' || l.peek(1) == '\r') {
l.advance() // backslash
if l.cur() == '\r' {
l.advance()
}
if l.cur() == '\n' {
l.advance()
}
continue
}
break
}
start := l.pos()
r := l.cur()
switch {
case r == 0:
return l.make(token.EOF, start, "")
case r == '\n':
l.advance()
return l.make(token.Newline, start, "\n")
case r == '/':
switch l.peek(1) {
case '/':
return l.lineComment(start)
case '*':
return l.blockComment(start)
default:
l.advance()
return l.make(token.Slash, start, "/")
}
case r == '"':
return l.string(start)
case r == '\'':
return l.runeLit(start)
case isIdentStart(r):
return l.ident(start)
case isDigit(r):
return l.number(start)
default:
return l.punct(start)
}
}
}
// lineComment consumes a // comment up to, but not including, the newline.
func (l *Lexer) lineComment(start token.Position) token.Token {
var b strings.Builder
for l.cur() != 0 && l.cur() != '\n' {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Comment, start, b.String())
}
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
func (l *Lexer) blockComment(start token.Position) token.Token {
var b strings.Builder
b.WriteRune(l.cur()) // '/'
l.advance()
b.WriteRune(l.cur()) // '*'
l.advance()
for l.cur() != 0 {
if l.cur() == '*' && l.peek(1) == '/' {
b.WriteString("*/")
l.advance()
l.advance()
break
}
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Comment, start, b.String())
}
// string consumes a double-quoted string literal, honouring backslash escapes.
func (l *Lexer) string(start token.Position) token.Token {
var b strings.Builder
b.WriteRune('"')
l.advance() // opening quote
for l.cur() != 0 && l.cur() != '\n' {
r := l.cur()
b.WriteRune(r)
l.advance()
if r == '\\' {
if l.cur() != 0 && l.cur() != '\n' {
b.WriteRune(l.cur())
l.advance()
}
continue
}
if r == '"' {
return l.make(token.String, start, b.String())
}
}
// Unterminated string: return what we have rather than failing.
return l.make(token.String, start, b.String())
}
// runeLit consumes a single-quoted rune literal such as 'a' or '\n'.
func (l *Lexer) runeLit(start token.Position) token.Token {
var b strings.Builder
b.WriteRune('\'')
l.advance() // opening quote
for l.cur() != 0 && l.cur() != '\n' {
r := l.cur()
b.WriteRune(r)
l.advance()
if r == '\\' {
if l.cur() != 0 && l.cur() != '\n' {
b.WriteRune(l.cur())
l.advance()
}
continue
}
if r == '\'' {
return l.make(token.Rune, start, b.String())
}
}
return l.make(token.Rune, start, b.String())
}
// ident consumes an identifier: letters, digits, '_', '.', and the middle dot.
func (l *Lexer) ident(start token.Position) token.Token {
var b strings.Builder
for isIdentChar(l.cur()) {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Ident, start, b.String())
}
// number consumes an integer or floating-point literal. The sign is never
// part of the literal; it is scanned separately as a Minus or Plus token.
func (l *Lexer) number(start token.Position) token.Token {
var b strings.Builder
// Base prefixes.
if l.cur() == '0' && (l.peek(1) == 'x' || l.peek(1) == 'X') {
b.WriteRune(l.cur())
l.advance()
b.WriteRune(l.cur())
l.advance()
for isHexDigit(l.cur()) {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Number, start, b.String())
}
if l.cur() == '0' && (l.peek(1) == 'b' || l.peek(1) == 'B') {
b.WriteRune(l.cur())
l.advance()
b.WriteRune(l.cur())
l.advance()
for l.cur() == '0' || l.cur() == '1' {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Number, start, b.String())
}
if l.cur() == '0' && (l.peek(1) == 'o' || l.peek(1) == 'O') {
b.WriteRune(l.cur())
l.advance()
b.WriteRune(l.cur())
l.advance()
for l.cur() >= '0' && l.cur() <= '7' {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Number, start, b.String())
}
// Decimal, possibly fractional and/or with an exponent.
for isDigit(l.cur()) {
b.WriteRune(l.cur())
l.advance()
}
if l.cur() == '.' && isDigit(l.peek(1)) {
b.WriteRune(l.cur())
l.advance()
for isDigit(l.cur()) {
b.WriteRune(l.cur())
l.advance()
}
}
if l.cur() == 'e' || l.cur() == 'E' {
b.WriteRune(l.cur())
l.advance()
if l.cur() == '+' || l.cur() == '-' {
b.WriteRune(l.cur())
l.advance()
}
for isDigit(l.cur()) {
b.WriteRune(l.cur())
l.advance()
}
}
return l.make(token.Number, start, b.String())
}
// punct consumes a single punctuation or operator token, handling the
// multi-character operators <<, >> and ->.
func (l *Lexer) punct(start token.Position) token.Token {
r := l.cur()
switch r {
case '(':
l.advance()
return l.make(token.LParen, start, "(")
case ')':
l.advance()
return l.make(token.RParen, start, ")")
case ',':
l.advance()
return l.make(token.Comma, start, ",")
case '+':
l.advance()
return l.make(token.Plus, start, "+")
case '-':
if l.peek(1) == '>' {
l.advance()
l.advance()
return l.make(token.Arrow, start, "->")
}
l.advance()
return l.make(token.Minus, start, "-")
case '*':
l.advance()
return l.make(token.Star, start, "*")
case ':':
l.advance()
return l.make(token.Colon, start, ":")
case '$':
l.advance()
return l.make(token.Dollar, start, "$")
case '<':
if l.peek(1) == '<' {
l.advance()
l.advance()
return l.make(token.LShift, start, "<<")
}
l.advance()
return l.make(token.LAngle, start, "<")
case '>':
if l.peek(1) == '>' {
l.advance()
l.advance()
return l.make(token.RShift, start, ">>")
}
l.advance()
return l.make(token.RAngle, start, ">")
case '@':
l.advance()
return l.make(token.At, start, "@")
case '#':
l.advance()
return l.make(token.Hash, start, "#")
default:
// Unknown rune: emit it as Illegal and move on.
l.advance()
return l.make(token.Illegal, start, string(r))
}
}
func isDigit(r rune) bool { return r >= '0' && r <= '9' }
func isHexDigit(r rune) bool {
return isDigit(r) || (r >= 'a' && r <= 'f') || (r >= 'A' && r <= 'F')
}
func isIdentStart(r rune) bool {
return r == '_' || r == middleDot || unicode.IsLetter(r)
}
func isIdentChar(r rune) bool {
return isIdentStart(r) || isDigit(r) || r == '.'
}
+155
View File
@@ -0,0 +1,155 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lexer
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// kinds tokenizes src and returns the kind sequence, dropping Newline/EOF.
func kinds(src string) []token.Kind {
var out []token.Kind
for _, t := range Tokenize(src) {
if t.Kind == token.Newline || t.Kind == token.EOF {
continue
}
out = append(out, t.Kind)
}
return out
}
// texts tokenizes src and returns the literal text of each significant token.
func texts(src string) []string {
var out []string
for _, t := range Tokenize(src) {
if t.Kind == token.Newline || t.Kind == token.EOF {
continue
}
out = append(out, t.Text)
}
return out
}
func eq[T comparable](t *testing.T, got, want []T) {
t.Helper()
if len(got) != len(want) {
t.Fatalf("length mismatch:\n got %v\n want %v", got, want)
}
for i := range got {
if got[i] != want[i] {
t.Fatalf("index %d:\n got %v\n want %v", i, got, want)
}
}
}
func TestTextDirective(t *testing.T) {
eq(t, texts("TEXT ·analyzeO1RangeAVX2(SB), NOSPLIT, $0-65"),
[]string{"TEXT", "·analyzeO1RangeAVX2", "(", "SB", ")", ",", "NOSPLIT", ",", "$", "0", "-", "65"})
}
func TestDataAndGlobl(t *testing.T) {
eq(t, texts("GLOBL ·idx16(SB), RODATA, $64"),
[]string{"GLOBL", "·idx16", "(", "SB", ")", ",", "RODATA", ",", "$", "64"})
eq(t, texts("DATA ·idx16+0(SB)/4, $1"),
[]string{"DATA", "·idx16", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "1"})
}
func TestStaticSymbol(t *testing.T) {
// mask24<> is a file-local symbol; <> must lex as two angle tokens.
eq(t, texts("GLOBL mask24<>(SB), RODATA, $16"),
[]string{"GLOBL", "mask24", "<", ">", "(", "SB", ")", ",", "RODATA", ",", "$", "16"})
}
func TestNegativeImmediate(t *testing.T) {
eq(t, texts("ANDQ $-8, R10"),
[]string{"ANDQ", "$", "-", "8", ",", "R10"})
}
func TestHexImmediate(t *testing.T) {
eq(t, texts("DATA mask24<>+0(SB)/4, $0x80020100"),
[]string{"DATA", "mask24", "<", ">", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "0x80020100"})
}
func TestMemoryAddressing(t *testing.T) {
eq(t, texts("LEAQ (SI)(BX*4), R9"),
[]string{"LEAQ", "(", "SI", ")", "(", "BX", "*", "4", ")", ",", "R9"})
eq(t, texts("VMOVDQU32 Z0, 4(SI)(AX*1)"),
[]string{"VMOVDQU32", "Z0", ",", "4", "(", "SI", ")", "(", "AX", "*", "1", ")"})
}
func TestLabelAndComment(t *testing.T) {
eq(t, kinds("vec1:\n\tJMP vec1 // loop"),
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
}
func TestAVX512Mnemonics(t *testing.T) {
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
eq(t, texts("KTESTW K1, K1"),
[]string{"KTESTW", "K1", ",", "K1"})
}
func TestArm64Shifts(t *testing.T) {
eq(t, texts("ADD R0<<2, R1, R2"),
[]string{"ADD", "R0", "<<", "2", ",", "R1", ",", "R2"})
eq(t, texts("MOVD R3->4, R5"),
[]string{"MOVD", "R3", "->", "4", ",", "R5"})
}
func TestInclude(t *testing.T) {
eq(t, texts(`#include "textflag.h"`),
[]string{"#", "include", `"textflag.h"`})
}
func TestPositions(t *testing.T) {
toks := Tokenize("MOVQ AX, BX\nRET")
// Find RET and check it landed on line 2.
var ret token.Token
for _, tok := range toks {
if tok.Text == "RET" {
ret = tok
}
}
if ret.Pos.Line != 2 || ret.Pos.Column != 1 {
t.Fatalf("RET position = %v, want 2:1", ret.Pos)
}
}
func TestIllegalNeverPanics(t *testing.T) {
// A stray backtick and NUL-ish garbage must not crash the scanner.
toks := Tokenize("MOVQ ` , \x01 AX")
if len(toks) == 0 {
t.Fatal("expected tokens")
}
}
func TestBlockComment(t *testing.T) {
eq(t, kinds("MOVQ /* inline */ AX"),
[]token.Kind{token.Ident, token.Comment, token.Ident})
// An unterminated block comment is tolerated.
toks := Tokenize("MOVQ /* never closed")
if toks[len(toks)-2].Kind != token.Comment {
t.Errorf("expected a comment token, got %v", toks)
}
}
func TestRuneLiteral(t *testing.T) {
eq(t, texts("MOVL $'a', AX"),
[]string{"MOVL", "$", "'a'", ",", "AX"})
}
func TestFloatAndBases(t *testing.T) {
eq(t, texts("$1.5"), []string{"$", "1.5"})
eq(t, texts("$0b1010"), []string{"$", "0b1010"})
eq(t, texts("$0o755"), []string{"$", "0o755"})
eq(t, texts("$1e3"), []string{"$", "1e3"})
}
func TestOperatorVariants(t *testing.T) {
eq(t, texts("R0>>2"), []string{"R0", ">>", "2"})
eq(t, texts("@>"), []string{"@", ">"})
eq(t, texts("a/b"), []string{"a", "/", "b"})
}
+175
View File
@@ -0,0 +1,175 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import (
"go/ast"
"go/parser"
"go/token"
"strconv"
"strings"
)
// abiExpectedArgSize computes the argument-area size (parameters plus results,
// laid out with Go's alignment rules on a 64-bit target) from the `// func …`
// signature in a TEXT function's doc comment. It returns ok=false when there
// is no parseable signature or it uses a type whose size cannot be determined
// (a named type), so the caller can skip the check rather than guess.
//
// The signature is parsed with the standard library's Go parser, so every
// legal signature form (shared names such as `left, right []int32`, nested
// pointers, arrays, structs) is handled correctly.
func abiExpectedArgSize(doc string) (int64, bool) {
sig := signatureLine(doc)
if sig == "" {
return 0, false
}
fset := token.NewFileSet()
f, err := parser.ParseFile(fset, "sig.go", "package p\n"+sig+" {}\n", 0)
if err != nil || len(f.Decls) == 0 {
return 0, false
}
fn, ok := f.Decls[0].(*ast.FuncDecl)
if !ok || fn.Type == nil {
return 0, false
}
return signatureSize(fn.Type.Params, fn.Type.Results)
}
// signatureLine returns the first `func …` line from a doc comment, trimmed.
func signatureLine(doc string) string {
for _, line := range strings.Split(doc, "\n") {
if t := strings.TrimSpace(line); strings.HasPrefix(t, "func ") {
return t
}
}
return ""
}
// signatureSize lays out the parameters and results and returns the total byte
// size of the argument area, matching Go's ABI0 stack layout: parameters are
// laid out first, then the result area begins on a word (8-byte) boundary.
func signatureSize(params, results *ast.FieldList) (int64, bool) {
paramsSize, _, ok := fieldsSizeAlign(params)
if !ok {
return 0, false
}
resultsSize, _, ok := fieldsSizeAlign(results)
if !ok {
return 0, false
}
// With no results the argument area is exactly the parameter size. When
// there are results, the result area begins on a word (8-byte) boundary
// after the parameters (Go's ABI0 stack layout).
if resultsSize == 0 {
return int64(paramsSize), true
}
return int64(alignUp(paramsSize, 8) + resultsSize), true
}
// fieldsSizeAlign lays out a field list sequentially (each field aligned to its
// own alignment) and returns the total size and the maximum field alignment.
func fieldsSizeAlign(list *ast.FieldList) (size, align int, ok bool) {
if list == nil {
return 0, 1, true
}
offset, maxAlign := 0, 1
for _, field := range list.List {
es, ea, fieldOK := typeSizeAlign(field.Type)
if !fieldOK {
return 0, 0, false
}
n := len(field.Names)
if n == 0 {
n = 1
}
for i := 0; i < n; i++ {
offset = alignUp(offset, ea)
offset += es
}
if ea > maxAlign {
maxAlign = ea
}
}
return offset, maxAlign, true
}
// basicSizes maps built-in type names to {size, align} on a 64-bit target.
var basicSizes = map[string][2]int{
"bool": {1, 1}, "byte": {1, 1}, "int8": {1, 1}, "uint8": {1, 1},
"int16": {2, 2}, "uint16": {2, 2},
"int32": {4, 4}, "uint32": {4, 4}, "float32": {4, 4},
"int": {8, 8}, "int64": {8, 8}, "uint": {8, 8}, "uint64": {8, 8},
"uintptr": {8, 8}, "float64": {8, 8},
"complex64": {8, 4}, "complex128": {16, 8},
"string": {16, 8}, "any": {16, 8}, "error": {16, 8},
}
// typeSizeAlign returns the size and alignment in bytes of a type expression,
// or ok=false when the size cannot be determined (an unknown named type).
func typeSizeAlign(e ast.Expr) (size, align int, ok bool) {
switch t := e.(type) {
case *ast.Ident:
if sa, found := basicSizes[t.Name]; found {
return sa[0], sa[1], true
}
return 0, 0, false // named type of unknown size
case *ast.SelectorExpr:
if pkg, isIdent := t.X.(*ast.Ident); isIdent && pkg.Name == "unsafe" && t.Sel.Name == "Pointer" {
return 8, 8, true
}
return 0, 0, false
case *ast.ParenExpr:
return typeSizeAlign(t.X)
case *ast.StarExpr:
return 8, 8, true // pointer
case *ast.MapType, *ast.ChanType, *ast.FuncType:
return 8, 8, true // map / chan / func are pointer-sized
case *ast.InterfaceType:
return 16, 8, true
case *ast.Ellipsis:
return 24, 8, true // variadic parameter is a slice
case *ast.ArrayType:
if t.Len == nil {
return 24, 8, true // slice header
}
n, lenOK := arrayLength(t.Len)
es, ea, elemOK := typeSizeAlign(t.Elt)
if !lenOK || !elemOK {
return 0, 0, false
}
return n * es, ea, true
case *ast.StructType:
return structSizeAlign(t.Fields)
}
return 0, 0, false
}
// structSizeAlign lays out a struct's fields and returns its size (rounded up
// to its alignment) and alignment.
func structSizeAlign(fields *ast.FieldList) (size, align int, ok bool) {
size, align, ok = fieldsSizeAlign(fields)
if !ok {
return 0, 0, false
}
return alignUp(size, align), align, true
}
// arrayLength evaluates a constant array-length expression (a literal, for the
// kernels this toolkit targets).
func arrayLength(e ast.Expr) (int, bool) {
if lit, ok := e.(*ast.BasicLit); ok && lit.Kind == token.INT {
if v, err := strconv.Atoi(lit.Value); err == nil {
return v, true
}
}
return 0, false
}
func alignUp(offset, align int) int {
if align <= 1 {
return offset
}
return (offset + align - 1) &^ (align - 1)
}
+129
View File
@@ -0,0 +1,129 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import "testing"
// TestABIExpectedArgSize checks the Go ABI0 argument-area size computation
// against hand-verified signatures (the same layouts the go-flac kernels use).
func TestABIExpectedArgSize(t *testing.T) {
cases := []struct {
sig string
want int64
}{
{"func f()", 0},
{"func f(a int, b int)", 16},
{"func f(a int32)", 4}, // no results: no word-boundary padding
{"func f(a int32) (r int32)", 12}, // results begin on a word boundary: 4 -> 8, +4
{"func f(a int) int", 16},
{"func f(s []int32, p *[32]uint16) (x uint64, ok bool)", 41},
{"func f(left, right []int32, sums *[4]uint64)", 56},
{"func f(src []byte, dst []int32)", 48},
{"func f(a bool, b int64)", 16}, // bool at 0, int64 aligned to 8
{"func f(x struct{ a int32; b int64 })", 16},
}
for _, c := range cases {
got, ok := abiExpectedArgSize(c.sig)
if !ok {
t.Errorf("%s: could not compute size", c.sig)
continue
}
if got != c.want {
t.Errorf("%s: size = %d, want %d", c.sig, got, c.want)
}
}
}
func TestABIUnknownTypeSkipped(t *testing.T) {
// A bare named type of unknown size must abort the check rather than guess.
// (A *pointer* to a named type is still 8 bytes and is fine.)
if _, ok := abiExpectedArgSize("func f(s Stream)"); ok {
t.Error("bare named type should make the size undecidable")
}
if _, ok := abiExpectedArgSize("func f(s *Stream)"); !ok {
t.Error("pointer to a named type is decidable (8 bytes)")
}
}
// TestABIArgSizeRule checks the lint rule end to end.
func TestABIArgSizeRule(t *testing.T) {
// Matching: the declared arg size agrees with the signature.
clean := lintSrc(t, "#include \"textflag.h\"\n"+
"// func f(a int, b int)\n"+
"TEXT ·f(SB), NOSPLIT, $0-16\n"+
"\tMOVQ a+0(FP), AX\n"+
"\tRET\n")
if codes(clean)[CodeABIArgSize] != 0 {
t.Fatalf("matching arg size should not warn: %+v", clean)
}
// Mismatching: declared 8, signature implies 16.
bad := lintSrc(t, "#include \"textflag.h\"\n"+
"// func f(a int, b int)\n"+
"TEXT ·f(SB), NOSPLIT, $0-8\n"+
"\tMOVQ a+0(FP), AX\n"+
"\tRET\n")
if codes(bad)[CodeABIArgSize] != 1 {
t.Fatalf("mismatching arg size should warn once: %+v", bad)
}
}
// TestABIArgSizeSkipsRegisterABI verifies the check does not fire for functions
// that declare a zero arg area (register ABI) or never touch FP.
func TestABIArgSizeSkipsRegisterABI(t *testing.T) {
diags := lintSrc(t, "#include \"textflag.h\"\n"+
"// func f(a int, b int)\n"+
"TEXT ·f(SB), NOSPLIT, $0-0\n"+
"\tMOVQ AX, BX\n"+
"\tRET\n")
if codes(diags)[CodeABIArgSize] != 0 {
t.Fatalf("register-ABI function must not be checked: %+v", diags)
}
}
// TestUnreachableCode exercises the dead-code detection and its guard rails.
func TestUnreachableCode(t *testing.T) {
// Code after a RET is unreachable.
dead := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tRET\n"+
"\tMOVQ AX, BX\n")
if codes(dead)[CodeUnreachable] != 1 {
t.Fatalf("code after RET should be unreachable: %+v", dead)
}
// A label after the RET makes the following code reachable again.
live := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tRET\n"+
"again:\n"+
"\tJMP again\n")
if codes(live)[CodeUnreachable] != 0 {
t.Fatalf("code after a label is reachable: %+v", live)
}
}
// TestUnreachableGuards verifies the analysis is suppressed where reachability
// cannot be determined statically.
func TestUnreachableGuards(t *testing.T) {
// A PC-relative jump defeats the analysis for the whole function.
pcrel := lintSrcArch(t, "f_amd64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tJCC 2(PC)\n"+
"\tRET\n"+
"\tMOVQ AX, BX\n")
if codes(pcrel)[CodeUnreachable] != 0 {
t.Fatalf("PC-relative functions must be skipped: %+v", pcrel)
}
// A register-indirect branch (riscv JALR) defeats the analysis too.
indirect := lintSrcArch(t, "f_riscv64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tJALR X1, X5\n"+
"\tRET\n"+
"\tMOV X1, X2\n")
if codes(indirect)[CodeUnreachable] != 0 {
t.Fatalf("indirect-branch functions must be skipped: %+v", indirect)
}
}
+96
View File
@@ -0,0 +1,96 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// checkFuncdata validates the structure of FUNCDATA and PCDATA directives,
// which carry the GC stack-map information. The checks are deliberately
// shallow — they confirm the operands are well formed and that a literal index
// is within the small range the runtime uses — and never try to interpret a
// named index constant such as $PCDATA_StackMapIndex.
func checkFuncdata(t *ast.Text, cfg Config) []Diagnostic {
var out []Diagnostic
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "FUNCDATA":
out = append(out, checkFunCDATA(in, cfg)...)
case "PCDATA":
out = append(out, checkPCDATA(in, cfg)...)
}
}
return out
}
// checkFunCDATA validates `FUNCDATA $index, symbol(SB)`.
func checkFunCDATA(in *ast.Instr, cfg Config) []Diagnostic {
if cfg.Disable[CodeFuncdata] {
return nil
}
var out []Diagnostic
if len(in.Operands) != 2 {
return []Diagnostic{{
Pos: in.Mnemonic.Pos, End: in.Mnemonic.End, Severity: Warning, Code: CodeFuncdata,
Message: fmt.Sprintf("FUNCDATA expects 2 operands (index, symbol), got %d", len(in.Operands)),
}}
}
out = append(out, checkIndex(in.Operands[0], "FUNCDATA")...)
if sym := in.Operands[1].Addr.Sym; in.Operands[1].Kind != ast.OpAddr || sym == nil {
out = append(out, Diagnostic{
Pos: in.Operands[1].Pos, Severity: Warning, Code: CodeFuncdata,
Message: "FUNCDATA second operand must be a symbol reference",
})
}
return out
}
// checkPCDATA validates `PCDATA $index, $value`.
func checkPCDATA(in *ast.Instr, cfg Config) []Diagnostic {
if cfg.Disable[CodeFuncdata] {
return nil
}
if len(in.Operands) != 2 {
return []Diagnostic{{
Pos: in.Mnemonic.Pos, End: in.Mnemonic.End, Severity: Warning, Code: CodeFuncdata,
Message: fmt.Sprintf("PCDATA expects 2 operands (index, value), got %d", len(in.Operands)),
}}
}
var out []Diagnostic
out = append(out, checkIndex(in.Operands[0], "PCDATA")...)
if in.Operands[1].Kind != ast.OpImmediate {
out = append(out, Diagnostic{
Pos: in.Operands[1].Pos, Severity: Warning, Code: CodeFuncdata,
Message: "PCDATA value must be an immediate",
})
}
return out
}
// checkIndex validates an immediate index operand. A literal index must lie in
// the small range the runtime uses; a named constant (e.g. $PCDATA_StackMapIndex)
// cannot be evaluated and is accepted without a range check.
func checkIndex(op *ast.Operand, directive string) []Diagnostic {
if op.Kind != ast.OpImmediate {
return []Diagnostic{{
Pos: op.Pos, Severity: Warning, Code: CodeFuncdata,
Message: directive + " index must be an immediate",
}}
}
if op.Imm.HasVal && (op.Imm.Val < 0 || op.Imm.Val > 10) {
return []Diagnostic{{
Pos: op.Pos, Severity: Warning, Code: CodeFuncdata,
Message: fmt.Sprintf("%s index %d is outside the valid range 0–10", directive, op.Imm.Val),
}}
}
return nil
}
+60
View File
@@ -0,0 +1,60 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import (
"os"
"path/filepath"
"runtime"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGoRuntimeCorpus parses and lints every runtime .s file the local Go
// toolchain ships for all four supported architectures. This is the
// real-world regression net: it exercises the full breadth of each
// architecture's syntax (macros, addressing modes, branch aliases) against
// production assembly. It is skipped when the toolchain source is absent.
//
// The bar is zero parse errors and zero error-severity diagnostics — i.e. no
// false "unknown instruction" / "undefined label" findings on code the real
// assembler accepts. Advisory warnings are reported but not fatal, since they
// are heuristics that may legitimately differ across Go versions.
func TestGoRuntimeCorpus(t *testing.T) {
dir := filepath.Join(runtime.GOROOT(), "src", "runtime")
var files []string
for _, suffix := range []string{"_amd64.s", "_arm64.s", "_riscv64.s", "_loong64.s"} {
matches, _ := filepath.Glob(filepath.Join(dir, "*"+suffix))
files = append(files, matches...)
}
if len(files) == 0 {
t.Skip("Go toolchain source (src/runtime/*.s) not present")
}
warnings := 0
for _, path := range files {
src, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read %s: %v", path, err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Errorf("parse %s: %v", filepath.Base(path), errs)
continue
}
diags := File(f, Config{Arch: arch.FromFilename(path)})
for _, d := range diags {
if d.Severity == Error {
t.Errorf("%s:%d: error %s: %s", filepath.Base(path), d.Pos.Line, d.Code, d.Message)
} else {
warnings++
}
}
}
if warnings > 0 {
t.Logf("%d advisory warnings across %d files (non-fatal)", warnings, len(files))
}
}
+510
View File
@@ -0,0 +1,510 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package lint runs static checks over a parsed GAsm file. The rules are
// deliberately conservative: where a check cannot be certain (for example an
// instruction whose operand count varies), it stays silent rather than emit a
// false positive. Every diagnostic carries a stable rule code so callers can
// disable individual rules.
package lint
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Severity ranks a diagnostic.
type Severity int
// Diagnostic severities, mirroring the language-server protocol ordering.
const (
Error Severity = iota
Warning
Information
Hint
)
// String returns a lower-case label for the severity.
func (s Severity) String() string {
switch s {
case Error:
return "error"
case Warning:
return "warning"
case Information:
return "information"
default:
return "hint"
}
}
// Diagnostic is one lint finding.
type Diagnostic struct {
Pos token.Position
End token.Position
Severity Severity
Code string
Message string
}
// Config controls a lint run.
type Config struct {
// Arch is the target architecture. When it is arch.Unknown the
// architecture-specific rules (unknown instruction, operand count) are
// skipped because no instruction table can be selected.
Arch arch.Arch
// Disable lists rule codes to suppress.
Disable map[string]bool
}
// Rule codes.
const (
CodeUnknownInstr = "unknown-instruction"
CodeOperandCount = "operand-count"
CodeUndefinedLabel = "undefined-label"
CodeDuplicateLabel = "duplicate-label"
CodeMissingRet = "missing-ret"
CodeMissingTextflag = "missing-textflag-include"
CodeUnreachable = "unreachable-code"
CodeABIArgSize = "abi-argsize"
CodeNosplitFrame = "nosplit-frame"
CodeRegisterClobber = "register-clobber"
CodeFuncdata = "funcdata-pcdata"
)
// pseudoOps are assembler pseudo-operations that are valid instruction-position
// tokens but are not machine instructions and so absent from the arch tables.
var pseudoOps = map[string]bool{
"BYTE": true, "WORD": true, "LONG": true, "QUAD": true, "FLOAT": true,
"PCALIGN": true, "FUNCDATA": true, "PCDATA": true, "GO_ARGS": true,
}
// File lints a parsed file and returns the diagnostics in source order.
func File(f *ast.File, cfg Config) []Diagnostic {
var out []Diagnostic
tab := arch.ForArch(cfg.Arch)
archKnown := cfg.Arch != arch.Unknown
hasTextflag := false
usesFlags := false
var firstFlagPos token.Position
// Macros (in-file #define, or any #include other than textflag.h, which
// only defines flag constants) make label resolution unreliable.
macrosInPlay := len(f.Macros) > 0
// Preprocessor conditionals (#ifdef …) make control-flow analysis
// unreliable, since mutually exclusive branches look sequential.
hasConditionals := false
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Include:
if !strings.Contains(dd.Header.Text, "textflag.h") {
macrosInPlay = true
}
case *ast.Preproc:
if isConditionalDirective(dd.Raw) {
hasConditionals = true
}
}
}
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Include:
if strings.Contains(dd.Header.Text, "textflag.h") {
hasTextflag = true
}
case *ast.Text:
out = append(out, lintText(dd, tab, archKnown, cfg, f.Macros, !macrosInPlay, !hasConditionals)...)
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
usesFlags = true
firstFlagPos = dd.Pos()
}
case *ast.Globl:
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
usesFlags = true
firstFlagPos = dd.Pos()
}
}
}
if !cfg.Disable[CodeMissingTextflag] && usesFlags && !hasTextflag {
out = append(out, Diagnostic{
Pos: firstFlagPos,
Severity: Warning,
Code: CodeMissingTextflag,
Message: "TEXT/GLOBL flags are used but textflag.h is not #included",
})
}
sortDiagnostics(out)
return out
}
// lintText lints one TEXT function body. doLabelChecks is false for files
// that use macros (an in-file #define or a non-textflag #include): without a
// preprocessor we cannot resolve labels that macros define or reference, so the
// label and RET heuristics are suppressed there to avoid false positives.
func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros map[string]bool, doLabelChecks bool, doUnreachable bool) []Diagnostic {
var out []Diagnostic
defined := map[string]token.Position{}
referenced := map[string]token.Position{}
hasRet := false
lastTerminal := false
hasMacro := false
instrCount := 0
dead := false // inside a region unreachable from above
reportedDead := false // the current dead region has already been reported
hasPCRel := referencesPC(t) // PC-relative jumps defeat reachability analysis
hasIndirect := hasIndirectBranch(t) // register-indirect branches do too
// Unreachable-code analysis is only sound in functions whose control flow is
// fully label-resolvable: no PC-relative jumps, no register-indirect
// branches, and (file-level) no preprocessor conditionals.
analyzable := doUnreachable && !hasPCRel && !hasIndirect
for _, s := range t.Body {
switch st := s.(type) {
case *ast.Label:
name := st.Name.Text
if prev, dup := defined[name]; dup {
if !cfg.Disable[CodeDuplicateLabel] {
out = append(out, Diagnostic{
Pos: st.Name.Pos,
End: st.Name.End,
Severity: Error,
Code: CodeDuplicateLabel,
Message: fmt.Sprintf("label %q already defined at %s", name, prev),
})
}
} else {
defined[name] = st.Name.Pos
}
// A label is a jump target: code after it is reachable again.
dead = false
reportedDead = false
case *ast.Instr:
instrCount++
mnem := st.Mnemonic.Text
upper := strings.ToUpper(mnem)
// Unreachable code: a real instruction following a RET/UNDEF and
// before any label, in a function whose control flow is fully
// resolvable. Only RET/UNDEF are treated as terminators here — an
// unconditional jump may be one entry of a hand-arranged branch
// table (e.g. the generated callback tables), so it is not assumed
// to make the following code dead. Pseudo-ops and macro invocations
// are never flagged.
if analyzable && dead && !reportedDead && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) &&
!cfg.Disable[CodeUnreachable] {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Warning,
Code: CodeUnreachable,
Message: "unreachable code after terminating instruction",
})
reportedDead = true
}
// RET never falls through. (UNDEF is a trap/marker rather than a
// control-flow terminator: code placed after it is occasionally
// deliberate metadata, so it is not treated as making the following
// code dead.)
if upper == "RET" {
dead = true
}
// A function need not RET if it ends in an unconditional jump (tail
// call / loop) or in UNDEF (a deliberate trap that never returns).
lastTerminal = isUnconditionalJump(cfg.Arch, upper) || upper == "UNDEF"
if isMacroInvocation(mnem, macros) {
hasMacro = true
}
if upper == "RET" {
hasRet = true
}
if archKnown && !cfg.Disable[CodeUnknownInstr] && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) {
if _, ok := tab.Lookup(mnem); !ok {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Error,
Code: CodeUnknownInstr,
Message: fmt.Sprintf("unknown %s instruction %q", cfg.Arch, mnem),
})
}
}
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) {
if in, ok := tab.Lookup(mnem); ok && in.MinOps >= 0 {
n := len(st.Operands)
if n < in.MinOps || n > in.MaxOps {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Warning,
Code: CodeOperandCount,
Message: fmt.Sprintf("%s expects %s, got %d operand(s)",
mnem, countRange(in.MinOps, in.MaxOps), n),
})
}
}
}
if isJump(cfg.Arch, upper) {
for _, op := range st.Operands {
if name, pos, ok := localLabelRef(op); ok && !tab.IsRegister(name) && !arch.IsPseudoReg(name) {
referenced[name] = pos
}
}
}
}
}
// Undefined labels.
if doLabelChecks && !cfg.Disable[CodeUndefinedLabel] {
for name, pos := range referenced {
if _, ok := defined[name]; !ok {
out = append(out, Diagnostic{
Pos: pos,
Severity: Error,
Code: CodeUndefinedLabel,
Message: fmt.Sprintf("jump to undefined label %q", name),
})
}
}
}
// Missing RET heuristic. Functions that invoke a macro are skipped: the
// macro body (opaque to us) may supply the RET.
if doLabelChecks && !cfg.Disable[CodeMissingRet] && instrCount > 0 && !hasRet && !lastTerminal && !hasMacro {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeMissingRet,
Message: fmt.Sprintf("function %q has no RET", t.Name.Name),
})
}
// ABI conformance: the argument area declared in the TEXT directive should
// match the size computed from the // func signature in the doc comment.
// Only applies to stack-argument (ABI0) functions, which reference their
// arguments through FP; register-ABI functions declare a zero arg area. Also
// skipped when there is no parseable signature or it uses an unknown type.
if !cfg.Disable[CodeABIArgSize] {
got := int64(0)
if t.Args != nil && t.Args.Imm.HasVal {
got = t.Args.Imm.Val
}
// Only meaningful for stack-argument (ABI0) functions: a non-zero
// declared arg area that is actually addressed through FP.
if got > 0 && usesFPArgs(t) {
if want, ok := abiExpectedArgSize(t.Doc); ok {
if want != got {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeABIArgSize,
Message: fmt.Sprintf("TEXT declares arg size %d but the // func signature implies %d", got, want),
})
}
}
}
}
// Register liveness: a callee-saved register that is written but never
// saved and restored is clobbered across the call. The check runs over the
// control-flow graph and is skipped for macro-using files, where an opaque
// macro may perform the save/restore.
if doLabelChecks && archKnown && !cfg.Disable[CodeRegisterClobber] {
live := analyzeLiveness(t, cfg.Arch)
if clobbered := clobberedCalleeSaved(live, cfg.Arch); len(clobbered) > 0 {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeRegisterClobber,
Message: fmt.Sprintf("callee-saved register(s) %s written but never saved/restored", strings.Join(clobbered, ", ")),
})
}
}
// FUNCDATA / PCDATA structural validation.
out = append(out, checkFuncdata(t, cfg)...)
return out
}
// usesFPArgs reports whether a function references its arguments through the FP
// pseudo-register — i.e. it uses the stack-based ABI0 layout, where the
// declared argument size must match the signature.
func usesFPArgs(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
for _, op := range in.Operands {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
return true
}
}
}
return false
}
// referencesPC reports whether a function uses a PC-relative operand (e.g.
// `JMP 2(PC)`). Such jumps target a computed offset rather than a label, so
// reachability cannot be determined statically and the unreachable-code check
// is suppressed for the whole function.
func referencesPC(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
for _, op := range in.Operands {
if strings.Contains(strings.ReplaceAll(op.Raw, " ", ""), "(PC)") {
return true
}
}
}
return false
}
// hasIndirectBranch reports whether a function transfers control through a
// register (JALR/JR/JIRL/BR/BLR). Such targets are computed at runtime, so
// reachability cannot be determined statically and the unreachable-code check is
// suppressed for the whole function.
func hasIndirectBranch(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "JALR", "JR", "JIRL", "BR", "BLR":
return true
}
}
return false
}
// isMacroInvocation reports whether a mnemonic is a macro invocation rather
// than a machine instruction. No Plan 9 mnemonic contains an underscore, so an
// underscore is a reliable macro marker (the runtime headers define macros such
// as get_tls and NO_LOCAL_POINTERS). Names introduced by an in-file #define
// are recognised too (CALLFN, DISPATCH, …). Full macro expansion is out of
// scope; this only keeps the linter quiet on invocations it cannot expand.
func isMacroInvocation(mnem string, macros map[string]bool) bool {
return strings.Contains(mnem, "_") || macros[mnem]
}
// isConditionalDirective reports whether a preprocessor directive (the text
// after '#') is a conditional-compilation directive whose branches the parser
// cannot resolve.
func isConditionalDirective(raw string) bool {
fields := strings.Fields(raw)
if len(fields) == 0 {
return false
}
switch fields[0] {
case "if", "ifdef", "ifndef", "else", "elif", "endif":
return true
}
return false
}
// localLabelRef returns the name and position of a bare local-label reference
// operand (no pseudo-register, no memory base), if op is one.
func localLabelRef(op *ast.Operand) (string, token.Position, bool) {
if op == nil || op.Kind != ast.OpAddr || op.Addr.Sym == nil {
return "", token.Position{}, false
}
sym := op.Addr.Sym
if sym.Pseudo != "" || op.Addr.Base != "" || sym.Name == "" {
return "", token.Position{}, false
}
return sym.Name, op.Pos, true
}
// riscvBranches and loong64Branches are the conditional-branch mnemonics; they
// are listed explicitly rather than matched by a "B" prefix so that bit-manip
// instructions (BCLR, BSET, …) are never mistaken for branches.
var riscvBranches = map[string]bool{
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
"BEQZ": true, "BNEZ": true, "BLEZ": true, "BGEZ": true, "BLTZ": true, "BGTZ": true,
}
var loong64Branches = map[string]bool{
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
"BLEZ": true, "BLTZ": true, "BGEZ": true, "BGTZ": true,
}
// isJump reports whether the mnemonic is any branch.
func isJump(a arch.Arch, upper string) bool {
switch a {
case arch.ARM64:
return upper == "CALL" || upper == "BR" || upper == "BLR" || upper == "JMP" ||
strings.HasPrefix(upper, "B") ||
strings.HasPrefix(upper, "CBZ") || strings.HasPrefix(upper, "CBNZ") ||
strings.HasPrefix(upper, "TBZ") || strings.HasPrefix(upper, "TBNZ")
case arch.RISCV:
return upper == "CALL" || riscvBranches[upper] ||
upper == "JMP" || upper == "J" || upper == "JAL" || upper == "JALR" ||
upper == "JR" || upper == "BR"
case arch.LOONG64:
return upper == "CALL" || loong64Branches[upper] ||
upper == "JIRL" || upper == "JMP" || upper == "BR"
default: // amd64
return upper == "CALL" || strings.HasPrefix(upper, "J")
}
}
// isUnconditionalJump reports whether the mnemonic is an unconditional branch
// (used to suppress the missing-RET heuristic for tail calls and loops).
func isUnconditionalJump(a arch.Arch, upper string) bool {
switch a {
case arch.ARM64:
return upper == "B" || upper == "BR" || upper == "JMP"
case arch.RISCV:
return upper == "JMP" || upper == "J" || upper == "JAL" ||
upper == "JALR" || upper == "JR" || upper == "BR"
case arch.LOONG64:
return upper == "JMP" || upper == "JIRL" || upper == "BR"
default:
return upper == "JMP"
}
}
func countRange(min, max int) string {
if min == max {
return fmt.Sprintf("%d operand(s)", min)
}
return fmt.Sprintf("%d–%d operands", min, max)
}
// sortDiagnostics orders diagnostics by line, then column, then code.
func sortDiagnostics(d []Diagnostic) {
for i := 1; i < len(d); i++ {
for j := i; j > 0 && lessDiag(d[j], d[j-1]); j-- {
d[j], d[j-1] = d[j-1], d[j]
}
}
}
func lessDiag(a, b Diagnostic) bool {
if a.Pos.Line != b.Pos.Line {
return a.Pos.Line < b.Pos.Line
}
if a.Pos.Column != b.Pos.Column {
return a.Pos.Column < b.Pos.Column
}
return a.Code < b.Code
}
+242
View File
@@ -0,0 +1,242 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import (
"os"
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
func lintSrc(t *testing.T, src string) []Diagnostic {
t.Helper()
f, errs := parser.Parse("test_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
return File(f, Config{Arch: arch.AMD64})
}
// lintSrcArch lints src under the architecture inferred from filename.
func lintSrcArch(t *testing.T, filename, src string) []Diagnostic {
t.Helper()
f, errs := parser.Parse(filename, src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
return File(f, Config{Arch: arch.FromFilename(filename)})
}
func codes(diags []Diagnostic) map[string]int {
m := map[string]int{}
for _, d := range diags {
m[d.Code]++
}
return m
}
func TestFixtureIsClean(t *testing.T) {
src, err := os.ReadFile("../testdata/sample_amd64.s")
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse("sample_amd64.s", string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
// The fixture mirrors the go-flac kernels, which use callee-saved registers
// (BX, R13) without saving them; the register-clobber audit flags that by
// design. This test targets the other rules, so the audit is disabled here
// (it is covered by TestRegisterClobber).
diags := File(f, Config{Arch: arch.AMD64, Disable: map[string]bool{CodeRegisterClobber: true}})
if len(diags) != 0 {
t.Fatalf("expected no diagnostics on the fixture, got %+v", diags)
}
}
func TestUnknownInstruction(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
FOOBAR AX, BX
RET
`)
if codes(diags)[CodeUnknownInstr] != 1 {
t.Fatalf("want one unknown-instruction, got %+v", diags)
}
}
func TestUndefinedLabel(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
JMP nowhere
RET
`)
if codes(diags)[CodeUndefinedLabel] != 1 {
t.Fatalf("want one undefined-label, got %+v", diags)
}
}
func TestDuplicateLabel(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
loop:
ADDQ $1, AX
loop:
SUBQ $1, AX
JMP loop
RET
`)
if codes(diags)[CodeDuplicateLabel] != 1 {
t.Fatalf("want one duplicate-label, got %+v", diags)
}
}
func TestMissingRet(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
ADDQ $1, AX
`)
if codes(diags)[CodeMissingRet] != 1 {
t.Fatalf("want one missing-ret, got %+v", diags)
}
}
func TestOperandCount(t *testing.T) {
// RET takes zero operands; JMP takes exactly one.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
RET AX
JMP
RET
`)
c := codes(diags)
if c[CodeOperandCount] != 2 {
t.Fatalf("want two operand-count findings, got %+v", diags)
}
}
func TestMissingTextflag(t *testing.T) {
diags := lintSrc(t, `
TEXT ·f(SB), NOSPLIT, $0
RET
`)
if codes(diags)[CodeMissingTextflag] != 1 {
t.Fatalf("want one missing-textflag-include, got %+v", diags)
}
}
func TestDisableRule(t *testing.T) {
f, _ := parser.Parse("t_amd64.s", `
TEXT ·f(SB), NOSPLIT, $0
RET
`)
diags := File(f, Config{Arch: arch.AMD64, Disable: map[string]bool{CodeMissingTextflag: true}})
if len(diags) != 0 {
t.Fatalf("disabling the rule should silence it, got %+v", diags)
}
}
func TestMacroInvocationSkipped(t *testing.T) {
// DISPATCH is defined in-file; get_tls carries an underscore. Neither is a
// machine instruction, so both must be ignored by the unknown-instruction
// rule rather than flagged.
diags := lintSrc(t, `
#include "textflag.h"
#define DISPATCH CALL ·x(SB)
TEXT ·f(SB), NOSPLIT, $0
DISPATCH
get_tls CX
RET
`)
if codes(diags)[CodeUnknownInstr] != 0 {
t.Fatalf("macro invocations must not be flagged: %+v", diags)
}
}
func TestUndefIsTerminal(t *testing.T) {
// A function whose body is UNDEF traps and never returns; it needs no RET.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
UNDEF
`)
if codes(diags)[CodeMissingRet] != 0 {
t.Fatalf("UNDEF should count as terminal: %+v", diags)
}
}
func TestArm64BranchAlias(t *testing.T) {
diags := lintSrcArch(t, "f_arm64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
B done
done:
RET
`)
if len(diags) != 0 {
t.Fatalf("arm64 B to a defined label should be clean: %+v", diags)
}
}
func TestArm64AddressingSuffix(t *testing.T) {
// .W (pre-index) and .P (post-index) suffixes must resolve to the base
// instruction.
diags := lintSrcArch(t, "f_arm64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
LDP.W (R0), (R1, R2)
VST1.P (R3), (R4)
RET
`)
if codes(diags)[CodeUnknownInstr] != 0 {
t.Fatalf("suffixed load/store should be recognised: %+v", diags)
}
}
func TestMacrosInPlaySuppressesLabelRules(t *testing.T) {
// Including a non-textflag header means macros may define labels and supply
// the RET, so undefined-label and missing-ret are suppressed.
diags := lintSrcArch(t, "f_arm64.s", `
#include "go_asm.h"
TEXT ·f(SB), NOSPLIT, $0
JMP RARG0
`)
if codes(diags)[CodeUndefinedLabel] != 0 || codes(diags)[CodeMissingRet] != 0 {
t.Fatalf("label rules should be suppressed in macro files: %+v", diags)
}
}
// TestRealGoLibrariesHasNoErrors asserts that the production go-flac kernels
// lint free of errors. Skipped when the sibling repository is absent.
func TestRealGoLibrariesHasNoErrors(t *testing.T) {
matches, _ := filepath.Glob("../../go-libraries/go-*/*.s")
if len(matches) == 0 {
t.Skip("go-libraries repository not present")
}
for _, path := range matches {
src, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse %s: %v", path, errs)
}
a := arch.FromFilename(path)
diags := File(f, Config{Arch: a})
for _, d := range diags {
if d.Severity == Error {
t.Errorf("%s: %s %s: %s", filepath.Base(path), d.Pos, d.Code, d.Message)
}
}
}
}
+455
View File
@@ -0,0 +1,455 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import (
"fmt"
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// This file implements register liveness by dataflow over a function's
// control-flow graph, and the checks built on it. The def/use model is
// deliberately conservative: where an instruction's effect is uncertain it is
// treated as both a use and a def of its register operands, which can only
// suppress a finding, never invent one.
// regEffect is the register-level effect of one instruction.
type regEffect struct {
def []string // registers written (killed)
use []string // registers read
saveGPR []string // callee-saved-style: register written to the stack
restGPR []string // register restored from the stack
}
// analyzeLiveness builds the control-flow graph of a function and computes
// live-in/live-out register sets by iterative backward dataflow.
type liveness struct {
blocks []*block
liveIn []map[string]bool
}
type block struct {
label string // label that begins this block, if any
instrs []*ast.Instr
succ []int // successor block indices
}
func analyzeLiveness(t *ast.Text, a arch.Arch) *liveness {
l := &liveness{}
l.buildCFG(t)
l.dataflow(a)
return l
}
// buildCFG splits the function body into basic blocks and wires up successors.
func (l *liveness) buildCFG(t *ast.Text) {
labelToBlock := map[string]int{}
var cur *block
flush := func() {
if cur != nil && len(cur.instrs) > 0 {
l.blocks = append(l.blocks, cur)
}
cur = nil
}
startBlock := func(lbl string) {
flush()
cur = &block{label: lbl}
}
startBlock("")
for _, s := range t.Body {
switch st := s.(type) {
case *ast.Label:
// A label begins a new block and is a jump target.
startBlock(st.Name.Text)
labelToBlock[st.Name.Text] = len(l.blocks) // index once flushed
case *ast.Instr:
if cur == nil {
startBlock("")
}
cur.instrs = append(cur.instrs, st)
if terminates(st) || isConditionalBranch(st) {
startBlock("")
}
}
}
flush()
// Fix up label->block indices (labels were recorded before the following
// block was appended) and build successor edges.
for i, b := range l.blocks {
if b.label != "" {
labelToBlock[b.label] = i
}
}
for i, b := range l.blocks {
if len(b.instrs) == 0 {
if i+1 < len(l.blocks) {
b.succ = append(b.succ, i+1)
}
continue
}
last := b.instrs[len(b.instrs)-1]
mnem := strings.ToUpper(last.Mnemonic.Text)
switch {
case mnem == "RET" || mnem == "UNDEF":
// No successors.
case isConditionalBranch(last):
if tgt, ok := branchTarget(last); ok {
if j, found := labelToBlock[tgt]; found {
b.succ = append(b.succ, j)
}
}
if i+1 < len(l.blocks) {
b.succ = append(b.succ, i+1) // fall-through
}
case isUnconditionalBranchAny(mnem):
if tgt, ok := branchTarget(last); ok {
if j, found := labelToBlock[tgt]; found {
b.succ = append(b.succ, j)
}
}
default:
if i+1 < len(l.blocks) {
b.succ = append(b.succ, i+1)
}
}
}
}
// dataflow runs the standard backward liveness iteration to a fixed point.
func (l *liveness) dataflow(a arch.Arch) {
n := len(l.blocks)
l.liveIn = make([]map[string]bool, n)
liveOut := make([]map[string]bool, n)
use := make([]map[string]bool, n)
def := make([]map[string]bool, n)
for i, b := range l.blocks {
use[i], def[i] = blockUseDef(b, a)
l.liveIn[i] = map[string]bool{}
liveOut[i] = map[string]bool{}
}
for changed := true; changed; {
changed = false
for i := n - 1; i >= 0; i-- {
out := map[string]bool{}
for _, s := range l.blocks[i].succ {
for r := range l.liveIn[s] {
out[r] = true
}
}
if !sameSet(out, liveOut[i]) {
liveOut[i] = out
changed = true
}
// in = use ∪ (out − def)
in := map[string]bool{}
for r := range use[i] {
in[r] = true
}
for r := range out {
if !def[i][r] {
in[r] = true
}
}
if !sameSet(in, l.liveIn[i]) {
l.liveIn[i] = in
changed = true
}
}
}
}
// blockUseDef computes the registers used before definition (use) and the
// registers defined (def) within a basic block.
func blockUseDef(b *block, a arch.Arch) (use, def map[string]bool) {
use = map[string]bool{}
def = map[string]bool{}
for _, in := range b.instrs {
eff := instrEffect(in, a)
for _, r := range eff.use {
if !def[r] {
use[r] = true
}
}
for _, r := range eff.def {
def[r] = true
}
}
return use, def
}
// terminates reports whether an instruction ends basic-block flow unconditionally.
func terminates(in *ast.Instr) bool {
m := strings.ToUpper(in.Mnemonic.Text)
return m == "RET" || m == "UNDEF" || isUnconditionalBranchAny(m)
}
func isConditionalBranch(in *ast.Instr) bool {
m := strings.ToUpper(in.Mnemonic.Text)
// Conditional jumps/branches, but not the unconditional ones.
if isUnconditionalBranchAny(m) || m == "RET" || m == "UNDEF" || m == "CALL" {
return false
}
return strings.HasPrefix(m, "J") || strings.HasPrefix(m, "B") ||
strings.HasPrefix(m, "CBZ") || strings.HasPrefix(m, "CBNZ") ||
strings.HasPrefix(m, "TBZ") || strings.HasPrefix(m, "TBNZ") ||
strings.HasPrefix(m, "BEQ") || strings.HasPrefix(m, "BNE")
}
// isUnconditionalBranchAny is an arch-agnostic unconditional-branch test.
func isUnconditionalBranchAny(m string) bool {
switch m {
case "JMP", "J", "JR", "B", "BR", "JIRL":
return true
}
return false
}
// branchTarget returns the local-label target of a branch, if it is one.
func branchTarget(in *ast.Instr) (string, bool) {
for _, op := range in.Operands {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
}
return "", false
}
// instrEffect returns the register-level effect of one instruction.
func instrEffect(in *ast.Instr, a arch.Arch) regEffect {
var eff regEffect
mnem := strings.ToUpper(in.Mnemonic.Text)
// PUSH/POP move a register to/from the stack.
if strings.HasPrefix(mnem, "PUSH") {
for _, op := range in.Operands {
if r := gprName(op, a); r != "" {
eff.use = append(eff.use, r)
eff.saveGPR = append(eff.saveGPR, r)
}
}
return eff
}
if strings.HasPrefix(mnem, "POP") {
for _, op := range in.Operands {
if r := gprName(op, a); r != "" {
eff.def = append(eff.def, r)
eff.restGPR = append(eff.restGPR, r)
}
}
return eff
}
compare := isCompare(mnem)
dstIdx := dstIndex(in, a)
for i, op := range in.Operands {
r := gprName(op, a)
if r != "" {
if i == dstIdx && !compare {
eff.def = append(eff.def, r)
// Arithmetic also reads its destination.
eff.use = append(eff.use, r)
} else {
eff.use = append(eff.use, r)
}
}
// Detect saves/restores through the stack frame.
if isStackAddr(op) {
// The other operand (the register) is being saved or restored.
for j, other := range in.Operands {
if j == i {
continue
}
if rr := gprName(other, a); rr != "" {
if j == dstIdx && !compare {
eff.restGPR = append(eff.restGPR, rr) // reg loaded from stack
} else {
eff.saveGPR = append(eff.saveGPR, rr) // reg stored to stack
}
}
}
}
}
return eff
}
// dstIndex returns the operand index of the destination register: last for the
// Plan 9 (amd64) spelling, first for arm64/riscv64/loong64.
func dstIndex(in *ast.Instr, a arch.Arch) int {
if a == arch.AMD64 {
return len(in.Operands) - 1
}
return 0
}
// isCompare reports whether the mnemonic only reads its operands (setting flags).
func isCompare(m string) bool {
return strings.HasPrefix(m, "CMP") || strings.HasPrefix(m, "TEST") ||
strings.HasPrefix(m, "CMN") || strings.HasPrefix(m, "TST") ||
m == "FCMP" || m == "FCMPE"
}
// gprName returns the canonical general-purpose register name of an operand, or
// "" if the operand is not a bare GPR reference.
func gprName(op *ast.Operand, a arch.Arch) string {
if op == nil || op.Kind != ast.OpAddr || op.Addr.Sym == nil {
return ""
}
if op.Addr.Base != "" || op.Addr.Sym.Pseudo != "" || op.Addr.Sym.Name == "" {
return ""
}
name := op.Addr.Sym.Name
if r, ok := arch.ForArch(a).Register(name); ok && (r.Class == arch.GPR || r.Class == arch.GPRSub) {
return canonicalGPR(name)
}
return ""
}
// canonicalGPR maps a sized sub-register to its base GPR (amd64 only).
func canonicalGPR(name string) string {
upper := strings.ToUpper(name)
// Named 8/16/32-bit forms of the classic registers.
switch upper {
case "AL", "AH", "AX":
return "AX"
case "BL", "BH", "BX":
return "BX"
case "CL", "CH", "CX":
return "CX"
case "DL", "DH", "DX":
return "DX"
case "SIL":
return "SI"
case "DIL":
return "DI"
case "BPL":
return "BP"
case "SPL":
return "SP"
}
// Numbered sub-registers R8B/R8W/R8D → R8.
if len(upper) >= 3 && upper[0] == 'R' {
switch upper[len(upper)-1] {
case 'B', 'W', 'D':
return upper[:len(upper)-1]
}
}
return upper
}
// isStackAddr reports whether an operand addresses the stack frame
// (base SP, or an FP/SP-relative symbol).
func isStackAddr(op *ast.Operand) bool {
if op == nil || op.Kind != ast.OpAddr {
return false
}
if op.Addr.Base == "SP" {
return true
}
if op.Addr.Sym != nil && (op.Addr.Sym.Pseudo == "SP" || op.Addr.Sym.Pseudo == "FP") {
return true
}
return false
}
func sameSet(a, b map[string]bool) bool {
if len(a) != len(b) {
return false
}
for k := range a {
if !b[k] {
return false
}
}
return true
}
// calleeSavedGPRs returns the general-purpose registers an assembly function
// must preserve for its caller, using the register names the assembler accepts
// for each architecture.
func calleeSavedGPRs(a arch.Arch) map[string]bool {
switch a {
case arch.AMD64:
return gprSet("BX", "BP", "R12", "R13", "R14", "R15")
case arch.ARM64:
names := []string{"R29", "R30"} // FP, LR
for i := 19; i <= 28; i++ {
names = append(names, fmt.Sprintf("R%d", i))
}
return gprSet(names...)
case arch.RISCV:
// RA (X1) and the S registers (X8, X9, X18–X27) are callee-saved.
names := []string{"X1", "RA", "X8", "X9", "S0", "S1", "FP"}
for i := 18; i <= 27; i++ {
names = append(names, fmt.Sprintf("X%d", i))
}
for i := 2; i <= 11; i++ {
names = append(names, fmt.Sprintf("S%d", i))
}
return gprSet(names...)
case arch.LOONG64:
// RA (R1), FP (R22) and S0–S8 (R23–R31) are callee-saved.
names := []string{"R1", "RA", "R22", "FP"}
for i := 23; i <= 31; i++ {
names = append(names, fmt.Sprintf("R%d", i))
}
for i := 0; i <= 8; i++ {
names = append(names, fmt.Sprintf("S%d", i))
}
return gprSet(names...)
}
return nil
}
func gprSet(names ...string) map[string]bool {
m := make(map[string]bool, len(names))
for _, n := range names {
m[n] = true
}
return m
}
// clobberedCalleeSaved returns the callee-saved registers a function writes
// without also saving and restoring them — i.e. registers whose caller-owned
// value is lost across the call. It walks the blocks of the liveness analysis
// (so the control-flow graph is what supplies the instruction set) and
// aggregates each instruction's register effects.
func clobberedCalleeSaved(l *liveness, a arch.Arch) []string {
callee := calleeSavedGPRs(a)
if len(callee) == 0 {
return nil
}
def := map[string]bool{}
saved := map[string]bool{}
restored := map[string]bool{}
for _, b := range l.blocks {
for _, in := range b.instrs {
eff := instrEffect(in, a)
for _, r := range eff.def {
def[r] = true
}
for _, r := range eff.saveGPR {
saved[r] = true
}
for _, r := range eff.restGPR {
restored[r] = true
}
}
}
var out []string
for r := range callee {
if def[r] && !(saved[r] && restored[r]) {
out = append(out, r)
}
}
sort.Strings(out)
return out
}
+79
View File
@@ -0,0 +1,79 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import "testing"
// TestRegisterClobber detects writes to callee-saved registers that are not
// saved and restored.
func TestRegisterClobber(t *testing.T) {
// BX (callee-saved on amd64) is written but never saved → clobbered.
clob := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVQ CX, BX\n"+
"\tRET\n")
if codes(clob)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved callee-saved write should be flagged: %+v", clob)
}
// Saved and restored → preserved.
saved := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $8\n"+
"\tPUSHQ BX\n"+
"\tMOVQ CX, BX\n"+
"\tPOPQ BX\n"+
"\tRET\n")
if codes(saved)[CodeRegisterClobber] != 0 {
t.Fatalf("saved/restored register must not be flagged: %+v", saved)
}
// A caller-saved register (CX) is fine to write.
caller := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVQ $1, CX\n"+
"\tRET\n")
if codes(caller)[CodeRegisterClobber] != 0 {
t.Fatalf("caller-saved register must not be flagged: %+v", caller)
}
}
// TestFuncdata validates the FUNCDATA/PCDATA structural checks.
func TestFuncdata(t *testing.T) {
// Well formed: no findings.
good := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tFUNCDATA $0, gclocals·abc(SB)\n"+
"\tPCDATA $1, $0\n"+
"\tRET\n")
if codes(good)[CodeFuncdata] != 0 {
t.Fatalf("well-formed FUNCDATA/PCDATA must not be flagged: %+v", good)
}
// FUNCDATA with one operand.
bad1 := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tFUNCDATA $0\n"+
"\tRET\n")
if codes(bad1)[CodeFuncdata] == 0 {
t.Fatal("FUNCDATA with one operand should be flagged")
}
// PCDATA with a non-immediate value.
bad2 := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tPCDATA $0, AX\n"+
"\tRET\n")
if codes(bad2)[CodeFuncdata] == 0 {
t.Fatal("PCDATA with a register value should be flagged")
}
// FUNCDATA index out of range.
bad3 := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tFUNCDATA $99, gclocals·abc(SB)\n"+
"\tRET\n")
if codes(bad3)[CodeFuncdata] == 0 {
t.Fatal("out-of-range FUNCDATA index should be flagged")
}
}
+382
View File
@@ -0,0 +1,382 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lsp
import (
"sort"
"strings"
"unicode"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// textflagMacros are the flag names defined by textflag.h; they are highlighted
// as macros and offered as completions after a TEXT/GLOBL directive.
var textflagMacros = map[string]bool{
"NOPROFILE": true, "DUPOK": true, "NOSPLIT": true, "RODATA": true,
"NOPTR": true, "WRAPPER": true, "NEEDCTXT": true, "TOPFRAME": true,
"LEAF": true, "ABI0": true, "REFLECTDATA": true,
}
// completion builds the completion list for a document.
func (s *Server) completion(p completionParams) []CompletionItem {
a := arch.ForArch(arch.FromFilename(uriPath(p.TextDocument.URI)))
tab := a
items := []CompletionItem{
{Label: "TEXT", Kind: ciKeyword, Detail: "define a function"},
{Label: "DATA", Kind: ciKeyword, Detail: "initialise a data symbol"},
{Label: "GLOBL", Kind: ciKeyword, Detail: "declare a global symbol"},
}
for name := range textflagMacros {
items = append(items, CompletionItem{Label: name, Kind: ciKeyword, Detail: "textflag.h flag"})
}
for name, desc := range map[string]string{
"FP": "frame pointer (arguments/results)", "SP": "stack pointer",
"SB": "static base (globals)", "PC": "program counter",
} {
items = append(items, CompletionItem{Label: name, Kind: ciConstant, Detail: desc})
}
for _, in := range tab.Instructions() {
items = append(items, CompletionItem{
Label: in.Name, Kind: ciFunction, Detail: in.Summary, Documentation: in.Summary,
})
}
for _, r := range tab.Registers() {
kind := ciVariable
if r.Class == arch.Vector || r.Class == arch.Mask || r.Class == arch.Float || r.Class == arch.VecARM {
kind = ciClass
}
items = append(items, CompletionItem{Label: r.Name, Kind: kind, Detail: r.Desc})
}
// Local labels defined in the document.
if f, _ := parser.Parse("", s.docs[p.TextDocument.URI]); f != nil {
for _, name := range labelNames(f) {
items = append(items, CompletionItem{Label: name, Kind: ciModule, Detail: "local label"})
}
}
sort.Slice(items, func(i, j int) bool { return items[i].Label < items[j].Label })
return items
}
// hover returns documentation for the symbol under the cursor.
func (s *Server) hover(p hoverParams) *Hover {
text := s.docs[p.TextDocument.URI]
word, rng := wordAt(text, p.Position)
if word == "" {
return nil
}
a := arch.ForArch(arch.FromFilename(uriPath(p.TextDocument.URI)))
var md string
if in, ok := a.Lookup(word); ok {
md = "**" + in.Name + "** — " + in.Summary
} else if r, ok := a.Register(word); ok {
md = "**" + r.Name + "** — " + r.Class.String() + " register. " + r.Desc
} else if desc, ok := arch.PseudoRegDesc(word); ok {
md = "**" + strings.ToUpper(word) + "** — pseudo-register. " + desc
} else if textflagMacros[strings.ToUpper(word)] {
md = "**" + strings.ToUpper(word) + "** — textflag.h flag"
} else {
return nil
}
return &Hover{
Contents: markupContent{Kind: "markdown", Value: md},
Range: rng,
}
}
// documentSymbols returns functions and their labels, plus global symbols.
func (s *Server) documentSymbols(p documentSymbolParams) []DocumentSymbol {
text := s.docs[p.TextDocument.URI]
f, _ := parser.Parse(uriPath(p.TextDocument.URI), text)
if f == nil {
return nil
}
var out []DocumentSymbol
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Text:
sym := DocumentSymbol{
Name: dd.Name.Name,
Detail: "TEXT " + strings.Join(dd.Flags, " "),
Kind: symFunction,
Range: textRange(dd),
SelectionRange: symRange(dd.Name),
}
for _, st := range dd.Body {
if l, ok := st.(*ast.Label); ok {
sym.Children = append(sym.Children, DocumentSymbol{
Name: l.Name.Text,
Kind: symVariable,
Range: tokenRange(l.Name),
SelectionRange: tokenRange(l.Name),
})
}
}
out = append(out, sym)
case *ast.Globl:
out = append(out, DocumentSymbol{
Name: dd.Name.Name, Detail: "GLOBL", Kind: symConstant,
Range: symRange(dd.Name), SelectionRange: symRange(dd.Name),
})
case *ast.Data:
out = append(out, DocumentSymbol{
Name: dd.Name.Name, Detail: "DATA", Kind: symConstant,
Range: symRange(dd.Name), SelectionRange: symRange(dd.Name),
})
}
}
return out
}
// semTok is one classified token before delta encoding.
type semTok struct {
line, char, length, typ int
}
// semanticTokens encodes syntax highlighting as LSP semantic tokens.
func (s *Server) semanticTokens(p semanticTokensParams) SemanticTokens {
text := s.docs[p.TextDocument.URI]
a := arch.ForArch(arch.FromFilename(uriPath(p.TextDocument.URI)))
f, _ := parser.Parse("", text)
labels := map[string]bool{}
for _, name := range labelNames(f) {
labels[name] = true
}
toks := lexer.Tokenize(text)
lines := groupLines(toks)
var encoded []semTok
for _, line := range lines {
encoded = append(encoded, classifyLine(line, a, labels)...)
}
return SemanticTokens{Data: deltaEncode(encoded)}
}
// classifyLine assigns a semantic token type to each significant token on a line.
func classifyLine(line []token.Token, a *arch.Table, labels map[string]bool) []semTok {
if len(line) == 0 {
return nil
}
var out []semTok
first := firstSignificant(line)
if first < 0 {
return nil
}
isDirective := line[first].Kind == token.Ident &&
(line[first].Text == "TEXT" || line[first].Text == "DATA" || line[first].Text == "GLOBL")
isLabel := line[first].Kind == token.Ident && first+1 < len(line) &&
line[first+1].Kind == token.Colon
isInstr := !isDirective && !isLabel && line[first].Kind == token.Ident
mnemonicDone := false
for i, t := range line {
typ := -1
switch t.Kind {
case token.Comment:
typ = stComment
case token.Number:
typ = stNumber
case token.String, token.Rune:
typ = stString
case token.Hash:
typ = stMacro
case token.Ident:
typ = classifyIdent(line, i, first, t.Text, a, labels,
isDirective, isLabel, isInstr, &mnemonicDone)
case token.Colon, token.Comma, token.LParen, token.RParen,
token.Plus, token.Minus, token.Star, token.Slash, token.Dollar,
token.LAngle, token.RAngle, token.LShift, token.RShift, token.Arrow, token.At:
typ = stOperator
}
if typ >= 0 {
out = append(out, semTok{
line: t.Pos.Line - 1,
char: t.Pos.Column - 1,
length: runeLen(t.Text),
typ: typ,
})
}
}
return out
}
// classifyIdent decides the semantic type of an identifier token.
func classifyIdent(line []token.Token, i, first int, text string, a *arch.Table,
labels map[string]bool, isDirective, isLabel, isInstr bool, mnemonicDone *bool) int {
upper := strings.ToUpper(text)
switch {
case isDirective && i == first:
return stKeyword
case isDirective && textflagMacros[upper]:
return stMacro
case isLabel && i == first:
return stNamespace
case arch.IsPseudoReg(text):
return stProperty
case labels[text]:
return stNamespace
}
if r, ok := a.Register(text); ok {
switch r.Class {
case arch.Vector, arch.Mask, arch.Float, arch.VecARM:
return stType
default:
return stVariable
}
}
if isInstr && i == first && !*mnemonicDone {
*mnemonicDone = true
return stFunction
}
// Argument/symbol names and anything else.
return stVariable
}
// deltaEncode converts absolute token positions to the LSP relative encoding.
func deltaEncode(toks []semTok) []int {
data := make([]int, 0, len(toks)*5)
prevLine, prevChar := 0, 0
for _, t := range toks {
dLine := t.line - prevLine
dChar := t.char
if dLine == 0 {
dChar = t.char - prevChar
}
data = append(data, dLine, dChar, t.length, t.typ, 0)
prevLine, prevChar = t.line, t.char
}
return data
}
// --- shared helpers ---------------------------------------------------------
// wordAt extracts the identifier surrounding pos and its range.
func wordAt(text string, pos Position) (string, Range) {
lines := strings.Split(text, "\n")
if pos.Line < 0 || pos.Line >= len(lines) {
return "", Range{}
}
runes := []rune(lines[pos.Line])
col := pos.Character
if col < 0 || col > len(runes) {
return "", Range{}
}
isWord := func(r rune) bool {
return r == '_' || r == '\u00B7' || unicode.IsLetter(r) || unicode.IsDigit(r)
}
start, end := col, col
for start > 0 && isWord(runes[start-1]) {
start--
}
for end < len(runes) && isWord(runes[end]) {
end++
}
if start == end {
return "", Range{}
}
rng := Range{
Start: Position{Line: pos.Line, Character: start},
End: Position{Line: pos.Line, Character: end},
}
return string(runes[start:end]), rng
}
// labelNames collects every label defined in a file.
func labelNames(f *ast.File) []string {
if f == nil {
return nil
}
seen := map[string]bool{}
var out []string
collect := func(body []ast.Stmt) {
for _, st := range body {
if l, ok := st.(*ast.Label); ok && !seen[l.Name.Text] {
seen[l.Name.Text] = true
out = append(out, l.Name.Text)
}
}
}
for _, d := range f.Decls {
if t, ok := d.(*ast.Text); ok {
collect(t.Body)
}
}
collect(f.Orphans)
return out
}
// groupLines splits a token stream into lines, keeping Newline boundaries but
// dropping the Newline and EOF tokens themselves.
func groupLines(toks []token.Token) [][]token.Token {
var lines [][]token.Token
var cur []token.Token
for _, t := range toks {
if t.Kind == token.EOF {
break
}
if t.Kind == token.Newline {
lines = append(lines, cur)
cur = nil
continue
}
cur = append(cur, t)
}
if len(cur) > 0 {
lines = append(lines, cur)
}
return lines
}
func firstSignificant(line []token.Token) int {
for i, t := range line {
if t.Kind != token.Comment {
return i
}
}
return -1
}
func runeLen(s string) int { return len([]rune(s)) }
// symRange builds a range covering a symbol from its position and raw text.
func symRange(sym *ast.Symbol) Range {
start := Position{Line: sym.Pos.Line - 1, Character: sym.Pos.Column - 1}
end := start
end.Character += runeLen(sym.Name)
return Range{Start: start, End: end}
}
// tokenRange builds a range covering one token.
func tokenRange(t token.Token) Range {
return Range{
Start: Position{Line: t.Pos.Line - 1, Character: t.Pos.Column - 1},
End: Position{Line: t.End.Line - 1, Character: t.End.Column - 1},
}
}
// textRange spans a TEXT function from its keyword to the end of its body.
func textRange(t *ast.Text) Range {
start := Position{Line: t.Keyword.Pos.Line - 1, Character: t.Keyword.Pos.Column - 1}
end := start
end.Character += runeLen(t.Keyword.Text)
if n := len(t.Body); n > 0 {
last := t.Body[n-1]
if in, ok := last.(*ast.Instr); ok {
end = Position{Line: in.Mnemonic.End.Line - 1, Character: in.Mnemonic.End.Column - 1}
} else {
lp := last.Pos()
end = Position{Line: lp.Line - 1, Character: lp.Column - 1}
}
}
return Range{Start: start, End: end}
}
+255
View File
@@ -0,0 +1,255 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package lsp implements a Language Server Protocol server for GAsm. It
// speaks JSON-RPC 2.0 over any io.Reader/io.Writer pair (normally standard
// input/output) and provides completion, hover documentation, document
// symbols, diagnostics and semantic-token highlighting — all backed by the
// pure-Go lexer, parser, arch and lint packages. It is the vendor-neutral
// integration point: any LSP-capable editor can use it with no editor-specific
// plugin code.
package lsp
import "encoding/json"
// --- JSON-RPC 2.0 -----------------------------------------------------------
// rpcMessage is the common envelope for every JSON-RPC message.
type rpcMessage struct {
JSONRPC string `json:"jsonrpc"`
ID *json.RawMessage `json:"id,omitempty"`
Method string `json:"method,omitempty"`
Params json.RawMessage `json:"params,omitempty"`
Result any `json:"result,omitempty"`
Error *rpcError `json:"error,omitempty"`
}
type rpcError struct {
Code int `json:"code"`
Message string `json:"message"`
}
// JSON-RPC error codes used by LSP.
const (
errParse = -32700
errInvalidRequest = -32600
errMethodNotFound = -32601
errInvalidParams = -32602
errInternal = -32603
)
// --- LSP positions and ranges ----------------------------------------------
// Position is a zero-based line/character position, as LSP requires.
type Position struct {
Line int `json:"line"`
Character int `json:"character"`
}
// Range is a pair of positions.
type Range struct {
Start Position `json:"start"`
End Position `json:"end"`
}
// Location is a range within a document URI.
type Location struct {
URI string `json:"uri"`
Range Range `json:"range"`
}
// --- diagnostics ------------------------------------------------------------
// Diagnostic severities (LSP ordering: 1 = error).
const (
sevError = 1
sevWarning = 2
sevInformation = 3
sevHint = 4
)
// Diagnostic is one published finding.
type Diagnostic struct {
Range Range `json:"range"`
Severity int `json:"severity"`
Code string `json:"code,omitempty"`
Source string `json:"source,omitempty"`
Message string `json:"message"`
}
type publishDiagnosticsParams struct {
URI string `json:"uri"`
Diagnostics []Diagnostic `json:"diagnostics"`
}
// --- text document synchronisation -----------------------------------------
type textDocumentItem struct {
URI string `json:"uri"`
LanguageID string `json:"languageId"`
Version int `json:"version"`
Text string `json:"text"`
}
type didOpenParams struct {
TextDocument textDocumentItem `json:"textDocument"`
}
type versionedTextDocumentIdentifier struct {
URI string `json:"uri"`
Version int `json:"version"`
}
type textDocumentIdentifier struct {
URI string `json:"uri"`
}
type contentChangeEvent struct {
Text string `json:"text"`
}
type didChangeParams struct {
TextDocument versionedTextDocumentIdentifier `json:"textDocument"`
ContentChanges []contentChangeEvent `json:"contentChanges"`
}
type didCloseParams struct {
TextDocument textDocumentIdentifier `json:"textDocument"`
}
// --- completion -------------------------------------------------------------
// Completion item kinds (a useful subset).
const (
ciFunction = 3
ciField = 5
ciVariable = 6
ciClass = 7
ciModule = 9
ciKeyword = 14
ciConstant = 21
ciStruct = 22
)
// CompletionItem is one completion suggestion.
type CompletionItem struct {
Label string `json:"label"`
Kind int `json:"kind,omitempty"`
Detail string `json:"detail,omitempty"`
Documentation string `json:"documentation,omitempty"`
InsertText string `json:"insertText,omitempty"`
}
type completionParams struct {
TextDocument textDocumentIdentifier `json:"textDocument"`
Position Position `json:"position"`
}
// --- hover ------------------------------------------------------------------
type hoverParams struct {
TextDocument textDocumentIdentifier `json:"textDocument"`
Position Position `json:"position"`
}
// Hover is the hover response.
type Hover struct {
Contents markupContent `json:"contents"`
Range Range `json:"range,omitempty"`
}
type markupContent struct {
Kind string `json:"kind"`
Value string `json:"value"`
}
// --- document symbols -------------------------------------------------------
// Symbol kinds (a useful subset).
const (
symFunction = 12
symConstant = 14
symVariable = 13
)
// DocumentSymbol is a hierarchical symbol.
type DocumentSymbol struct {
Name string `json:"name"`
Detail string `json:"detail,omitempty"`
Kind int `json:"kind"`
Range Range `json:"range"`
SelectionRange Range `json:"selectionRange"`
Children []DocumentSymbol `json:"children,omitempty"`
}
type documentSymbolParams struct {
TextDocument textDocumentIdentifier `json:"textDocument"`
}
// --- semantic tokens --------------------------------------------------------
// semanticTokenTypes is the legend of token type names, in index order. The
// indices are referenced by the encoder below.
var semanticTokenTypes = []string{
"comment", // 0
"keyword", // 1
"function", // 2
"variable", // 3
"type", // 4
"number", // 5
"string", // 6
"operator", // 7
"property", // 8
"namespace", // 9
"macro", // 10
}
const (
stComment = 0
stKeyword = 1
stFunction = 2
stVariable = 3
stType = 4
stNumber = 5
stString = 6
stOperator = 7
stProperty = 8
stNamespace = 9
stMacro = 10
)
// SemanticTokensLegend advertises the token classification.
type SemanticTokensLegend struct {
TokenTypes []string `json:"tokenTypes"`
TokenModifiers []string `json:"tokenModifiers"`
}
// SemanticTokens is the encoded token payload.
type SemanticTokens struct {
Data []int `json:"data"`
}
type semanticTokensParams struct {
TextDocument textDocumentIdentifier `json:"textDocument"`
}
// --- initialize -------------------------------------------------------------
type initializeParams struct {
RootURI string `json:"rootUri"`
}
// ServerCapabilities advertises what this server provides.
type ServerCapabilities struct {
TextDocumentSync int `json:"textDocumentSync"`
CompletionProvider map[string]any `json:"completionProvider,omitempty"`
HoverProvider bool `json:"hoverProvider,omitempty"`
DocumentSymbolProvider bool `json:"documentSymbolProvider,omitempty"`
SemanticTokensProvider map[string]any `json:"semanticTokensProvider,omitempty"`
DiagnosticProvider map[string]any `json:"diagnosticProvider,omitempty"`
}
type initializeResult struct {
Capabilities ServerCapabilities `json:"capabilities"`
ServerInfo map[string]string `json:"serverInfo,omitempty"`
}
+249
View File
@@ -0,0 +1,249 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lsp
import (
"bufio"
"encoding/json"
"fmt"
"io"
"strconv"
"strings"
"sync"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Server is a GAsm language server bound to a byte stream.
type Server struct {
in *bufio.Reader
out io.Writer
mu sync.Mutex // guards writes to out
docs map[string]string
}
// New returns a server reading from in and writing to out.
func New(in io.Reader, out io.Writer) *Server {
return &Server{
in: bufio.NewReader(in),
out: out,
docs: make(map[string]string),
}
}
// Run serves requests until the input is exhausted or an exit is requested.
func (s *Server) Run() error {
for {
msg, err := s.read()
if err == io.EOF {
return nil
}
if err != nil {
return err
}
if exit := s.dispatch(msg); exit {
return nil
}
}
}
// read parses one Content-Length framed JSON-RPC message.
func (s *Server) read() (*rpcMessage, error) {
length := -1
for {
line, err := s.in.ReadString('\n')
if err != nil {
return nil, err
}
line = strings.TrimRight(line, "\r\n")
if line == "" {
break
}
if k, v, ok := strings.Cut(line, ":"); ok && strings.EqualFold(strings.TrimSpace(k), "Content-Length") {
length, _ = strconv.Atoi(strings.TrimSpace(v))
}
}
if length < 0 {
return nil, fmt.Errorf("missing Content-Length header")
}
body := make([]byte, length)
if _, err := io.ReadFull(s.in, body); err != nil {
return nil, err
}
var msg rpcMessage
if err := json.Unmarshal(body, &msg); err != nil {
return nil, err
}
return &msg, nil
}
// send marshals and writes one framed message.
func (s *Server) send(msg *rpcMessage) error {
msg.JSONRPC = "2.0"
body, err := json.Marshal(msg)
if err != nil {
return err
}
s.mu.Lock()
defer s.mu.Unlock()
if _, err := fmt.Fprintf(s.out, "Content-Length: %d\r\n\r\n", len(body)); err != nil {
return err
}
_, err = s.out.Write(body)
return err
}
func (s *Server) respond(id *json.RawMessage, result any) {
_ = s.send(&rpcMessage{ID: id, Result: result})
}
func (s *Server) respondError(id *json.RawMessage, code int, msg string) {
_ = s.send(&rpcMessage{ID: id, Error: &rpcError{Code: code, Message: msg}})
}
func (s *Server) notify(method string, params any) {
raw, _ := json.Marshal(params)
_ = s.send(&rpcMessage{Method: method, Params: raw})
}
// dispatch routes one message. It returns true when the server should stop.
func (s *Server) dispatch(msg *rpcMessage) (exit bool) {
switch msg.Method {
case "initialize":
s.respond(msg.ID, initializeResult{
Capabilities: ServerCapabilities{
TextDocumentSync: 1, // full sync
CompletionProvider: map[string]any{},
HoverProvider: true,
DocumentSymbolProvider: true,
SemanticTokensProvider: map[string]any{
"legend": SemanticTokensLegend{
TokenTypes: semanticTokenTypes,
TokenModifiers: []string{},
},
"full": true,
},
},
ServerInfo: map[string]string{"name": "gasm", "version": "0.1.0"},
})
case "initialized", "textDocument/didSave":
// Notifications with nothing to do.
case "textDocument/didOpen":
var p didOpenParams
if json.Unmarshal(msg.Params, &p) == nil {
s.docs[p.TextDocument.URI] = p.TextDocument.Text
s.publish(p.TextDocument.URI)
}
case "textDocument/didChange":
var p didChangeParams
if json.Unmarshal(msg.Params, &p) == nil && len(p.ContentChanges) > 0 {
// Full sync: the last change carries the whole document.
text := p.ContentChanges[len(p.ContentChanges)-1].Text
s.docs[p.TextDocument.URI] = text
s.publish(p.TextDocument.URI)
}
case "textDocument/didClose":
var p didCloseParams
if json.Unmarshal(msg.Params, &p) == nil {
delete(s.docs, p.TextDocument.URI)
// Clear diagnostics for the closed document.
s.notify("textDocument/publishDiagnostics", publishDiagnosticsParams{
URI: p.TextDocument.URI, Diagnostics: []Diagnostic{},
})
}
case "textDocument/completion":
var p completionParams
json.Unmarshal(msg.Params, &p)
s.respond(msg.ID, s.completion(p))
case "textDocument/hover":
var p hoverParams
json.Unmarshal(msg.Params, &p)
s.respond(msg.ID, s.hover(p))
case "textDocument/documentSymbol":
var p documentSymbolParams
json.Unmarshal(msg.Params, &p)
s.respond(msg.ID, s.documentSymbols(p))
case "textDocument/semanticTokens/full":
var p semanticTokensParams
json.Unmarshal(msg.Params, &p)
s.respond(msg.ID, s.semanticTokens(p))
case "shutdown":
s.respond(msg.ID, nil)
case "exit":
return true
default:
if msg.ID != nil {
s.respondError(msg.ID, errMethodNotFound, "method not supported: "+msg.Method)
}
}
return false
}
// publish parses and lints a document and pushes the diagnostics to the client.
func (s *Server) publish(uri string) {
text := s.docs[uri]
f, _ := parser.Parse(uri, text)
cfg := lint.Config{Arch: arch.FromFilename(uriPath(uri))}
diags := lint.File(f, cfg)
out := make([]Diagnostic, 0, len(diags))
for _, d := range diags {
out = append(out, Diagnostic{
Range: toRange(d.Pos.Line, d.Pos.Column, d.End),
Severity: lintSeverity(d.Severity),
Code: d.Code,
Source: "gasm",
Message: d.Message,
})
}
s.notify("textDocument/publishDiagnostics", publishDiagnosticsParams{URI: uri, Diagnostics: out})
}
// toRange converts one-based line/column plus an optional end position into an
// LSP range (zero-based).
func toRange(line, col int, end token.Position) Range {
start := Position{Line: line - 1, Character: col - 1}
finish := start
if end.IsValid() {
finish = Position{Line: end.Line - 1, Character: end.Column - 1}
} else {
finish.Character = start.Character + 1
}
return Range{Start: start, End: finish}
}
func lintSeverity(s lint.Severity) int {
switch s {
case lint.Error:
return sevError
case lint.Warning:
return sevWarning
case lint.Information:
return sevInformation
default:
return sevHint
}
}
// uriPath strips a file:// scheme and returns the path component.
func uriPath(uri string) string {
if rest, ok := strings.CutPrefix(uri, "file://"); ok {
return rest
}
return uri
}
+294
View File
@@ -0,0 +1,294 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lsp
import (
"bufio"
"bytes"
"encoding/json"
"fmt"
"io"
"strconv"
"strings"
"testing"
)
const cleanDoc = "#include \"textflag.h\"\n" +
"TEXT ·foo(SB), NOSPLIT, $0\n" +
"\tMOVQ AX, CX\n" +
"loop:\n" +
"\tJMP loop\n" +
"\tRET\n"
const badDoc = "#include \"textflag.h\"\n" +
"TEXT ·foo(SB), NOSPLIT, $0\n" +
"\tNOSUCHINSTR AX, BX\n" +
"\tJMP missing\n"
// frame renders one Content-Length framed JSON-RPC message.
func frame(id any, method string, params any) string {
msg := map[string]any{"jsonrpc": "2.0"}
if id != nil {
msg["id"] = id
}
if method != "" {
msg["method"] = method
}
if params != nil {
msg["params"] = params
}
body, _ := json.Marshal(msg)
return fmt.Sprintf("Content-Length: %d\r\n\r\n%s", len(body), body)
}
// run feeds input to a server and returns every parsed output message.
func run(t *testing.T, input string) []rpcMessage {
t.Helper()
var out bytes.Buffer
srv := New(strings.NewReader(input), &out)
if err := srv.Run(); err != nil {
t.Fatalf("server run: %v", err)
}
return readFrames(t, &out)
}
// readFrames parses all framed messages from a buffer.
func readFrames(t *testing.T, r io.Reader) []rpcMessage {
t.Helper()
br := bufio.NewReader(r)
var msgs []rpcMessage
for {
length := -1
for {
line, err := br.ReadString('\n')
if err == io.EOF {
return msgs
}
if err != nil {
t.Fatalf("read header: %v", err)
}
line = strings.TrimRight(line, "\r\n")
if line == "" {
break
}
if k, v, ok := strings.Cut(line, ":"); ok && strings.EqualFold(strings.TrimSpace(k), "Content-Length") {
length, _ = strconv.Atoi(strings.TrimSpace(v))
}
}
if length < 0 {
return msgs
}
body := make([]byte, length)
if _, err := io.ReadFull(br, body); err != nil {
t.Fatalf("read body: %v", err)
}
var m rpcMessage
if err := json.Unmarshal(body, &m); err != nil {
t.Fatalf("unmarshal: %v", err)
}
msgs = append(msgs, m)
}
}
// session builds a standard scripting of messages around a document.
func session(uri, text string) string {
var b strings.Builder
b.WriteString(frame(1, "initialize", map[string]any{"rootUri": ""}))
b.WriteString(frame(nil, "initialized", map[string]any{}))
b.WriteString(frame(nil, "textDocument/didOpen", map[string]any{
"textDocument": map[string]any{
"uri": uri, "languageId": "gasm", "version": 1, "text": text,
},
}))
return b.String()
}
func findByID(msgs []rpcMessage, n int) *rpcMessage {
for i := range msgs {
if msgs[i].ID != nil {
var id int
if json.Unmarshal(*msgs[i].ID, &id) == nil && id == n {
return &msgs[i]
}
}
}
return nil
}
func findMethod(msgs []rpcMessage, method string) *rpcMessage {
for i := range msgs {
if msgs[i].Method == method {
return &msgs[i]
}
}
return nil
}
func TestInitialize(t *testing.T) {
msgs := run(t, session("file:///f_amd64.s", cleanDoc)+frame(nil, "exit", nil))
resp := findByID(msgs, 1)
if resp == nil {
t.Fatal("no initialize response")
}
var res initializeResult
if err := json.Unmarshal(mustResult(t, resp), &res); err != nil {
t.Fatal(err)
}
if !res.Capabilities.HoverProvider || res.Capabilities.SemanticTokensProvider == nil {
t.Fatalf("unexpected capabilities: %+v", res.Capabilities)
}
}
func TestDiagnosticsClean(t *testing.T) {
msgs := run(t, session("file:///f_amd64.s", cleanDoc)+frame(nil, "exit", nil))
pub := findMethod(msgs, "textDocument/publishDiagnostics")
if pub == nil {
t.Fatal("no publishDiagnostics notification")
}
var p publishDiagnosticsParams
json.Unmarshal(pub.Params, &p)
if len(p.Diagnostics) != 0 {
t.Fatalf("clean doc should have no diagnostics, got %+v", p.Diagnostics)
}
}
func TestDiagnosticsErrors(t *testing.T) {
msgs := run(t, session("file:///f_amd64.s", badDoc)+frame(nil, "exit", nil))
pub := findMethod(msgs, "textDocument/publishDiagnostics")
if pub == nil {
t.Fatal("no publishDiagnostics notification")
}
var p publishDiagnosticsParams
json.Unmarshal(pub.Params, &p)
codes := map[string]bool{}
for _, d := range p.Diagnostics {
codes[d.Code] = true
}
if !codes["unknown-instruction"] || !codes["undefined-label"] {
t.Fatalf("expected unknown-instruction and undefined-label, got %+v", p.Diagnostics)
}
}
func TestCompletion(t *testing.T) {
in := session("file:///f_amd64.s", cleanDoc) +
frame(2, "textDocument/completion", map[string]any{
"textDocument": map[string]any{"uri": "file:///f_amd64.s"},
"position": map[string]any{"line": 2, "character": 1},
}) + frame(nil, "exit", nil)
msgs := run(t, in)
resp := findByID(msgs, 2)
if resp == nil {
t.Fatal("no completion response")
}
var items []CompletionItem
if err := json.Unmarshal(mustResult(t, resp), &items); err != nil {
t.Fatal(err)
}
labels := map[string]bool{}
for _, it := range items {
labels[it.Label] = true
}
for _, want := range []string{"MOVQ", "AX", "TEXT", "NOSPLIT", "loop"} {
if !labels[want] {
t.Errorf("completion missing %q", want)
}
}
}
func TestHover(t *testing.T) {
in := session("file:///f_amd64.s", cleanDoc) +
frame(3, "textDocument/hover", map[string]any{
"textDocument": map[string]any{"uri": "file:///f_amd64.s"},
"position": map[string]any{"line": 2, "character": 2}, // on MOVQ
}) + frame(nil, "exit", nil)
msgs := run(t, in)
resp := findByID(msgs, 3)
if resp == nil {
t.Fatal("no hover response")
}
var h Hover
if err := json.Unmarshal(mustResult(t, resp), &h); err != nil {
t.Fatal(err)
}
if !strings.Contains(h.Contents.Value, "MOVQ") {
t.Fatalf("hover = %q, want MOVQ docs", h.Contents.Value)
}
}
func TestDocumentSymbols(t *testing.T) {
in := session("file:///f_amd64.s", cleanDoc) +
frame(4, "textDocument/documentSymbol", map[string]any{
"textDocument": map[string]any{"uri": "file:///f_amd64.s"},
}) + frame(nil, "exit", nil)
msgs := run(t, in)
resp := findByID(msgs, 4)
if resp == nil {
t.Fatal("no documentSymbol response")
}
var syms []DocumentSymbol
if err := json.Unmarshal(mustResult(t, resp), &syms); err != nil {
t.Fatal(err)
}
if len(syms) == 0 || syms[0].Name != "foo" {
t.Fatalf("symbols = %+v, want function foo", syms)
}
found := false
for _, c := range syms[0].Children {
if c.Name == "loop" {
found = true
}
}
if !found {
t.Errorf("function foo should contain label loop: %+v", syms[0].Children)
}
}
func TestSemanticTokens(t *testing.T) {
in := session("file:///f_amd64.s", cleanDoc) +
frame(5, "textDocument/semanticTokens/full", map[string]any{
"textDocument": map[string]any{"uri": "file:///f_amd64.s"},
}) + frame(nil, "exit", nil)
msgs := run(t, in)
resp := findByID(msgs, 5)
if resp == nil {
t.Fatal("no semanticTokens response")
}
var st SemanticTokens
if err := json.Unmarshal(mustResult(t, resp), &st); err != nil {
t.Fatal(err)
}
if len(st.Data) == 0 || len(st.Data)%5 != 0 {
t.Fatalf("semantic tokens data invalid: len=%d", len(st.Data))
}
// There must be at least one "function" (mnemonic) and one "comment"-free
// keyword token; sanity-check that a MOVQ-classified function token exists.
seenFunction := false
for i := 3; i < len(st.Data); i += 5 {
if st.Data[i] == stFunction {
seenFunction = true
}
}
if !seenFunction {
t.Error("expected at least one function (mnemonic) semantic token")
}
}
func TestMethodNotFound(t *testing.T) {
in := frame(9, "bogus/method", map[string]any{}) + frame(nil, "exit", nil)
msgs := run(t, in)
resp := findByID(msgs, 9)
if resp == nil || resp.Error == nil || resp.Error.Code != errMethodNotFound {
t.Fatalf("expected method-not-found error, got %+v", resp)
}
}
// mustResult re-marshals a response result into raw JSON for typed decoding.
func mustResult(t *testing.T, m *rpcMessage) []byte {
t.Helper()
b, err := json.Marshal(m.Result)
if err != nil {
t.Fatal(err)
}
return b
}
+619
View File
@@ -0,0 +1,619 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package parser turns a GAsm token stream into an abstract syntax tree. It
// is line-oriented, matching how the Plan 9 assembler itself reads a file, and
// tolerant: a malformed line is reported as an error but never aborts the
// parse of the rest of the file.
package parser
import (
"fmt"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Error is a single parse diagnostic.
type Error struct {
Pos token.Position
Msg string
}
func (e Error) Error() string {
return e.Pos.String() + ": " + e.Msg
}
// Parse scans and parses src, returning the file and any diagnostics. The
// returned file is usable even when errors is non-empty.
func Parse(path, src string) (*ast.File, []error) {
toks := lexer.Tokenize(src)
lines := splitLines(toks)
p := &state{path: path}
p.parse(lines)
return p.file, p.errs
}
// state carries the mutable context for one parse.
type state struct {
path string
file *ast.File
errs []error
curText *ast.Text // the TEXT body labels/instructions attach to
pending []string // comment lines awaiting a TEXT to become its Doc
}
func (p *state) errorf(pos token.Position, format string, args ...any) {
p.errs = append(p.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)})
}
// splitLines groups the token stream into lines, dropping the Newline tokens.
func splitLines(toks []token.Token) [][]token.Token {
var lines [][]token.Token
var cur []token.Token
for _, t := range toks {
if t.Kind == token.EOF {
break
}
if t.Kind == token.Newline {
lines = append(lines, cur)
cur = nil
continue
}
cur = append(cur, t)
}
if len(cur) > 0 {
lines = append(lines, cur)
}
return lines
}
func (p *state) parse(lines [][]token.Token) {
p.file = &ast.File{Path: p.path, Macros: map[string]bool{}}
for _, line := range lines {
line = trimSpace(line)
if len(line) == 0 {
// Blank line: a comment block ends here only if it was not
// directly preceding a declaration; keep pending doc intact
// across a single blank line is not desired, so reset.
p.pending = nil
continue
}
p.parseLine(line)
}
}
// trimSpace is a no-op placeholder kept for symmetry; the lexer already drops
// horizontal whitespace, but this documents the intent.
func trimSpace(line []token.Token) []token.Token { return line }
func (p *state) parseLine(line []token.Token) {
first := line[0]
// A lone comment accumulates as documentation for a following TEXT.
if len(line) == 1 && first.Kind == token.Comment {
p.pending = append(p.pending, commentText(first.Text))
return
}
// Preprocessor line.
if first.Kind == token.Hash {
p.parsePreproc(line)
p.pending = nil
return
}
// Directives and instructions are identified by a leading identifier.
if first.Kind == token.Ident {
switch first.Text {
case "TEXT":
p.parseText(line)
return
case "GLOBL":
p.file.Decls = append(p.file.Decls, p.parseGlobl(line))
p.curText = nil
p.pending = nil
return
case "DATA":
p.file.Decls = append(p.file.Decls, p.parseData(line))
p.curText = nil
p.pending = nil
return
}
// Label (ident immediately followed by a colon).
if len(line) >= 2 && line[1].Kind == token.Colon {
lbl := &ast.Label{Name: line[0], Colon: line[1]}
p.addStmt(lbl)
// A label may share its line with an instruction: "loop: MOVQ …".
if rest := dropColon(line); len(rest) > 0 {
p.parseInstr(rest)
}
p.pending = nil
return
}
// Otherwise it is an instruction.
p.parseInstr(line)
p.pending = nil
return
}
p.errorf(first.Pos, "unexpected token %s at start of line", first.Kind)
p.pending = nil
}
// dropColon removes the leading "ident :" of a label, returning the remainder.
func dropColon(line []token.Token) []token.Token {
if len(line) >= 2 && line[1].Kind == token.Colon {
return line[2:]
}
return nil
}
func (p *state) addStmt(s ast.Stmt) {
if p.curText != nil {
p.curText.Body = append(p.curText.Body, s)
return
}
p.file.Orphans = append(p.file.Orphans, s)
}
func (p *state) parsePreproc(line []token.Token) {
hash := line[0]
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
line[2].Kind == token.String {
p.file.Decls = append(p.file.Decls, &ast.Include{
Hash: hash,
Name: line[1],
Header: line[2],
})
return
}
// Record macro names so the linter can recognise their invocations.
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "define" &&
line[2].Kind == token.Ident {
p.file.Macros[line[2].Text] = true
}
p.file.Decls = append(p.file.Decls, &ast.Preproc{
Hash: hash,
Raw: joinRaw(line[1:]),
})
}
func (p *state) parseText(line []token.Token) {
text := &ast.Text{Keyword: line[0]}
if len(p.pending) > 0 {
text.Doc = strings.Join(p.pending, "\n")
}
p.pending = nil
rest := line[1:]
sym, n := parseSymbolPrefix(rest)
if sym == nil {
p.errorf(line[0].Pos, "TEXT missing a symbol name")
}
text.Name = sym
rest = rest[n:]
// Consume flags (identifiers, possibly '|' joined) up to the frame '$'.
rest = skipComma(rest)
for len(rest) > 0 && rest[0].Kind != token.Dollar {
if rest[0].Kind == token.Ident {
text.Flags = append(text.Flags, rest[0].Text)
}
// Commas, '|' (Illegal) and anything else between flags is skipped.
rest = rest[1:]
}
// Frame: $number ; optional args: -number.
if len(rest) > 0 && rest[0].Kind == token.Dollar {
text.Frame = parseOperand(rest[:2]) // "$" "number"
rest = rest[2:]
if len(rest) >= 2 && rest[0].Kind == token.Minus && rest[1].Kind == token.Number {
text.Args = &ast.Operand{
Kind: ast.OpImmediate,
Imm: ast.Immediate{Val: parseInt(rest[1].Text), HasVal: true},
Raw: "-" + rest[1].Text,
Pos: rest[0].Pos,
}
rest = rest[2:]
}
}
p.file.Decls = append(p.file.Decls, text)
p.curText = text
}
func (p *state) parseGlobl(line []token.Token) *ast.Globl {
g := &ast.Globl{Keyword: line[0]}
rest := skipComma(line[1:])
sym, n := parseSymbolPrefix(rest)
g.Name = sym
rest = skipComma(rest[n:])
for len(rest) > 0 && rest[0].Kind != token.Dollar {
if rest[0].Kind == token.Ident {
g.Flags = append(g.Flags, rest[0].Text)
}
rest = rest[1:]
}
if len(rest) > 0 && rest[0].Kind == token.Dollar {
g.Size = parseOperand(rest)
}
return g
}
func (p *state) parseData(line []token.Token) *ast.Data {
d := &ast.Data{Keyword: line[0]}
rest := line[1:]
// The name may carry a /width suffix: ·idx+0(SB)/4. Split it off the
// symbol group that precedes the first top-level comma.
nameGroup, valuePart := splitFirstComma(rest)
nameGroup, width := splitTrailingWidth(nameGroup)
sym, _ := parseSymbolPrefix(nameGroup)
d.Name = sym
d.Width = width
if len(valuePart) > 0 {
d.Value = parseOperand(stripComment(valuePart))
}
return d
}
func (p *state) parseInstr(line []token.Token) {
body, comment := splitTrailingComment(line)
if len(body) == 0 {
return
}
instr := &ast.Instr{Mnemonic: body[0], Comment: comment}
for _, grp := range splitOperands(body[1:]) {
if op := parseOperand(grp); op != nil {
instr.Operands = append(instr.Operands, op)
}
}
p.addStmt(instr)
}
// --- symbol parsing ---------------------------------------------------------
var pseudoRegs = map[string]bool{"FP": true, "SP": true, "SB": true, "PC": true}
// parseSymbolPrefix parses a leading symbol reference from g and returns it
// together with the number of tokens consumed. It returns (nil, 0) when no
// symbol is present.
func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) {
if len(g) == 0 || g[0].Kind != token.Ident {
return nil, 0
}
sym := &ast.Symbol{Pos: g[0].Pos}
i := 0
setName(g[0].Text, sym)
i++
if i+1 < len(g) && g[i].Kind == token.LAngle && g[i+1].Kind == token.RAngle {
sym.Static = true
i += 2
}
if i < len(g) && g[i].Kind == token.Plus {
i++
if i < len(g) && g[i].Kind == token.Number {
sym.Offset, sym.HasOff = parseInt(g[i].Text), true
i++
}
}
if i+2 < len(g) && g[i].Kind == token.LParen && g[i+1].Kind == token.Ident &&
pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen {
sym.Pseudo = g[i+1].Text
i += 3
}
sym.Raw = joinRaw(g[:i])
return sym, i
}
// setName splits a raw identifier on the middle dot into package and name.
func setName(raw string, sym *ast.Symbol) {
const dot = "\u00B7"
switch {
case strings.HasPrefix(raw, dot):
sym.Pkg = ""
sym.Name = strings.TrimPrefix(raw, dot)
case strings.Contains(raw, dot):
parts := strings.SplitN(raw, dot, 2)
sym.Pkg = parts[0]
sym.Name = parts[1]
default:
sym.Name = raw
}
}
// --- operand parsing --------------------------------------------------------
// parseOperand parses one operand group into an Operand.
func parseOperand(g []token.Token) *ast.Operand {
g = stripComment(g)
if len(g) == 0 {
return nil
}
op := &ast.Operand{Raw: joinRaw(g), Pos: g[0].Pos}
if g[0].Kind == token.Dollar {
op.Kind = ast.OpImmediate
op.Imm = parseImmediate(g[1:])
return op
}
op.Kind = ast.OpAddr
op.Addr = parseAddress(g)
return op
}
// parseImmediate parses the tokens following a '$'.
func parseImmediate(g []token.Token) ast.Immediate {
var imm ast.Immediate
if len(g) == 0 {
return imm
}
// $sym(…) form.
if findPseudoParen(g) >= 0 || (g[0].Kind == token.Ident) {
if sym, n := parseSymbolPrefix(g); sym != nil && (sym.Pseudo != "" || sym.Static) {
imm.Sym = sym
_ = n
return imm
}
}
i := 0
if g[i].Kind == token.Minus {
imm.Neg = true
i++
} else if g[i].Kind == token.Plus {
i++
}
if i < len(g) && g[i].Kind == token.Number {
text := g[i].Text
if v, ok := tryInt(text); ok {
imm.Val = v
imm.HasVal = true
} else {
imm.Float = text
}
i++
} else if i < len(g) && (g[i].Kind == token.String || g[i].Kind == token.Rune) {
imm.Str = g[i].Text
i++
}
return imm
}
// parseAddress parses a non-immediate operand.
func parseAddress(g []token.Token) ast.Address {
var addr ast.Address
if len(g) == 0 {
return addr
}
// Symbol-with-pseudo form: name[<>][+off](PSEUDO).
if idx := findPseudoParen(g); idx >= 0 {
sym, _ := parseSymbolPrefix(g[:idx+3])
addr.Sym = sym
return addr
}
i := 0
// Optional leading displacement before a '(' base group.
if isSignedNumber(g, i) && i+1 < len(g) && g[i+1].Kind == token.LParen {
neg := false
if g[i].Kind == token.Minus {
neg = true
i++
} else if g[i].Kind == token.Plus {
i++
}
if i < len(g) && g[i].Kind == token.Number {
addr.Offset = parseInt(g[i].Text)
addr.HasOff = true
if neg {
addr.Offset = -addr.Offset
}
i++
}
}
// First parenthesised group: the base register.
if i < len(g) && g[i].Kind == token.LParen {
i++
if i < len(g) && g[i].Kind == token.Ident {
addr.Base = g[i].Text
i++
}
if i < len(g) && g[i].Kind == token.RParen {
i++
}
}
// Optional second group: (index*scale) or (index).
if i < len(g) && g[i].Kind == token.LParen {
i++
if i < len(g) && g[i].Kind == token.Ident {
addr.Index = g[i].Text
i++
}
if i < len(g) && g[i].Kind == token.Star {
i++
if i < len(g) && g[i].Kind == token.Number {
addr.Scale = int(parseInt(g[i].Text))
i++
}
}
if i < len(g) && g[i].Kind == token.RParen {
i++
}
}
// Bare name (register, label or symbol) possibly with an arm64 shift.
if addr.Base == "" && addr.Sym == nil && g[0].Kind == token.Ident {
sym := &ast.Symbol{Pos: g[0].Pos}
setName(g[0].Text, sym)
sym.Raw = g[0].Text
addr.Sym = sym
i = 1
}
// Any remaining tokens form a verbatim shift/extension suffix (arm64).
if i > 0 && i < len(g) {
addr.Shift = joinRaw(g[i:])
}
return addr
}
// findPseudoParen returns the index of the '(' that begins a (PSEUDO) group,
// or -1 when none is present.
func findPseudoParen(g []token.Token) int {
for i := 0; i+2 < len(g); i++ {
if g[i].Kind == token.LParen && g[i+1].Kind == token.Ident &&
pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen {
return i
}
}
return -1
}
// --- token helpers ----------------------------------------------------------
// splitOperands splits a token slice on top-level commas (commas outside any
// parenthesis group).
func splitOperands(g []token.Token) [][]token.Token {
var out [][]token.Token
var cur []token.Token
depth := 0
for _, t := range g {
switch t.Kind {
case token.LParen:
depth++
cur = append(cur, t)
case token.RParen:
depth--
cur = append(cur, t)
case token.Comma:
if depth == 0 {
if len(cur) > 0 {
out = append(out, cur)
}
cur = nil
} else {
cur = append(cur, t)
}
case token.Comment:
// A comment terminates the operand list.
if len(cur) > 0 {
out = append(out, cur)
}
return out
default:
cur = append(cur, t)
}
}
if len(cur) > 0 {
out = append(out, cur)
}
return out
}
// splitFirstComma splits g at the first top-level comma.
func splitFirstComma(g []token.Token) (before, after []token.Token) {
depth := 0
for i, t := range g {
switch t.Kind {
case token.LParen:
depth++
case token.RParen:
depth--
case token.Comma:
if depth == 0 {
return g[:i], g[i+1:]
}
}
}
return g, nil
}
// splitTrailingComment separates a trailing comment from the line body.
func splitTrailingComment(g []token.Token) (body []token.Token, comment string) {
for i, t := range g {
if t.Kind == token.Comment {
return g[:i], commentText(t.Text)
}
}
return g, ""
}
// stripComment removes a trailing comment token from a group.
func stripComment(g []token.Token) []token.Token {
for i, t := range g {
if t.Kind == token.Comment {
return g[:i]
}
}
return g
}
// splitTrailingWidth removes a "/width" suffix from a DATA name group.
func splitTrailingWidth(g []token.Token) ([]token.Token, int) {
for i := 0; i+1 < len(g); i++ {
if g[i].Kind == token.Slash && g[i+1].Kind == token.Number {
return g[:i], int(parseInt(g[i+1].Text))
}
}
return g, 0
}
func skipComma(g []token.Token) []token.Token {
if len(g) > 0 && g[0].Kind == token.Comma {
return g[1:]
}
return g
}
func isSignedNumber(g []token.Token, i int) bool {
if i >= len(g) {
return false
}
if g[i].Kind == token.Number {
return true
}
if (g[i].Kind == token.Minus || g[i].Kind == token.Plus) &&
i+1 < len(g) && g[i+1].Kind == token.Number {
return true
}
return false
}
func joinRaw(g []token.Token) string {
parts := make([]string, len(g))
for i, t := range g {
parts[i] = t.Text
}
return strings.Join(parts, " ")
}
// commentText removes a leading // or /* marker from a comment token's text.
func commentText(s string) string {
if strings.HasPrefix(s, "//") {
return strings.TrimSpace(strings.TrimPrefix(s, "//"))
}
if strings.HasPrefix(s, "/*") {
s = strings.TrimPrefix(s, "/*")
s = strings.TrimSuffix(s, "*/")
return strings.TrimSpace(s)
}
return s
}
func parseInt(text string) int64 {
v, _ := tryInt(text)
return v
}
func tryInt(text string) (int64, bool) {
v, err := strconv.ParseInt(text, 0, 64)
if err != nil {
return 0, false
}
return v, true
}
+241
View File
@@ -0,0 +1,241 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package parser
import (
"os"
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
func mustParse(t *testing.T, path string) *ast.File {
t.Helper()
src, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read %s: %v", path, err)
}
file, errs := Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse %s: %v", path, errs)
}
return file
}
func texts(f *ast.File) []*ast.Text {
var out []*ast.Text
for _, d := range f.Decls {
if t, ok := d.(*ast.Text); ok {
out = append(out, t)
}
}
return out
}
func TestParseSample(t *testing.T) {
f := mustParse(t, "../testdata/sample_amd64.s")
// Includes, GLOBL/DATA and two TEXT functions.
var includes, globls, datas int
for _, d := range f.Decls {
switch d.(type) {
case *ast.Include:
includes++
case *ast.Globl:
globls++
case *ast.Data:
datas++
}
}
if includes != 1 {
t.Errorf("includes = %d, want 1", includes)
}
if globls != 2 {
t.Errorf("globls = %d, want 2", globls)
}
if datas != 4 {
t.Errorf("datas = %d, want 4", datas)
}
txts := texts(f)
if len(txts) != 2 {
t.Fatalf("text functions = %d, want 2", len(txts))
}
fn := txts[0]
if fn.Name.Name != "analyzeO1RangeAVX2" {
t.Errorf("name = %q, want analyzeO1RangeAVX2", fn.Name.Name)
}
if fn.Name.Pseudo != "SB" {
t.Errorf("pseudo = %q, want SB", fn.Name.Pseudo)
}
if len(fn.Flags) != 1 || fn.Flags[0] != "NOSPLIT" {
t.Errorf("flags = %v, want [NOSPLIT]", fn.Flags)
}
if fn.Frame == nil || !fn.Frame.Imm.HasVal || fn.Frame.Imm.Val != 0 {
t.Errorf("frame = %+v, want $0", fn.Frame)
}
if fn.Args == nil || fn.Args.Imm.Val != 65 {
t.Errorf("args = %+v, want 65", fn.Args)
}
if fn.Doc == "" {
t.Error("expected a doc comment on the first TEXT")
}
// The body must contain the two labels vec1 and vec1done.
labels := map[string]bool{}
for _, s := range fn.Body {
if l, ok := s.(*ast.Label); ok {
labels[l.Name.Text] = true
}
}
for _, want := range []string{"vec1", "vec1done"} {
if !labels[want] {
t.Errorf("missing label %q", want)
}
}
}
func TestOperandStructure(t *testing.T) {
f := mustParse(t, "../testdata/sample_amd64.s")
fn := texts(f)[0]
// Index instructions by mnemonic for targeted checks.
byMnem := map[string]*ast.Instr{}
for _, s := range fn.Body {
if in, ok := s.(*ast.Instr); ok {
byMnem[in.Mnemonic.Text] = in
}
}
// MOVQ swin_base+0(FP), SI — the first MOVQ in the body.
var mov *ast.Instr
for _, s := range fn.Body {
if in, ok := s.(*ast.Instr); ok && in.Mnemonic.Text == "MOVQ" {
mov = in
break
}
}
if mov == nil {
t.Fatal("MOVQ not found")
}
src := mov.Operands[0]
if src.Kind != ast.OpAddr || src.Addr.Sym == nil {
t.Fatalf("src operand = %+v, want symbol address", src)
}
if src.Addr.Sym.Name != "swin_base" || src.Addr.Sym.Pseudo != "FP" || src.Addr.Sym.Offset != 0 {
t.Errorf("src symbol = %+v, want swin_base+0(FP)", src.Addr.Sym)
}
if mov.Operands[1].Addr.Sym.Name != "SI" {
t.Errorf("dst = %+v, want SI", mov.Operands[1].Addr)
}
// LEAQ (SI)(BX*4), R9
leaq := byMnem["LEAQ"]
if leaq == nil {
t.Fatal("LEAQ not found")
}
mem := leaq.Operands[0].Addr
if mem.Base != "SI" || mem.Index != "BX" || mem.Scale != 4 {
t.Errorf("LEAQ addr = %+v, want base SI index BX scale 4", mem)
}
// ANDQ $-8, R10
andq := byMnem["ANDQ"]
if andq == nil {
t.Fatal("ANDQ not found")
}
imm := andq.Operands[0]
if imm.Kind != ast.OpImmediate || !imm.Imm.Neg || imm.Imm.Val != 8 {
t.Errorf("ANDQ imm = %+v, want -8", imm.Imm)
}
}
func TestAVX512Operands(t *testing.T) {
f := mustParse(t, "../testdata/sample_amd64.s")
fn := texts(f)[1]
byMnem := map[string]*ast.Instr{}
for _, s := range fn.Body {
if in, ok := s.(*ast.Instr); ok {
byMnem[in.Mnemonic.Text] = in
}
}
// VALIGND $15, Z9, Z0, Z1 — four operands.
val := byMnem["VALIGND"]
if val == nil {
t.Fatal("VALIGND not found")
}
if len(val.Operands) != 4 {
t.Errorf("VALIGND operands = %d, want 4", len(val.Operands))
}
if val.Operands[0].Kind != ast.OpImmediate || val.Operands[0].Imm.Val != 15 {
t.Errorf("VALIGND first operand = %+v, want $15", val.Operands[0])
}
// VMOVDQU32 Z0, 4(SI)(AX*1)
vmov := byMnem["VMOVDQU32"]
if vmov == nil {
t.Fatal("VMOVDQU32 not found")
}
dst := vmov.Operands[len(vmov.Operands)-1].Addr
if dst.Offset != 4 || dst.Base != "SI" || dst.Index != "AX" || dst.Scale != 1 {
t.Errorf("VMOVDQU32 dst = %+v, want 4(SI)(AX*1)", dst)
}
// KTESTW K1, K1 — mask registers parse as bare names.
kt := byMnem["KTESTW"]
if kt == nil || len(kt.Operands) != 2 {
t.Fatalf("KTESTW = %+v, want two operands", kt)
}
}
func TestDataWidthAndStatic(t *testing.T) {
f := mustParse(t, "../testdata/sample_amd64.s")
var datas []*ast.Data
for _, d := range f.Decls {
if dd, ok := d.(*ast.Data); ok {
datas = append(datas, dd)
}
}
if datas[0].Width != 4 {
t.Errorf("first DATA width = %d, want 4", datas[0].Width)
}
if datas[0].Name.Pseudo != "SB" || datas[0].Name.Offset != 0 {
t.Errorf("first DATA name = %+v, want +0(SB)", datas[0].Name)
}
if datas[0].Value.Kind != ast.OpImmediate || datas[0].Value.Imm.Val != 1 {
t.Errorf("first DATA value = %+v, want $1", datas[0].Value)
}
// The mask24<> entries are static.
if !datas[2].Name.Static {
t.Errorf("mask24 DATA should be static, got %+v", datas[2].Name)
}
}
// TestParseRealGoLibraries parses every .s file in the sibling go-libraries
// repository when it is checked out, asserting a clean, error-free parse. It
// is skipped when the repository is not present.
func TestParseRealGoLibraries(t *testing.T) {
matches, _ := filepath.Glob("../../go-libraries/go-*/*.s")
if len(matches) == 0 {
t.Skip("go-libraries repository not present next to gasm-devkit")
}
for _, path := range matches {
src, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read %s: %v", path, err)
}
file, errs := Parse(path, string(src))
if len(errs) > 0 {
t.Errorf("parse %s: %v", path, errs)
continue
}
if len(texts(file)) == 0 {
t.Errorf("parse %s: no TEXT functions found", path)
}
t.Logf("%s: %d decls, %d functions", filepath.Base(path), len(file.Decls), len(texts(file)))
}
}
+56
View File
@@ -0,0 +1,56 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
#include "textflag.h"
// Index vector for the order-2 ramp.
GLOBL ·idx16(SB), RODATA, $64
DATA ·idx16+0(SB)/4, $1
DATA ·idx16+4(SB)/4, $2
// A file-local (static) constant table.
GLOBL mask24<>(SB), RODATA, $16
DATA mask24<>+0(SB)/4, $0x80020100
DATA mask24<>+4(SB)/4, $0x80050403
// func analyzeO1RangeAVX2(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool)
TEXT ·analyzeO1RangeAVX2(SB), NOSPLIT, $0-65
MOVQ swin_base+0(FP), SI
MOVQ dstP_base+24(FP), DI
MOVQ dstP_len+32(FP), BX
MOVQ hist+48(FP), R13
VPCMPEQD Y0, Y0, Y0
VPSLLD $31, Y0, Y0
LEAQ (SI)(BX*4), R9
MOVQ BX, R10
ANDQ $-8, R10
vec1:
CMPQ SI, R10
JGE vec1done
VMOVDQU (SI), Y1
VMOVDQU 4(SI), Y2
VPSUBD Y1, Y2, Y3
ADDQ $32, SI
JMP vec1
vec1done:
MOVQ AX, partSum+56(FP)
MOVB AL, overflow+64(FP)
VZEROUPPER
RET
// func decodeFixedO1AVX512(samples []int32, residual []int32)
TEXT ·decodeFixedO1AVX512(SB), NOSPLIT, $0-48
MOVQ samples_base+0(FP), SI
VPBROADCASTD AX, Z15
VMOVDQU32 (DI)(AX*1), Z0
VALIGND $15, Z9, Z0, Z1
VFMADD231PD Z14, Z12, Z10
VPCMPEQD Z0, Z3, K1
KTESTW K1, K1
VPSRAQ X31, Z8, Z8
VMOVDQU32 Z0, 4(SI)(AX*1)
RET
+108
View File
@@ -0,0 +1,108 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package token defines the lexical tokens of Go's Plan 9 assembler (GAsm)
// and the source positions attached to them. It has no dependencies and is
// shared by the lexer, parser, formatter, linter and language server.
package token
import "strconv"
// Kind classifies a lexical token.
type Kind int
// The token kinds. The zero value is Illegal so that an uninitialised Token
// is obviously invalid.
const (
Illegal Kind = iota
EOF
Newline
Comment
// Literals and names.
Ident // instruction mnemonic, label, register or symbol name
Number // integer or floating-point literal (sign carried separately)
String // "..."
Rune // '.'
// Punctuation and operators.
LParen // (
RParen // )
Comma // ,
Plus // +
Minus // -
Star // *
Slash // /
Colon // :
Dollar // $
LAngle // <
RAngle // >
LShift // <<
RShift // >>
Arrow // ->
At // @
Hash // #
)
var kindNames = map[Kind]string{
Illegal: "ILLEGAL",
EOF: "EOF",
Newline: "NEWLINE",
Comment: "COMMENT",
Ident: "IDENT",
Number: "NUMBER",
String: "STRING",
Rune: "RUNE",
LParen: "(",
RParen: ")",
Comma: ",",
Plus: "+",
Minus: "-",
Star: "*",
Slash: "/",
Colon: ":",
Dollar: "$",
LAngle: "<",
RAngle: ">",
LShift: "<<",
RShift: ">>",
Arrow: "->",
At: "@",
Hash: "#",
}
// String returns a human-readable name for the kind.
func (k Kind) String() string {
if s, ok := kindNames[k]; ok {
return s
}
return "Kind(" + strconv.Itoa(int(k)) + ")"
}
// Position is a byte offset plus one-based line and column within a file.
type Position struct {
Offset int // byte offset, zero-based
Line int // line number, one-based
Column int // column number, one-based (in runes)
}
// String renders the position as "line:column".
func (p Position) String() string {
return strconv.Itoa(p.Line) + ":" + strconv.Itoa(p.Column)
}
// IsValid reports whether the position carries a real line number.
func (p Position) IsValid() bool { return p.Line > 0 }
// Token is a single lexical token together with its literal text and span.
type Token struct {
Kind Kind
Text string
Pos Position // inclusive start
End Position // exclusive end
}
// String renders the token for diagnostics and debugging.
func (t Token) String() string {
return t.Pos.String() + " " + t.Kind.String() + " " + strconv.Quote(t.Text)
}
+49
View File
@@ -0,0 +1,49 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package token
import "testing"
func TestKindString(t *testing.T) {
cases := map[Kind]string{
EOF: "EOF",
Ident: "IDENT",
Number: "NUMBER",
LParen: "(",
LShift: "<<",
Arrow: "->",
Illegal: "ILLEGAL",
}
for k, want := range cases {
if got := k.String(); got != want {
t.Errorf("Kind(%d).String() = %q, want %q", int(k), got, want)
}
}
// An out-of-range kind falls back to the numeric form.
if got := Kind(9999).String(); got != "Kind(9999)" {
t.Errorf("Kind(9999).String() = %q", got)
}
}
func TestPosition(t *testing.T) {
p := Position{Offset: 10, Line: 3, Column: 7}
if got := p.String(); got != "3:7" {
t.Errorf("Position.String() = %q, want 3:7", got)
}
if !p.IsValid() {
t.Error("position with a line should be valid")
}
var zero Position
if zero.IsValid() {
t.Error("zero position should be invalid")
}
}
func TestTokenString(t *testing.T) {
tok := Token{Kind: Ident, Text: "MOVQ", Pos: Position{Line: 1, Column: 2}}
got := tok.String()
if got == "" || got[0] == ' ' {
t.Errorf("Token.String() = %q", got)
}
}