feat: gasm-devkit 0.1.0 — GAsm lexer, parser, linter, formatter, LSP and amd64 assembler
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
+12
@@ -0,0 +1,12 @@
|
||||
# Binaries
|
||||
/gasm
|
||||
/bin/
|
||||
*.exe
|
||||
|
||||
# Test and coverage artefacts
|
||||
coverage.out
|
||||
*.test
|
||||
|
||||
# Editor detritus
|
||||
*.swp
|
||||
.DS_Store
|
||||
@@ -0,0 +1,28 @@
|
||||
BSD 3-Clause License
|
||||
|
||||
Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are met:
|
||||
|
||||
1. Redistributions of source code must retain the above copyright notice,
|
||||
this list of conditions and the following disclaimer.
|
||||
|
||||
2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimer in the documentation
|
||||
and/or other materials provided with the distribution.
|
||||
|
||||
3. Neither the name of the copyright holder nor the names of its
|
||||
contributors may be used to endorse or promote products derived from
|
||||
this software without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
+180
@@ -0,0 +1,180 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Command gen regenerates the architecture instruction tables from the Go
|
||||
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
|
||||
// complete set of mnemonics it accepts for each architecture in
|
||||
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
|
||||
// names so gasm-devkit supports every instruction the real assembler does,
|
||||
// with no hand-maintained (and therefore inevitably incomplete) lists.
|
||||
//
|
||||
// Usage (via the justfile):
|
||||
//
|
||||
// just gen
|
||||
//
|
||||
// The generated files are committed; regenerating requires a Go installation
|
||||
// but the toolkit itself has no dependency on the toolchain source at runtime.
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// archDirs maps a gasm-devkit architecture name to its obj sub-directory.
|
||||
var archDirs = []struct {
|
||||
arch string
|
||||
sub string
|
||||
}{
|
||||
{"amd64", "x86"},
|
||||
{"arm64", "arm64"},
|
||||
{"riscv", "riscv"},
|
||||
{"loong64", "loong64"},
|
||||
}
|
||||
|
||||
func main() {
|
||||
goroot := strings.TrimSpace(runGoEnvGOROOT())
|
||||
if goroot == "" {
|
||||
fatal("could not determine GOROOT")
|
||||
}
|
||||
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
|
||||
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
|
||||
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
|
||||
common, err := extractInstrs(commonPath)
|
||||
if err != nil {
|
||||
fatal("extract common: %v", err)
|
||||
}
|
||||
common = filterCommon(common)
|
||||
if err := writeCommon(common); err != nil {
|
||||
fatal("write common: %v", err)
|
||||
}
|
||||
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
|
||||
|
||||
for _, a := range archDirs {
|
||||
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
|
||||
names, err := extractInstrs(path)
|
||||
if err != nil {
|
||||
fatal("extract %s: %v", a.arch, err)
|
||||
}
|
||||
if err := writeGen(a.arch, a.sub, names); err != nil {
|
||||
fatal("write %s: %v", a.arch, err)
|
||||
}
|
||||
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names), a.arch)
|
||||
}
|
||||
}
|
||||
|
||||
// filterCommon drops opcode names that are not user-writable instructions.
|
||||
func filterCommon(names []string) []string {
|
||||
drop := map[string]bool{"XXX": true, "LAST": true}
|
||||
var out []string
|
||||
for _, n := range names {
|
||||
if !drop[n] {
|
||||
out = append(out, n)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// writeCommon emits arch/common_gen.go.
|
||||
func writeCommon(names []string) error {
|
||||
var b strings.Builder
|
||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
||||
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n\n")
|
||||
b.WriteString("package arch\n\n")
|
||||
b.WriteString("// commonGeneratedInstrs is the set of opcodes shared by every architecture\n")
|
||||
b.WriteString("// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).\n")
|
||||
b.WriteString("var commonGeneratedInstrs = []string{\n")
|
||||
for _, n := range names {
|
||||
fmt.Fprintf(&b, "\t%q,\n", n)
|
||||
}
|
||||
b.WriteString("}\n")
|
||||
return os.WriteFile(filepath.Join("arch", "common_gen.go"), []byte(b.String()), 0o644)
|
||||
}
|
||||
|
||||
// extractInstrs parses an anames.go file and returns the sorted, de-duplicated
|
||||
// instruction names from its `var Anames = []string{...}` literal.
|
||||
func extractInstrs(path string) ([]string, error) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, path, nil, 0)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
seen := map[string]bool{}
|
||||
var names []string
|
||||
for _, decl := range f.Decls {
|
||||
gd, ok := decl.(*ast.GenDecl)
|
||||
if !ok || gd.Tok != token.VAR {
|
||||
continue
|
||||
}
|
||||
for _, spec := range gd.Specs {
|
||||
vs, ok := spec.(*ast.ValueSpec)
|
||||
if !ok || len(vs.Names) == 0 || vs.Names[0].Name != "Anames" {
|
||||
continue
|
||||
}
|
||||
for _, val := range vs.Values {
|
||||
cl, ok := val.(*ast.CompositeLit)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
for _, elt := range cl.Elts {
|
||||
if lit := stringLit(elt); lit != "" && lit != "LAST" && !seen[lit] {
|
||||
seen[lit] = true
|
||||
names = append(names, lit)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
return names, nil
|
||||
}
|
||||
|
||||
// stringLit returns the string value of a composite-literal element, whether it
|
||||
// is a plain literal or a keyed entry such as `obj.A_ARCHSPECIFIC: "AAA"`.
|
||||
func stringLit(elt ast.Expr) string {
|
||||
switch e := elt.(type) {
|
||||
case *ast.BasicLit:
|
||||
if e.Kind == token.STRING {
|
||||
return strings.Trim(e.Value, `"`)
|
||||
}
|
||||
case *ast.KeyValueExpr:
|
||||
return stringLit(e.Value)
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// writeGen emits arch/<arch>_gen.go.
|
||||
func writeGen(arch, sub string, names []string) error {
|
||||
var b strings.Builder
|
||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
||||
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n\n")
|
||||
b.WriteString("package arch\n\n")
|
||||
b.WriteString("// " + arch + "GeneratedInstrs is the complete set of " + arch +
|
||||
" mnemonics accepted by\n// Go's Plan 9 assembler.\n")
|
||||
b.WriteString("var " + arch + "GeneratedInstrs = []string{\n")
|
||||
for _, n := range names {
|
||||
fmt.Fprintf(&b, "\t%q,\n", n)
|
||||
}
|
||||
b.WriteString("}\n")
|
||||
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
|
||||
}
|
||||
|
||||
func runGoEnvGOROOT() string {
|
||||
out, err := exec.Command("go", "env", "GOROOT").Output()
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
return string(out)
|
||||
}
|
||||
|
||||
func fatal(format string, args ...any) {
|
||||
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
|
||||
os.Exit(1)
|
||||
}
|
||||
+317
@@ -0,0 +1,317 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package arch
|
||||
|
||||
import "fmt"
|
||||
|
||||
func buildAMD64() *Table {
|
||||
return newTable(AMD64, amd64Registers(), mergedInstrs(amd64Summaries(), commonGeneratedInstrs, amd64GeneratedInstrs, amd64Aliases()))
|
||||
}
|
||||
|
||||
// amd64Aliases are the traditional x86 conditional-jump spellings (plus a few
|
||||
// instruction aliases) that Go's assembler accepts and maps onto its canonical
|
||||
// opcodes. They are user-writable but absent from the generated opcode table,
|
||||
// so they are listed explicitly here.
|
||||
func amd64Aliases() []string {
|
||||
return []string{
|
||||
"JA", "JAE", "JB", "JBE", "JC", "JCC", "JCS", "JE", "JG", "JHI", "JHS",
|
||||
"JL", "JLO", "JLS", "JMI", "JNA", "JNAE", "JNB", "JNBE", "JNC", "JNG",
|
||||
"JNGE", "JNL", "JNLE", "JNO", "JNP", "JNS", "JNZ", "JO", "JOC", "JOS",
|
||||
"JP", "JPC", "JPE", "JPL", "JPO", "JPS", "JS", "JZ",
|
||||
"MASKMOVDQU", "MOVDQ2Q", "MOVNTDQ", "MOVOA", "PSLLDQ", "PSRLDQ",
|
||||
"MOVD", "PADDD", "MOVBELL", "MOVBEQQ", "MOVBEWW",
|
||||
}
|
||||
}
|
||||
|
||||
// amd64Summaries returns the curated documentation/operand-count table keyed by
|
||||
// upper-case mnemonic. It enriches the complete generated name list; names
|
||||
// without a curated entry are still recognised, just without a summary.
|
||||
func amd64Summaries() map[string]Instr { return toMap(amd64Curated()) }
|
||||
|
||||
// amd64Registers builds the amd64 register file. Numbered registers are
|
||||
// generated; the irregularly named ones are listed explicitly.
|
||||
func amd64Registers() []Register {
|
||||
var regs []Register
|
||||
add := func(name string, class RegClass, desc string) {
|
||||
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
|
||||
}
|
||||
|
||||
// 64-bit general-purpose registers.
|
||||
for _, n := range []string{"AX", "BX", "CX", "DX", "SI", "DI", "BP", "SP"} {
|
||||
add(n, GPR, "64-bit general-purpose register")
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
|
||||
}
|
||||
// 8-bit low/high sub-registers.
|
||||
for _, n := range []string{"AL", "BL", "CL", "DL", "SIL", "DIL", "BPL", "SPL"} {
|
||||
add(n, GPRSub, "8-bit low sub-register")
|
||||
}
|
||||
for _, n := range []string{"AH", "BH", "CH", "DH"} {
|
||||
add(n, GPRSub, "8-bit high sub-register")
|
||||
}
|
||||
// Sized numbered sub-registers.
|
||||
for i := 8; i <= 15; i++ {
|
||||
add(fmt.Sprintf("R%dB", i), GPRSub, "8-bit sub-register")
|
||||
add(fmt.Sprintf("R%dW", i), GPRSub, "16-bit sub-register")
|
||||
add(fmt.Sprintf("R%dD", i), GPRSub, "32-bit sub-register")
|
||||
}
|
||||
// SIMD vector registers: X (SSE), Y (AVX2), Z (AVX-512).
|
||||
for i := 0; i <= 15; i++ {
|
||||
add(fmt.Sprintf("X%d", i), Vector, "128-bit SSE/AVX vector register")
|
||||
add(fmt.Sprintf("Y%d", i), Vector, "256-bit AVX2 vector register")
|
||||
add(fmt.Sprintf("Z%d", i), Vector, "512-bit AVX-512 vector register")
|
||||
}
|
||||
for i := 16; i <= 31; i++ {
|
||||
add(fmt.Sprintf("Z%d", i), Vector, "512-bit AVX-512 vector register")
|
||||
}
|
||||
// AVX-512 mask registers.
|
||||
for i := 0; i <= 7; i++ {
|
||||
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
|
||||
}
|
||||
return regs
|
||||
}
|
||||
|
||||
// i builds an instruction with an unknown/variable operand count.
|
||||
func i(name, summary string) Instr {
|
||||
return Instr{Name: name, Summary: summary, MinOps: -1, MaxOps: -1}
|
||||
}
|
||||
|
||||
// ic builds an instruction with an explicit operand-count range.
|
||||
func ic(name, summary string, min, max int) Instr {
|
||||
return Instr{Name: name, Summary: summary, MinOps: min, MaxOps: max}
|
||||
}
|
||||
|
||||
// amd64Curated returns the hand-written subset of amd64 instructions that carry
|
||||
// a summary and/or an explicit operand-count range. The authoritative,
|
||||
// complete instruction set is amd64GeneratedInstrs (see amd64_gen.go).
|
||||
func amd64Curated() []Instr {
|
||||
var t []Instr
|
||||
|
||||
// Data movement.
|
||||
for _, s := range []string{"B", "W", "L", "Q"} {
|
||||
t = append(t, ic("MOV"+s, "Move "+s+"-width value", 2, 2))
|
||||
}
|
||||
for _, m := range []string{
|
||||
"MOVBLZX", "MOVBQSX", "MOVWLZX", "MOVWQSX", "MOVWLSX", "MOVLQSX", "MOVBLSX", "MOVQL",
|
||||
} {
|
||||
t = append(t, ic(m, "Sign/zero-extending move", 2, 2))
|
||||
}
|
||||
for _, m := range []string{"MOVO", "MOVOU"} {
|
||||
t = append(t, ic(m, "Move 16-byte aligned/unaligned vector", 2, 2))
|
||||
}
|
||||
for _, s := range []string{"B", "W", "L", "Q"} {
|
||||
t = append(t, ic("LEA"+s, "Load effective address", 2, 2))
|
||||
}
|
||||
for _, s := range []string{"B", "W", "L", "Q"} {
|
||||
t = append(t, ic("XCHG"+s, "Exchange operands", 2, 2))
|
||||
}
|
||||
|
||||
// Integer arithmetic and logic.
|
||||
for _, op := range []string{"ADD", "SUB", "AND", "OR", "XOR", "ADC", "SBB"} {
|
||||
for _, s := range []string{"B", "W", "L", "Q"} {
|
||||
t = append(t, ic(op+s, op+" integer", 2, 2))
|
||||
}
|
||||
}
|
||||
for _, s := range []string{"B", "W", "L", "Q"} {
|
||||
t = append(t, ic("INC"+s, "Increment", 1, 1))
|
||||
t = append(t, ic("DEC"+s, "Decrement", 1, 1))
|
||||
t = append(t, ic("NEG"+s, "Two's-complement negate", 1, 1))
|
||||
t = append(t, ic("NOT"+s, "Bitwise complement", 1, 1))
|
||||
}
|
||||
for _, s := range []string{"B", "W", "L", "Q"} {
|
||||
t = append(t, i("IMUL"+s, "Signed multiply"))
|
||||
t = append(t, ic("IMUL3"+s, "Signed multiply by immediate", 3, 3))
|
||||
t = append(t, ic("MUL"+s, "Unsigned multiply", 1, 1))
|
||||
t = append(t, ic("DIV"+s, "Unsigned divide", 1, 1))
|
||||
t = append(t, ic("IDIV"+s, "Signed divide", 1, 1))
|
||||
}
|
||||
|
||||
// Shifts and rotates.
|
||||
for _, op := range []string{"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR"} {
|
||||
for _, s := range []string{"B", "W", "L", "Q"} {
|
||||
t = append(t, i(op+s, op+" shift/rotate"))
|
||||
}
|
||||
}
|
||||
for _, s := range []string{"W", "L", "Q"} {
|
||||
t = append(t, i("SHLD"+s, "Double-precision left shift"))
|
||||
t = append(t, i("SHRD"+s, "Double-precision right shift"))
|
||||
}
|
||||
|
||||
// Compare and test.
|
||||
for _, s := range []string{"B", "W", "L", "Q"} {
|
||||
t = append(t, ic("CMP"+s, "Compare (subtract, flags only)", 2, 2))
|
||||
t = append(t, ic("TEST"+s, "AND, flags only", 2, 2))
|
||||
}
|
||||
for _, op := range []string{"BT", "BTS", "BTR", "BTC"} {
|
||||
for _, s := range []string{"W", "L", "Q"} {
|
||||
t = append(t, i(op+s, "Bit test"+op[1:]))
|
||||
}
|
||||
}
|
||||
|
||||
// Control flow.
|
||||
t = append(t, ic("JMP", "Unconditional jump", 1, 1))
|
||||
for _, cc := range []string{
|
||||
"EQ", "NE", "Z", "NZ", "L", "LE", "G", "GE", "LT", "GT", "MI", "PL",
|
||||
"B", "BE", "A", "AE", "CS", "CC", "HI", "LS", "C", "NC",
|
||||
"S", "NS", "O", "NO", "P", "NP", "PE", "PO", "OS", "OC",
|
||||
"CXZ", "ECXZ", "RCXZ",
|
||||
} {
|
||||
t = append(t, ic("J"+cc, "Conditional jump", 1, 1))
|
||||
}
|
||||
t = append(t, ic("CALL", "Call subroutine", 1, 1))
|
||||
t = append(t, ic("RET", "Return from subroutine", 0, 0))
|
||||
t = append(t, ic("RETF", "Far return", 0, 0))
|
||||
t = append(t, ic("NOP", "No operation", 0, 1))
|
||||
t = append(t, i("INT", "Software interrupt"))
|
||||
t = append(t, ic("SYSCALL", "System call", 0, 0))
|
||||
t = append(t, ic("HLT", "Halt", 0, 0))
|
||||
t = append(t, ic("UD2", "Undefined instruction (trap)", 0, 0))
|
||||
|
||||
// Conditional set and move.
|
||||
for _, cc := range []string{
|
||||
"EQ", "NE", "L", "LE", "G", "GE", "LT", "GT", "B", "BE", "A", "AE",
|
||||
"CS", "CC", "HI", "LS", "S", "NS", "O", "NO", "P", "NP", "MI", "PL",
|
||||
} {
|
||||
t = append(t, ic("SET"+cc, "Set byte on condition", 1, 1))
|
||||
}
|
||||
for _, s := range []string{"L", "Q", "W"} {
|
||||
for _, cc := range []string{"EQ", "NE", "LT", "LE", "GT", "GE"} {
|
||||
t = append(t, ic("CMOV"+s+cc, "Conditional move", 2, 2))
|
||||
}
|
||||
}
|
||||
|
||||
// Bit scanning and counting.
|
||||
for _, op := range []string{"LZCNT", "TZCNT", "POPCNT", "BSF", "BSR"} {
|
||||
for _, s := range []string{"W", "L", "Q"} {
|
||||
t = append(t, ic(op+s, op+" bit operation", 2, 2))
|
||||
}
|
||||
}
|
||||
for _, s := range []string{"L", "Q"} {
|
||||
t = append(t, ic("BSWAP"+s, "Byte-swap", 1, 1))
|
||||
}
|
||||
for _, m := range []string{"CDQ", "CQO", "CBW", "CWDE", "CDQE"} {
|
||||
t = append(t, ic(m, "Sign-extend accumulator", 0, 0))
|
||||
}
|
||||
for _, m := range []string{"CPUID", "RDTSC", "LFENCE", "SFENCE", "MFENCE", "PAUSE"} {
|
||||
t = append(t, ic(m, "Serialising/system instruction", 0, 0))
|
||||
}
|
||||
|
||||
// SIMD data movement.
|
||||
for _, m := range []string{
|
||||
"VMOVDQU", "VMOVDQA", "VMOVUPS", "VMOVUPD", "VMOVAPS", "VMOVAPD",
|
||||
"VMOVSD", "VMOVSS", "VMOVQ", "VMOVD",
|
||||
"VMOVDQU32", "VMOVDQU64", "VMOVDQA32", "VMOVDQA64",
|
||||
"MOVDQU", "MOVDQA", "MOVUPS", "MOVUPD", "MOVAPS", "MOVAPD", "MOVSD", "MOVSS", "MOVD",
|
||||
} {
|
||||
t = append(t, i(m, "SIMD move"))
|
||||
}
|
||||
|
||||
// SIMD integer logic and arithmetic.
|
||||
for _, m := range []string{
|
||||
"VPXOR", "VPXORD", "VPXORQ", "VPAND", "VPANDN", "VPANDD", "VPANDND", "VPOR", "VPORD", "VPORQ",
|
||||
"VPADDB", "VPADDW", "VPADDD", "VPADDQ",
|
||||
"VPSUBB", "VPSUBW", "VPSUBD", "VPSUBQ",
|
||||
"VPMULLW", "VPMULLD", "VPMULLQ", "VPMULDQ", "VPMULUDQ", "VPMULHUW", "VPMULHW",
|
||||
"VPADUSB", "VPADUSW", "VPSUBUSB", "VPSUBUSW",
|
||||
"VPMINSB", "VPMINSW", "VPMINSD", "VPMAXSB", "VPMAXSW", "VPMAXSD",
|
||||
"VPABSB", "VPABSW", "VPABSD", "VPABSQ",
|
||||
"VPSLLW", "VPSLLD", "VPSLLQ", "VPSRLW", "VPSRLD", "VPSRLQ",
|
||||
"VPSRAW", "VPSRAD", "VPSRAQ", "VPSRAVD", "VPSRAVQ", "VPSLLVD", "VPSLLVQ", "VPSRLVD", "VPSRLVQ",
|
||||
"VPAVGB", "VPAVGW",
|
||||
"VPACKSSDW", "VPACKSSWB", "VPACKUSDW", "VPACKUSWB",
|
||||
} {
|
||||
t = append(t, i(m, "Packed integer SIMD"))
|
||||
}
|
||||
|
||||
// SIMD comparison.
|
||||
for _, m := range []string{
|
||||
"VPCMPEQB", "VPCMPEQW", "VPCMPEQD", "VPCMPEQQ",
|
||||
"VPCMPGTB", "VPCMPGTW", "VPCMPGTD", "VPCMPGTQ",
|
||||
"VPCMPB", "VPCMPW", "VPCMPD", "VPCMPQ",
|
||||
} {
|
||||
t = append(t, i(m, "Packed compare"))
|
||||
}
|
||||
|
||||
// SIMD unpack, shuffle, permute, broadcast, extract, insert.
|
||||
for _, m := range []string{
|
||||
"VPUNPCKLBW", "VPUNPCKLWD", "VPUNPCKLDQ", "VPUNPCKLQDQ",
|
||||
"VPUNPCKHBW", "VPUNPCKHWD", "VPUNPCKHDQ", "VPUNPCKHQDQ",
|
||||
"VPSHUFD", "VPSHUFHW", "VPSHUFLW", "VPSHUFB",
|
||||
"VEXTRACTI128", "VEXTRACTF128", "VEXTRACTI32X4", "VEXTRACTI64X4",
|
||||
"VEXTRACTF32X4", "VEXTRACTF64X4", "VEXTRACTI32X8", "VEXTRACTI64X2",
|
||||
"VINSERTI128", "VINSERTF128", "VINSERTI32X4", "VINSERTI64X4", "VINSERTF32X4", "VINSERTF64X4",
|
||||
"VPERMQ", "VPERMD", "VPERMPS", "VPERM2I128", "VPERM2F128",
|
||||
"VPBROADCASTD", "VPBROADCASTQ", "VPBROADCASTB", "VPBROADCASTW",
|
||||
"VBROADCASTSD", "VBROADCASTSS", "VBROADCASTI128", "VBROADCASTI32X4",
|
||||
"VALIGND", "VALIGNQ", "VPBLENDD", "VPBLENDW", "VBLENDVPD", "VBLENDVPS",
|
||||
"VSHUFPD", "VSHUFPS",
|
||||
} {
|
||||
t = append(t, i(m, "Shuffle / permute / broadcast"))
|
||||
}
|
||||
|
||||
// SIMD sign/zero extension and truncation.
|
||||
for _, m := range []string{
|
||||
"VPMOVSXBW", "VPMOVSXBD", "VPMOVSXBQ", "VPMOVSXWD", "VPMOVSXWQ", "VPMOVSXDQ",
|
||||
"VPMOVZXBW", "VPMOVZXBD", "VPMOVZXBQ", "VPMOVZXWD", "VPMOVZXWQ", "VPMOVZXDQ",
|
||||
"VPMOVDW", "VPMOVQW", "VPMOVQD", "VPMOVDB", "VPMOVWB", "VPMOVQB",
|
||||
"VPMOVMSKB", "VMOVMSKPS", "VMOVMSKPD", "VMOVQ2DQ", "VMOVDQ2Q",
|
||||
} {
|
||||
t = append(t, i(m, "Packed extend / truncate / mask"))
|
||||
}
|
||||
|
||||
// SIMD floating point.
|
||||
for _, m := range []string{
|
||||
"VADDPD", "VADDPS", "VADDSD", "VADDSS",
|
||||
"VSUBPD", "VSUBPS", "VSUBSD", "VSUBSS",
|
||||
"VMULPD", "VMULPS", "VMULSD", "VMULSS",
|
||||
"VDIVPD", "VDIVPS", "VDIVSD", "VDIVSS",
|
||||
"VMINPD", "VMINPS", "VMINSD", "VMINSS", "VMAXPD", "VMAXPS", "VMAXSD", "VMAXSS",
|
||||
"VXORPD", "VXORPS", "VANDPD", "VANDPS", "VANDNPD", "VANDNPS", "VORPD", "VORPS",
|
||||
"VUNPCKHPD", "VUNPCKLPD", "VUNPCKHPS", "VUNPCKLPS",
|
||||
"VSQRTPD", "VSQRTPS", "VSQRTSD", "VSQRTSS", "VRSQRTPS", "VRCPPS",
|
||||
"VCMPPD", "VCMPPS", "VCMPSD", "VCMPSS",
|
||||
} {
|
||||
t = append(t, i(m, "Packed/scalar floating point"))
|
||||
}
|
||||
|
||||
// FMA.
|
||||
for _, ord := range []string{"132", "213", "231"} {
|
||||
for _, sfx := range []string{"PD", "PS", "SD", "SS"} {
|
||||
t = append(t, i("VFMADD"+ord+sfx, "Fused multiply-add"))
|
||||
t = append(t, i("VFMSUB"+ord+sfx, "Fused multiply-subtract"))
|
||||
t = append(t, i("VFNMADD"+ord+sfx, "Fused negated multiply-add"))
|
||||
t = append(t, i("VFNMSUB"+ord+sfx, "Fused negated multiply-subtract"))
|
||||
}
|
||||
}
|
||||
|
||||
// SIMD conversion.
|
||||
for _, m := range []string{
|
||||
"VCVTDQ2PD", "VCVTDQ2PS", "VCVTPD2DQ", "VCVTPS2DQ", "VCVTPD2PS", "VCVTPS2PD",
|
||||
"VCVTQQ2PD", "VCVTQQ2PS", "VCVTUQQ2PD", "VCVTUQQ2PS",
|
||||
"VCVTTPD2DQ", "VCVTTPS2DQ", "VCVTSI2SD", "VCVTSI2SS", "VCVTSD2SI", "VCVTSS2SI",
|
||||
"VCVTSD2SS", "VCVTSS2SD",
|
||||
"CVTSL2SD", "CVTSL2SS", "CVTSQ2SD", "CVTSQ2SS", "CVTTSD2SL", "CVTTSD2SQ", "CVTTSS2SL",
|
||||
} {
|
||||
t = append(t, i(m, "Numeric conversion"))
|
||||
}
|
||||
|
||||
// SIMD zeroing.
|
||||
t = append(t, ic("VZEROUPPER", "Zero upper halves of YMM/ZMM", 0, 0))
|
||||
t = append(t, ic("VZEROALL", "Zero all YMM/ZMM state", 0, 0))
|
||||
|
||||
// AVX-512 mask register operations.
|
||||
for _, s := range []string{"B", "W", "D", "Q"} {
|
||||
t = append(t, ic("KMOV"+s, "Move mask register", 2, 2))
|
||||
t = append(t, ic("KTEST"+s, "Test mask registers", 2, 2))
|
||||
t = append(t, i("KAND"+s, "AND masks"))
|
||||
t = append(t, i("KOR"+s, "OR masks"))
|
||||
t = append(t, i("KXOR"+s, "XOR masks"))
|
||||
t = append(t, i("KNOT"+s, "NOT mask"))
|
||||
t = append(t, i("KANDN"+s, "AND-NOT masks"))
|
||||
t = append(t, i("KUNPCK"+s, "Unpack masks"))
|
||||
}
|
||||
|
||||
return t
|
||||
}
|
||||
+1609
File diff suppressed because it is too large
Load Diff
+275
@@ -0,0 +1,275 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package arch provides architecture-specific metadata for GAsm: the register
|
||||
// files and instruction tables for amd64 and arm64. The metadata powers
|
||||
// completion, hover documentation, semantic highlighting and the "unknown
|
||||
// instruction" lint. It is pure data with no dependency on the parser, so it
|
||||
// can be consulted from any layer.
|
||||
package arch
|
||||
|
||||
import (
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Arch identifies a target instruction set.
|
||||
type Arch string
|
||||
|
||||
// Supported architectures.
|
||||
const (
|
||||
AMD64 Arch = "amd64"
|
||||
ARM64 Arch = "arm64"
|
||||
RISCV Arch = "riscv"
|
||||
LOONG64 Arch = "loong64"
|
||||
Unknown Arch = ""
|
||||
)
|
||||
|
||||
// FromFilename guesses the target architecture from a source file name. Go
|
||||
// assembly files conventionally carry a GOARCH suffix such as "_amd64.s",
|
||||
// "_arm64.s", "_riscv64.s" or "_loong64.s". It returns Unknown when no suffix
|
||||
// matches.
|
||||
func FromFilename(name string) Arch {
|
||||
lower := strings.ToLower(name)
|
||||
switch {
|
||||
case strings.Contains(lower, "_amd64"):
|
||||
return AMD64
|
||||
case strings.Contains(lower, "_arm64"):
|
||||
return ARM64
|
||||
case strings.Contains(lower, "_riscv64"), strings.Contains(lower, "_riscv"):
|
||||
return RISCV
|
||||
case strings.Contains(lower, "_loong64"), strings.Contains(lower, "_loong"):
|
||||
return LOONG64
|
||||
default:
|
||||
return Unknown
|
||||
}
|
||||
}
|
||||
|
||||
// RegClass classifies a register for highlighting and completion grouping.
|
||||
type RegClass int
|
||||
|
||||
// Register classes.
|
||||
const (
|
||||
GPR RegClass = iota // general-purpose integer register
|
||||
GPRSub // sized sub-register (AL, R8D, …)
|
||||
Vector // SSE/AVX/AVX-512 vector (X/Y/Z)
|
||||
Mask // AVX-512 mask register (K)
|
||||
Float // arm64 floating-point register (F)
|
||||
VecARM // arm64 SIMD/vector register (V)
|
||||
Special // architecture-special register
|
||||
)
|
||||
|
||||
// String returns a short label for the class.
|
||||
func (c RegClass) String() string {
|
||||
switch c {
|
||||
case GPR:
|
||||
return "general-purpose"
|
||||
case GPRSub:
|
||||
return "sub-register"
|
||||
case Vector:
|
||||
return "vector"
|
||||
case Mask:
|
||||
return "mask"
|
||||
case Float:
|
||||
return "float"
|
||||
case VecARM:
|
||||
return "vector (arm64)"
|
||||
case Special:
|
||||
return "special"
|
||||
default:
|
||||
return "register"
|
||||
}
|
||||
}
|
||||
|
||||
// Register describes one architectural register.
|
||||
type Register struct {
|
||||
Name string
|
||||
Class RegClass
|
||||
Desc string
|
||||
}
|
||||
|
||||
// Instr describes one instruction mnemonic.
|
||||
type Instr struct {
|
||||
Name string
|
||||
Summary string
|
||||
// MinOps and MaxOps bound the operand count; -1 means "unknown/variable"
|
||||
// and disables the operand-count lint for that instruction.
|
||||
MinOps int
|
||||
MaxOps int
|
||||
}
|
||||
|
||||
// Table is the metadata for one architecture.
|
||||
type Table struct {
|
||||
Arch Arch
|
||||
regs map[string]Register
|
||||
regList []Register
|
||||
instrs map[string]Instr
|
||||
instrList []Instr
|
||||
}
|
||||
|
||||
func newTable(a Arch, regs []Register, instrs []Instr) *Table {
|
||||
t := &Table{
|
||||
Arch: a,
|
||||
regs: make(map[string]Register, len(regs)),
|
||||
regList: regs,
|
||||
instrs: make(map[string]Instr, len(instrs)),
|
||||
instrList: instrs,
|
||||
}
|
||||
for _, r := range regs {
|
||||
t.regs[strings.ToUpper(r.Name)] = r
|
||||
}
|
||||
for _, in := range instrs {
|
||||
t.instrs[strings.ToUpper(in.Name)] = in
|
||||
}
|
||||
return t
|
||||
}
|
||||
|
||||
// IsRegister reports whether name is a register of this architecture.
|
||||
func (t *Table) IsRegister(name string) bool {
|
||||
_, ok := t.regs[strings.ToUpper(name)]
|
||||
return ok
|
||||
}
|
||||
|
||||
// Register returns the named register.
|
||||
func (t *Table) Register(name string) (Register, bool) {
|
||||
r, ok := t.regs[strings.ToUpper(name)]
|
||||
return r, ok
|
||||
}
|
||||
|
||||
// Registers returns all registers in definition order.
|
||||
func (t *Table) Registers() []Register { return t.regList }
|
||||
|
||||
// Lookup returns the metadata for a mnemonic (case-insensitive).
|
||||
func (t *Table) Lookup(mnemonic string) (Instr, bool) {
|
||||
key := strings.ToUpper(mnemonic)
|
||||
if in, ok := t.instrs[key]; ok {
|
||||
return in, true
|
||||
}
|
||||
// arm64 load/store instructions take a .P (post-index) or .W (pre-index)
|
||||
// addressing suffix that the assembler front-end strips; mirror that so the
|
||||
// base instruction is still recognised.
|
||||
if t.Arch == ARM64 {
|
||||
for _, suffix := range []string{".P", ".W"} {
|
||||
if base, ok := strings.CutSuffix(key, suffix); ok {
|
||||
if in, found := t.instrs[base]; found {
|
||||
return in, true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return Instr{}, false
|
||||
}
|
||||
|
||||
// Instructions returns all instructions in definition order.
|
||||
func (t *Table) Instructions() []Instr { return t.instrList }
|
||||
|
||||
// pseudoRegs are the Plan 9 pseudo-registers, valid on every architecture.
|
||||
var pseudoRegs = map[string]string{
|
||||
"FP": "frame pointer: references function arguments and results",
|
||||
"SP": "stack pointer: the top of the local stack frame",
|
||||
"SB": "static base: references global symbols",
|
||||
"PC": "program counter",
|
||||
}
|
||||
|
||||
// IsPseudoReg reports whether name is a Plan 9 pseudo-register.
|
||||
func IsPseudoReg(name string) bool {
|
||||
_, ok := pseudoRegs[strings.ToUpper(name)]
|
||||
return ok
|
||||
}
|
||||
|
||||
// PseudoRegDesc returns the description of a pseudo-register.
|
||||
func PseudoRegDesc(name string) (string, bool) {
|
||||
d, ok := pseudoRegs[strings.ToUpper(name)]
|
||||
return d, ok
|
||||
}
|
||||
|
||||
var (
|
||||
amd64Table *Table
|
||||
arm64Table *Table
|
||||
riscvTable *Table
|
||||
loong64Table *Table
|
||||
)
|
||||
|
||||
func init() {
|
||||
amd64Table = buildAMD64()
|
||||
arm64Table = buildARM64()
|
||||
riscvTable = buildRISCV()
|
||||
loong64Table = buildLOONG64()
|
||||
}
|
||||
|
||||
// ForArch returns the table for a, or the amd64 table for Unknown so that
|
||||
// callers always get a usable default.
|
||||
func ForArch(a Arch) *Table {
|
||||
switch a {
|
||||
case ARM64:
|
||||
return arm64Table
|
||||
case RISCV:
|
||||
return riscvTable
|
||||
case LOONG64:
|
||||
return loong64Table
|
||||
default:
|
||||
return amd64Table
|
||||
}
|
||||
}
|
||||
|
||||
// fixedArity lists the few instructions whose operand count is reliable on
|
||||
// every architecture; relaxCounts leaves these untouched.
|
||||
var fixedArity = map[string]bool{
|
||||
"RET": true, "NOP": true, "JMP": true, "CALL": true, "UNDEF": true,
|
||||
}
|
||||
|
||||
// relaxCounts clears operand-count bounds for every instruction except the
|
||||
// fixed-arity ones. It is applied to architectures (arm64, riscv64, loong64)
|
||||
// whose instructions have too many operand forms for a single fixed count to be
|
||||
// reliable, so the operand-count lint stays silent rather than guess.
|
||||
func relaxCounts(instrs []Instr) []Instr {
|
||||
for i := range instrs {
|
||||
if !fixedArity[strings.ToUpper(instrs[i].Name)] {
|
||||
instrs[i].MinOps = -1
|
||||
instrs[i].MaxOps = -1
|
||||
}
|
||||
}
|
||||
return instrs
|
||||
}
|
||||
|
||||
// mergedInstrs combines the common opcode list with an architecture-specific
|
||||
// list (de-duplicated, common first) and enriches the result with the curated
|
||||
// summaries map.
|
||||
func mergedInstrs(summaries map[string]Instr, nameSets ...[]string) []Instr {
|
||||
seen := make(map[string]bool)
|
||||
var names []string
|
||||
for _, set := range nameSets {
|
||||
for _, n := range set {
|
||||
if !seen[n] {
|
||||
seen[n] = true
|
||||
names = append(names, n)
|
||||
}
|
||||
}
|
||||
}
|
||||
return buildInstrs(names, summaries)
|
||||
}
|
||||
|
||||
// buildInstrs merges the complete generated instruction name list with a
|
||||
// curated summaries map (keyed by upper-case mnemonic). Instructions without a
|
||||
// curated entry get an empty summary and an unknown operand count, which keeps
|
||||
// the operand-count lint silent for them.
|
||||
func buildInstrs(names []string, summaries map[string]Instr) []Instr {
|
||||
out := make([]Instr, 0, len(names))
|
||||
for _, n := range names {
|
||||
if in, ok := summaries[strings.ToUpper(n)]; ok {
|
||||
in.Name = n
|
||||
out = append(out, in)
|
||||
} else {
|
||||
out = append(out, Instr{Name: n, MinOps: -1, MaxOps: -1})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// toMap converts a curated instruction slice into an upper-case-keyed map.
|
||||
func toMap(list []Instr) map[string]Instr {
|
||||
m := make(map[string]Instr, len(list))
|
||||
for _, in := range list {
|
||||
m[strings.ToUpper(in.Name)] = in
|
||||
}
|
||||
return m
|
||||
}
|
||||
@@ -0,0 +1,161 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package arch
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestFromFilename(t *testing.T) {
|
||||
cases := map[string]Arch{
|
||||
"avx2_amd64.s": AMD64,
|
||||
"foo_arm64.s": ARM64,
|
||||
"portable.s": Unknown,
|
||||
"decode_ARM64.S": ARM64,
|
||||
"kernels_amd64.s": AMD64,
|
||||
"kernel_riscv64.s": RISCV,
|
||||
"kernel_loong64.s": LOONG64,
|
||||
}
|
||||
for name, want := range cases {
|
||||
if got := FromFilename(name); got != want {
|
||||
t.Errorf("FromFilename(%q) = %q, want %q", name, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAMD64Registers(t *testing.T) {
|
||||
tab := ForArch(AMD64)
|
||||
for _, reg := range []string{"AX", "BX", "R15", "AL", "X0", "Y15", "Z31", "K7"} {
|
||||
if !tab.IsRegister(reg) {
|
||||
t.Errorf("amd64: %s should be a register", reg)
|
||||
}
|
||||
}
|
||||
for _, not := range []string{"vec1", "R16", "Z32", "K8", "swin_base"} {
|
||||
if tab.IsRegister(not) {
|
||||
t.Errorf("amd64: %s should NOT be a register", not)
|
||||
}
|
||||
}
|
||||
if r, ok := tab.Register("Y0"); !ok || r.Class != Vector {
|
||||
t.Errorf("Y0 class = %+v, want Vector", r)
|
||||
}
|
||||
if r, ok := tab.Register("K1"); !ok || r.Class != Mask {
|
||||
t.Errorf("K1 class = %+v, want Mask", r)
|
||||
}
|
||||
}
|
||||
|
||||
func TestARM64Registers(t *testing.T) {
|
||||
tab := ForArch(ARM64)
|
||||
for _, reg := range []string{"R0", "R30", "SP", "ZR", "F0", "V31"} {
|
||||
if !tab.IsRegister(reg) {
|
||||
t.Errorf("arm64: %s should be a register", reg)
|
||||
}
|
||||
}
|
||||
if tab.IsRegister("AX") {
|
||||
t.Error("arm64: AX must not be a register")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAMD64Instructions(t *testing.T) {
|
||||
tab := ForArch(AMD64)
|
||||
// Every mnemonic used in the go-flac kernels must be known.
|
||||
used := []string{
|
||||
"MOVQ", "MOVL", "MOVB", "LEAQ", "ADDQ", "SUBL", "ANDQ", "ORL", "XORL",
|
||||
"CMPQ", "CMPL", "TESTQ", "IMUL3L", "IMULQ", "INCW", "LZCNTL", "CMOVLGT",
|
||||
"SETNE", "JMP", "JGE", "JNE", "JLE", "JLT", "JGT", "JZ", "JNZ", "RET",
|
||||
"VPCMPEQD", "VPSLLD", "VPXOR", "VPSUBD", "VPADDD", "VPADDQ", "VMOVDQU",
|
||||
"VEXTRACTI128", "VPSHUFD", "VMOVQ", "VMOVMSKPS", "VZEROUPPER", "VPUNPCKLDQ",
|
||||
"VPUNPCKHDQ", "VPSRAD", "VPOR", "VPMOVSXDQ", "VPMULDQ", "VPCMPGTQ", "VPSRLQ",
|
||||
"VPANDN", "VPMOVMSKB", "VPBROADCASTD", "VPSHUFB", "VPACKSSDW", "VPERMQ",
|
||||
"VPERM2I128", "VPMOVZXDQ", "VFMADD231PD", "VCVTDQ2PD", "VMOVUPD", "VMULPD",
|
||||
"VADDPD", "VADDSD", "VMOVSD", "VMULSD", "CVTSL2SD", "VEXTRACTF128",
|
||||
"VPMOVSXWD", "MOVWLSX", "MOVBLZX", "MOVLQSX",
|
||||
// AVX-512.
|
||||
"VMOVDQU32", "VPXORD", "VALIGND", "VPERMD", "VPMULLD", "VPMOVDW", "VPSRAQ",
|
||||
"KTESTW", "KMOVW", "VPXORQ", "VPMOVQD", "VEXTRACTF64X4", "VEXTRACTI64X4",
|
||||
"VPBROADCASTQ", "VPMULLQ", "VPCMPEQD",
|
||||
}
|
||||
for _, m := range used {
|
||||
if _, ok := tab.Lookup(m); !ok {
|
||||
t.Errorf("amd64: instruction %s is missing from the table", m)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestOperandCounts(t *testing.T) {
|
||||
tab := ForArch(AMD64)
|
||||
if in, _ := tab.Lookup("RET"); in.MinOps != 0 || in.MaxOps != 0 {
|
||||
t.Errorf("RET counts = %d/%d, want 0/0", in.MinOps, in.MaxOps)
|
||||
}
|
||||
if in, _ := tab.Lookup("JMP"); in.MinOps != 1 || in.MaxOps != 1 {
|
||||
t.Errorf("JMP counts = %d/%d, want 1/1", in.MinOps, in.MaxOps)
|
||||
}
|
||||
if in, _ := tab.Lookup("MOVQ"); in.MinOps != 2 || in.MaxOps != 2 {
|
||||
t.Errorf("MOVQ counts = %d/%d, want 2/2", in.MinOps, in.MaxOps)
|
||||
}
|
||||
if in, _ := tab.Lookup("IMUL3L"); in.MinOps != 3 || in.MaxOps != 3 {
|
||||
t.Errorf("IMUL3L counts = %d/%d, want 3/3", in.MinOps, in.MaxOps)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPseudoRegs(t *testing.T) {
|
||||
for _, p := range []string{"FP", "SP", "SB", "PC"} {
|
||||
if !IsPseudoReg(p) {
|
||||
t.Errorf("%s should be a pseudo-register", p)
|
||||
}
|
||||
}
|
||||
if IsPseudoReg("AX") {
|
||||
t.Error("AX must not be a pseudo-register")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCVRegisters(t *testing.T) {
|
||||
tab := ForArch(RISCV)
|
||||
for _, reg := range []string{"X0", "X31", "F0", "F31", "ZERO", "RA", "SP", "A0", "S11", "T6", "FA0"} {
|
||||
if !tab.IsRegister(reg) {
|
||||
t.Errorf("riscv: %s should be a register", reg)
|
||||
}
|
||||
}
|
||||
if tab.IsRegister("AX") {
|
||||
t.Error("riscv: AX must not be a register")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLOONG64Registers(t *testing.T) {
|
||||
tab := ForArch(LOONG64)
|
||||
for _, reg := range []string{"R0", "R31", "F0", "F31", "V0", "V31", "X0", "X31"} {
|
||||
if !tab.IsRegister(reg) {
|
||||
t.Errorf("loong64: %s should be a register", reg)
|
||||
}
|
||||
}
|
||||
if tab.IsRegister("AX") {
|
||||
t.Error("loong64: AX must not be a register")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRISCVInstructions(t *testing.T) {
|
||||
tab := ForArch(RISCV)
|
||||
for _, m := range []string{"ADD", "ADDI", "SUB", "MUL", "DIV", "BEQ", "BNE", "JAL", "JALR", "LW", "SW", "FADDD", "AMOSWAPD"} {
|
||||
if _, ok := tab.Lookup(m); !ok {
|
||||
t.Errorf("riscv: instruction %s is missing", m)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestLOONG64Instructions(t *testing.T) {
|
||||
tab := ForArch(LOONG64)
|
||||
for _, m := range []string{"ADD", "ADDD", "SUBD", "MULD", "BEQ", "BNE", "BGE", "BGEZ", "JIRL", "MOVD", "MOVW"} {
|
||||
if _, ok := tab.Lookup(m); !ok {
|
||||
t.Errorf("loong64: instruction %s is missing", m)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestGeneratedTableSize sanity-checks that the toolchain-derived tables are
|
||||
// the full instruction sets, not a partial hand-written subset.
|
||||
func TestGeneratedTableSize(t *testing.T) {
|
||||
min := map[Arch]int{AMD64: 1000, ARM64: 400, RISCV: 800, LOONG64: 600}
|
||||
for a, want := range min {
|
||||
if n := len(ForArch(a).Instructions()); n < want {
|
||||
t.Errorf("%s: only %d instructions, want >= %d (generation incomplete?)", a, n, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
+181
@@ -0,0 +1,181 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package arch
|
||||
|
||||
import "fmt"
|
||||
|
||||
func buildARM64() *Table {
|
||||
return newTable(ARM64, arm64Registers(), relaxCounts(mergedInstrs(arm64Summaries(), commonGeneratedInstrs, arm64GeneratedInstrs, arm64Aliases())))
|
||||
}
|
||||
|
||||
// arm64Aliases are the branch/jump spellings the assembler front-end accepts in
|
||||
// addition to the generated opcode table (notably the unconditional B and BL).
|
||||
func arm64Aliases() []string {
|
||||
return []string{
|
||||
"B", "BL", "BCS", "BHS", "BCC", "BLO", "BMI", "BPL", "BVS", "BVC",
|
||||
"BHI", "BLS", "CBZW", "CBNZW", "ADR", "ADRP",
|
||||
}
|
||||
}
|
||||
|
||||
// arm64Summaries returns the curated documentation/operand-count table keyed by
|
||||
// upper-case mnemonic; it enriches the complete generated name list.
|
||||
func arm64Summaries() map[string]Instr { return toMap(arm64Curated()) }
|
||||
|
||||
// arm64Registers builds the arm64 (AArch64) register file.
|
||||
func arm64Registers() []Register {
|
||||
var regs []Register
|
||||
add := func(name string, class RegClass, desc string) {
|
||||
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
|
||||
}
|
||||
|
||||
// General-purpose integer registers R0–R30.
|
||||
for i := 0; i <= 30; i++ {
|
||||
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
|
||||
}
|
||||
add("ZR", Special, "zero register (reads as 0)")
|
||||
add("SP", Special, "stack pointer")
|
||||
add("LR", Special, "link register (alias of R30)")
|
||||
add("PC", Special, "program counter")
|
||||
add("RSP", Special, "stack pointer (alias)")
|
||||
|
||||
// Floating-point / SIMD registers: F (scalar FP) and V (vector).
|
||||
for i := 0; i <= 31; i++ {
|
||||
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
|
||||
add(fmt.Sprintf("V%d", i), VecARM, "128-bit SIMD/vector register")
|
||||
}
|
||||
return regs
|
||||
}
|
||||
|
||||
// arm64Curated returns the hand-written subset of arm64 (AArch64) instructions
|
||||
// that carry a summary and/or an operand-count range. 32-bit operations carry
|
||||
// a W suffix. The authoritative, complete set is arm64GeneratedInstrs.
|
||||
func arm64Curated() []Instr {
|
||||
var t []Instr
|
||||
|
||||
// Data movement (loads and stores are MOVx with a memory operand).
|
||||
for _, m := range []string{
|
||||
"MOVB", "MOVBU", "MOVH", "MOVHU", "MOVW", "MOVWU", "MOVD",
|
||||
"FMOVS", "FMOVD",
|
||||
} {
|
||||
t = append(t, ic(m, "Move / load / store", 2, 2))
|
||||
}
|
||||
for _, m := range []string{"MOVK", "MOVN", "MOVZ", "MOVKW", "MOVNW", "MOVZW"} {
|
||||
t = append(t, i(m, "Move wide constant"))
|
||||
}
|
||||
for _, m := range []string{"ADR", "ADRP"} {
|
||||
t = append(t, ic(m, "Address of label/page", 2, 2))
|
||||
}
|
||||
|
||||
// Integer arithmetic and logic (64-bit and W 32-bit forms).
|
||||
for _, op := range []string{"ADD", "ADDS", "SUB", "SUBS", "AND", "ANDS", "ORR", "ORN", "EOR", "EON", "BIC", "BICS", "ADC", "ADCS", "SBC", "SBCS"} {
|
||||
t = append(t, i(op, op+" (64-bit)"))
|
||||
t = append(t, i(op+"W", op+" (32-bit)"))
|
||||
}
|
||||
for _, op := range []string{"NEG", "NGC", "MVN"} {
|
||||
t = append(t, i(op, op+" (64-bit)"))
|
||||
t = append(t, i(op+"W", op+" (32-bit)"))
|
||||
}
|
||||
for _, op := range []string{"MUL", "MNEG", "SMULL", "UMULL", "SMULH", "UMULH", "MADD", "MSUB", "SMADDL", "UMADDL", "SMSUBL", "UMSUBL"} {
|
||||
t = append(t, i(op, "Multiply / multiply-accumulate"))
|
||||
}
|
||||
for _, op := range []string{"UDIV", "SDIV", "UDIVW", "SDIVW"} {
|
||||
t = append(t, ic(op, "Divide", 3, 3))
|
||||
}
|
||||
|
||||
// Shifts, rotates and bit manipulation.
|
||||
for _, op := range []string{"LSL", "LSR", "ASR", "ROR"} {
|
||||
t = append(t, i(op, op+" shift"))
|
||||
t = append(t, i(op+"W", op+" shift (32-bit)"))
|
||||
}
|
||||
for _, op := range []string{"LSLV", "LSRV", "ASRV", "RORV", "LSLVW", "LSRVW", "ASRVW", "RORVW"} {
|
||||
t = append(t, i(op, "Variable shift"))
|
||||
}
|
||||
for _, op := range []string{"RBIT", "REV", "REV16", "REV32", "REV64", "CLZ", "CLS", "RBITW", "REVW", "CLZW", "CLSW"} {
|
||||
t = append(t, ic(op, "Bit manipulation", 2, 2))
|
||||
}
|
||||
for _, op := range []string{"UBFX", "SBFX", "UBFM", "SBFM", "BFXIL", "EXTR"} {
|
||||
t = append(t, i(op, "Bitfield extract"))
|
||||
}
|
||||
|
||||
// Compare and test.
|
||||
for _, op := range []string{"CMP", "CMN", "TST"} {
|
||||
t = append(t, i(op, op+" (64-bit)"))
|
||||
t = append(t, i(op+"W", op+" (32-bit)"))
|
||||
}
|
||||
|
||||
// Conditional select.
|
||||
for _, op := range []string{"CSEL", "CSINC", "CSINV", "CSNEG", "CSET", "CSETM", "CINC", "CINV", "CNEG"} {
|
||||
t = append(t, i(op, "Conditional select"))
|
||||
t = append(t, i(op+"W", "Conditional select (32-bit)"))
|
||||
}
|
||||
for _, op := range []string{"CCMP", "CCMN", "CCMPW", "CCMNW"} {
|
||||
t = append(t, i(op, "Conditional compare"))
|
||||
}
|
||||
|
||||
// Control flow.
|
||||
t = append(t, ic("B", "Unconditional branch", 1, 1))
|
||||
t = append(t, ic("BL", "Branch with link", 1, 1))
|
||||
for _, cc := range []string{
|
||||
"EQ", "NE", "CS", "HS", "CC", "LO", "MI", "PL", "VS", "VC",
|
||||
"HI", "LS", "GE", "LT", "GT", "LE", "AL", "NV",
|
||||
} {
|
||||
t = append(t, ic("B"+cc, "Conditional branch", 1, 1))
|
||||
}
|
||||
for _, op := range []string{"CBZ", "CBNZ", "TBZ", "TBNZ"} {
|
||||
t = append(t, i(op, "Compare/test and branch"))
|
||||
t = append(t, i(op+"W", "Compare/test and branch (32-bit)"))
|
||||
}
|
||||
t = append(t, ic("RET", "Return", 0, 1))
|
||||
t = append(t, ic("BR", "Branch to register", 1, 1))
|
||||
t = append(t, ic("BLR", "Branch with link to register", 1, 1))
|
||||
t = append(t, ic("NOP", "No operation", 0, 1))
|
||||
t = append(t, ic("BRK", "Breakpoint", 0, 1))
|
||||
for _, op := range []string{"SVC", "HVC", "SMC"} {
|
||||
t = append(t, i(op, "Exception generation"))
|
||||
}
|
||||
for _, op := range []string{"DMB", "DSB", "ISB"} {
|
||||
t = append(t, i(op, "Barrier"))
|
||||
}
|
||||
for _, op := range []string{"MRS", "MSR"} {
|
||||
t = append(t, ic(op, "System register access", 2, 2))
|
||||
}
|
||||
|
||||
// Atomics (LSE and load-exclusive/store-exclusive).
|
||||
for _, op := range []string{
|
||||
"LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW",
|
||||
"LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW",
|
||||
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL", "CASAL",
|
||||
} {
|
||||
t = append(t, i(op, "Atomic memory operation"))
|
||||
}
|
||||
|
||||
// Floating-point scalar.
|
||||
for _, op := range []string{
|
||||
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
|
||||
"FMADD", "FMSUB", "FNMADD", "FNMSUB", "FCMP", "FCMPE",
|
||||
"FCVT", "FCVTZS", "FCVTZU", "FCVTNS", "FCVTNU", "FCVTAS", "FCVTAU",
|
||||
"SCVTF", "UCVTF", "FRINTM", "FRINTN", "FRINTP", "FRINTZ",
|
||||
} {
|
||||
t = append(t, i(op, "Floating-point operation"))
|
||||
}
|
||||
t = append(t, i("FMOV", "Floating-point move"))
|
||||
|
||||
// NEON / SIMD vector (arrangement carried by the operand suffix).
|
||||
for _, op := range []string{
|
||||
"VADD", "VSUB", "VMUL", "VMLA", "VMLS", "VNEG", "VABS", "VMIN", "VMAX",
|
||||
"VAND", "VORR", "VEOR", "VBIC", "VBIF", "VBSL", "VNOT",
|
||||
"VDUP", "VMOV", "VMOVI", "VMOVQ",
|
||||
"VLD1", "VLD2", "VLD3", "VLD4", "VST1", "VST2", "VST3", "VST4",
|
||||
"VCNT", "VREV16", "VREV32", "VREV64", "VUZP1", "VUZP2", "VZIP1", "VZIP2", "VTRN1", "VTRN2",
|
||||
"VSHL", "VSHR", "VSSHLL", "VUSHR", "VEXT", "VTBL", "VTBX",
|
||||
"VADDV", "VUMAXV", "VUMINV", "VSMAXV", "VSMINV",
|
||||
"VFADD", "VFSUB", "VFMUL", "VFDIV", "VFNEG", "VFABS", "VFMIN", "VFMAX",
|
||||
"VFMLA", "VFMLS", "VFCVT", "VSCVTF", "VUCVTF", "VFCMEQ", "VFCMGT", "VFCMLT",
|
||||
"VCMPEQ", "VCMPGT", "VCMPGE", "VSHLL",
|
||||
} {
|
||||
t = append(t, i(op, "NEON SIMD vector operation"))
|
||||
}
|
||||
|
||||
return t
|
||||
}
|
||||
@@ -0,0 +1,547 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
|
||||
|
||||
package arch
|
||||
|
||||
// arm64GeneratedInstrs is the complete set of arm64 mnemonics accepted by
|
||||
// Go's Plan 9 assembler.
|
||||
var arm64GeneratedInstrs = []string{
|
||||
"ADC",
|
||||
"ADCS",
|
||||
"ADCSW",
|
||||
"ADCW",
|
||||
"ADD",
|
||||
"ADDS",
|
||||
"ADDSW",
|
||||
"ADDW",
|
||||
"ADR",
|
||||
"ADRP",
|
||||
"AESD",
|
||||
"AESE",
|
||||
"AESIMC",
|
||||
"AESMC",
|
||||
"AND",
|
||||
"ANDS",
|
||||
"ANDSW",
|
||||
"ANDW",
|
||||
"ASR",
|
||||
"ASRW",
|
||||
"AT",
|
||||
"AUTIA1716",
|
||||
"AUTIASP",
|
||||
"AUTIB1716",
|
||||
"AUTIBSP",
|
||||
"BCC",
|
||||
"BCS",
|
||||
"BEQ",
|
||||
"BFI",
|
||||
"BFIW",
|
||||
"BFM",
|
||||
"BFMW",
|
||||
"BFXIL",
|
||||
"BFXILW",
|
||||
"BGE",
|
||||
"BGT",
|
||||
"BHI",
|
||||
"BHS",
|
||||
"BIC",
|
||||
"BICS",
|
||||
"BICSW",
|
||||
"BICW",
|
||||
"BLE",
|
||||
"BLO",
|
||||
"BLS",
|
||||
"BLT",
|
||||
"BMI",
|
||||
"BNE",
|
||||
"BPL",
|
||||
"BRK",
|
||||
"BTI",
|
||||
"BVC",
|
||||
"BVS",
|
||||
"CASAD",
|
||||
"CASALB",
|
||||
"CASALD",
|
||||
"CASALH",
|
||||
"CASALW",
|
||||
"CASAW",
|
||||
"CASB",
|
||||
"CASD",
|
||||
"CASH",
|
||||
"CASLD",
|
||||
"CASLW",
|
||||
"CASPD",
|
||||
"CASPW",
|
||||
"CASW",
|
||||
"CBNZ",
|
||||
"CBNZW",
|
||||
"CBZ",
|
||||
"CBZW",
|
||||
"CCMN",
|
||||
"CCMNW",
|
||||
"CCMP",
|
||||
"CCMPW",
|
||||
"CINC",
|
||||
"CINCW",
|
||||
"CINV",
|
||||
"CINVW",
|
||||
"CLREX",
|
||||
"CLS",
|
||||
"CLSW",
|
||||
"CLZ",
|
||||
"CLZW",
|
||||
"CMN",
|
||||
"CMNW",
|
||||
"CMP",
|
||||
"CMPW",
|
||||
"CNEG",
|
||||
"CNEGW",
|
||||
"CRC32B",
|
||||
"CRC32CB",
|
||||
"CRC32CH",
|
||||
"CRC32CW",
|
||||
"CRC32CX",
|
||||
"CRC32H",
|
||||
"CRC32W",
|
||||
"CRC32X",
|
||||
"CSEL",
|
||||
"CSELW",
|
||||
"CSET",
|
||||
"CSETM",
|
||||
"CSETMW",
|
||||
"CSETW",
|
||||
"CSINC",
|
||||
"CSINCW",
|
||||
"CSINV",
|
||||
"CSINVW",
|
||||
"CSNEG",
|
||||
"CSNEGW",
|
||||
"DC",
|
||||
"DCPS1",
|
||||
"DCPS2",
|
||||
"DCPS3",
|
||||
"DMB",
|
||||
"DRPS",
|
||||
"DSB",
|
||||
"DWORD",
|
||||
"EON",
|
||||
"EONW",
|
||||
"EOR",
|
||||
"EORW",
|
||||
"ERET",
|
||||
"EXTR",
|
||||
"EXTRW",
|
||||
"FABSD",
|
||||
"FABSS",
|
||||
"FADDD",
|
||||
"FADDS",
|
||||
"FCCMPD",
|
||||
"FCCMPED",
|
||||
"FCCMPES",
|
||||
"FCCMPS",
|
||||
"FCMPD",
|
||||
"FCMPED",
|
||||
"FCMPES",
|
||||
"FCMPS",
|
||||
"FCSELD",
|
||||
"FCSELS",
|
||||
"FCVTDH",
|
||||
"FCVTDS",
|
||||
"FCVTHD",
|
||||
"FCVTHS",
|
||||
"FCVTSD",
|
||||
"FCVTSH",
|
||||
"FCVTZSD",
|
||||
"FCVTZSDW",
|
||||
"FCVTZSS",
|
||||
"FCVTZSSW",
|
||||
"FCVTZUD",
|
||||
"FCVTZUDW",
|
||||
"FCVTZUS",
|
||||
"FCVTZUSW",
|
||||
"FDIVD",
|
||||
"FDIVS",
|
||||
"FLDPD",
|
||||
"FLDPQ",
|
||||
"FLDPS",
|
||||
"FMADDD",
|
||||
"FMADDS",
|
||||
"FMAXD",
|
||||
"FMAXNMD",
|
||||
"FMAXNMS",
|
||||
"FMAXS",
|
||||
"FMIND",
|
||||
"FMINNMD",
|
||||
"FMINNMS",
|
||||
"FMINS",
|
||||
"FMOVD",
|
||||
"FMOVQ",
|
||||
"FMOVS",
|
||||
"FMSUBD",
|
||||
"FMSUBS",
|
||||
"FMULD",
|
||||
"FMULS",
|
||||
"FNEGD",
|
||||
"FNEGS",
|
||||
"FNMADDD",
|
||||
"FNMADDS",
|
||||
"FNMSUBD",
|
||||
"FNMSUBS",
|
||||
"FNMULD",
|
||||
"FNMULS",
|
||||
"FRINTAD",
|
||||
"FRINTAS",
|
||||
"FRINTID",
|
||||
"FRINTIS",
|
||||
"FRINTMD",
|
||||
"FRINTMS",
|
||||
"FRINTND",
|
||||
"FRINTNS",
|
||||
"FRINTPD",
|
||||
"FRINTPS",
|
||||
"FRINTXD",
|
||||
"FRINTXS",
|
||||
"FRINTZD",
|
||||
"FRINTZS",
|
||||
"FSQRTD",
|
||||
"FSQRTS",
|
||||
"FSTPD",
|
||||
"FSTPQ",
|
||||
"FSTPS",
|
||||
"FSUBD",
|
||||
"FSUBS",
|
||||
"HINT",
|
||||
"HLT",
|
||||
"HVC",
|
||||
"IC",
|
||||
"ISB",
|
||||
"LDADDAB",
|
||||
"LDADDAD",
|
||||
"LDADDAH",
|
||||
"LDADDALB",
|
||||
"LDADDALD",
|
||||
"LDADDALH",
|
||||
"LDADDALW",
|
||||
"LDADDAW",
|
||||
"LDADDB",
|
||||
"LDADDD",
|
||||
"LDADDH",
|
||||
"LDADDLB",
|
||||
"LDADDLD",
|
||||
"LDADDLH",
|
||||
"LDADDLW",
|
||||
"LDADDW",
|
||||
"LDAR",
|
||||
"LDARB",
|
||||
"LDARH",
|
||||
"LDARW",
|
||||
"LDAXP",
|
||||
"LDAXPW",
|
||||
"LDAXR",
|
||||
"LDAXRB",
|
||||
"LDAXRH",
|
||||
"LDAXRW",
|
||||
"LDCLRAB",
|
||||
"LDCLRAD",
|
||||
"LDCLRAH",
|
||||
"LDCLRALB",
|
||||
"LDCLRALD",
|
||||
"LDCLRALH",
|
||||
"LDCLRALW",
|
||||
"LDCLRAW",
|
||||
"LDCLRB",
|
||||
"LDCLRD",
|
||||
"LDCLRH",
|
||||
"LDCLRLB",
|
||||
"LDCLRLD",
|
||||
"LDCLRLH",
|
||||
"LDCLRLW",
|
||||
"LDCLRW",
|
||||
"LDEORAB",
|
||||
"LDEORAD",
|
||||
"LDEORAH",
|
||||
"LDEORALB",
|
||||
"LDEORALD",
|
||||
"LDEORALH",
|
||||
"LDEORALW",
|
||||
"LDEORAW",
|
||||
"LDEORB",
|
||||
"LDEORD",
|
||||
"LDEORH",
|
||||
"LDEORLB",
|
||||
"LDEORLD",
|
||||
"LDEORLH",
|
||||
"LDEORLW",
|
||||
"LDEORW",
|
||||
"LDORAB",
|
||||
"LDORAD",
|
||||
"LDORAH",
|
||||
"LDORALB",
|
||||
"LDORALD",
|
||||
"LDORALH",
|
||||
"LDORALW",
|
||||
"LDORAW",
|
||||
"LDORB",
|
||||
"LDORD",
|
||||
"LDORH",
|
||||
"LDORLB",
|
||||
"LDORLD",
|
||||
"LDORLH",
|
||||
"LDORLW",
|
||||
"LDORW",
|
||||
"LDP",
|
||||
"LDPSW",
|
||||
"LDPW",
|
||||
"LDXP",
|
||||
"LDXPW",
|
||||
"LDXR",
|
||||
"LDXRB",
|
||||
"LDXRH",
|
||||
"LDXRW",
|
||||
"LSL",
|
||||
"LSLW",
|
||||
"LSR",
|
||||
"LSRW",
|
||||
"MADD",
|
||||
"MADDW",
|
||||
"MNEG",
|
||||
"MNEGW",
|
||||
"MOVB",
|
||||
"MOVBU",
|
||||
"MOVD",
|
||||
"MOVH",
|
||||
"MOVHU",
|
||||
"MOVK",
|
||||
"MOVKW",
|
||||
"MOVN",
|
||||
"MOVNW",
|
||||
"MOVP",
|
||||
"MOVPD",
|
||||
"MOVPQ",
|
||||
"MOVPS",
|
||||
"MOVPSW",
|
||||
"MOVPW",
|
||||
"MOVW",
|
||||
"MOVWU",
|
||||
"MOVZ",
|
||||
"MOVZW",
|
||||
"MRS",
|
||||
"MSR",
|
||||
"MSUB",
|
||||
"MSUBW",
|
||||
"MUL",
|
||||
"MULW",
|
||||
"MVN",
|
||||
"MVNW",
|
||||
"NEG",
|
||||
"NEGS",
|
||||
"NEGSW",
|
||||
"NEGW",
|
||||
"NGC",
|
||||
"NGCS",
|
||||
"NGCSW",
|
||||
"NGCW",
|
||||
"NOOP",
|
||||
"ORN",
|
||||
"ORNW",
|
||||
"ORR",
|
||||
"ORRW",
|
||||
"PACIASP",
|
||||
"PACIBSP",
|
||||
"PRFM",
|
||||
"PRFUM",
|
||||
"RBIT",
|
||||
"RBITW",
|
||||
"REM",
|
||||
"REMW",
|
||||
"REV",
|
||||
"REV16",
|
||||
"REV16W",
|
||||
"REV32",
|
||||
"REVW",
|
||||
"ROR",
|
||||
"RORW",
|
||||
"SBC",
|
||||
"SBCS",
|
||||
"SBCSW",
|
||||
"SBCW",
|
||||
"SBFIZ",
|
||||
"SBFIZW",
|
||||
"SBFM",
|
||||
"SBFMW",
|
||||
"SBFX",
|
||||
"SBFXW",
|
||||
"SCVTFD",
|
||||
"SCVTFS",
|
||||
"SCVTFWD",
|
||||
"SCVTFWS",
|
||||
"SDIV",
|
||||
"SDIVW",
|
||||
"SEV",
|
||||
"SEVL",
|
||||
"SHA1C",
|
||||
"SHA1H",
|
||||
"SHA1M",
|
||||
"SHA1P",
|
||||
"SHA1SU0",
|
||||
"SHA1SU1",
|
||||
"SHA256H",
|
||||
"SHA256H2",
|
||||
"SHA256SU0",
|
||||
"SHA256SU1",
|
||||
"SHA512H",
|
||||
"SHA512H2",
|
||||
"SHA512SU0",
|
||||
"SHA512SU1",
|
||||
"SMADDL",
|
||||
"SMC",
|
||||
"SMNEGL",
|
||||
"SMSUBL",
|
||||
"SMULH",
|
||||
"SMULL",
|
||||
"STLR",
|
||||
"STLRB",
|
||||
"STLRH",
|
||||
"STLRW",
|
||||
"STLXP",
|
||||
"STLXPW",
|
||||
"STLXR",
|
||||
"STLXRB",
|
||||
"STLXRH",
|
||||
"STLXRW",
|
||||
"STP",
|
||||
"STPW",
|
||||
"STXP",
|
||||
"STXPW",
|
||||
"STXR",
|
||||
"STXRB",
|
||||
"STXRH",
|
||||
"STXRW",
|
||||
"SUB",
|
||||
"SUBS",
|
||||
"SUBSW",
|
||||
"SUBW",
|
||||
"SVC",
|
||||
"SWPAB",
|
||||
"SWPAD",
|
||||
"SWPAH",
|
||||
"SWPALB",
|
||||
"SWPALD",
|
||||
"SWPALH",
|
||||
"SWPALW",
|
||||
"SWPAW",
|
||||
"SWPB",
|
||||
"SWPD",
|
||||
"SWPH",
|
||||
"SWPLB",
|
||||
"SWPLD",
|
||||
"SWPLH",
|
||||
"SWPLW",
|
||||
"SWPW",
|
||||
"SXTB",
|
||||
"SXTBW",
|
||||
"SXTH",
|
||||
"SXTHW",
|
||||
"SXTW",
|
||||
"SYS",
|
||||
"SYSL",
|
||||
"TBNZ",
|
||||
"TBZ",
|
||||
"TLBI",
|
||||
"TST",
|
||||
"TSTW",
|
||||
"UBFIZ",
|
||||
"UBFIZW",
|
||||
"UBFM",
|
||||
"UBFMW",
|
||||
"UBFX",
|
||||
"UBFXW",
|
||||
"UCVTFD",
|
||||
"UCVTFS",
|
||||
"UCVTFWD",
|
||||
"UCVTFWS",
|
||||
"UDIV",
|
||||
"UDIVW",
|
||||
"UMADDL",
|
||||
"UMNEGL",
|
||||
"UMSUBL",
|
||||
"UMULH",
|
||||
"UMULL",
|
||||
"UREM",
|
||||
"UREMW",
|
||||
"UXTB",
|
||||
"UXTBW",
|
||||
"UXTH",
|
||||
"UXTHW",
|
||||
"UXTW",
|
||||
"VADD",
|
||||
"VADDP",
|
||||
"VADDV",
|
||||
"VAND",
|
||||
"VBCAX",
|
||||
"VBIF",
|
||||
"VBIT",
|
||||
"VBSL",
|
||||
"VCMEQ",
|
||||
"VCMTST",
|
||||
"VCNT",
|
||||
"VDUP",
|
||||
"VEOR",
|
||||
"VEOR3",
|
||||
"VEXT",
|
||||
"VFMLA",
|
||||
"VFMLS",
|
||||
"VLD1",
|
||||
"VLD1R",
|
||||
"VLD2",
|
||||
"VLD2R",
|
||||
"VLD3",
|
||||
"VLD3R",
|
||||
"VLD4",
|
||||
"VLD4R",
|
||||
"VMOV",
|
||||
"VMOVD",
|
||||
"VMOVI",
|
||||
"VMOVQ",
|
||||
"VMOVS",
|
||||
"VORR",
|
||||
"VPMULL",
|
||||
"VPMULL2",
|
||||
"VRAX1",
|
||||
"VRBIT",
|
||||
"VREV16",
|
||||
"VREV32",
|
||||
"VREV64",
|
||||
"VSHL",
|
||||
"VSLI",
|
||||
"VSRI",
|
||||
"VST1",
|
||||
"VST2",
|
||||
"VST3",
|
||||
"VST4",
|
||||
"VSUB",
|
||||
"VTBL",
|
||||
"VTBX",
|
||||
"VTRN1",
|
||||
"VTRN2",
|
||||
"VUADDLV",
|
||||
"VUADDW",
|
||||
"VUADDW2",
|
||||
"VUMAX",
|
||||
"VUMIN",
|
||||
"VUSHLL",
|
||||
"VUSHLL2",
|
||||
"VUSHR",
|
||||
"VUSRA",
|
||||
"VUXTL",
|
||||
"VUXTL2",
|
||||
"VUZP1",
|
||||
"VUZP2",
|
||||
"VXAR",
|
||||
"VZIP1",
|
||||
"VZIP2",
|
||||
"WFE",
|
||||
"WFI",
|
||||
"WORD",
|
||||
"YIELD",
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/util.go from the Go toolchain.
|
||||
|
||||
package arch
|
||||
|
||||
// commonGeneratedInstrs is the set of opcodes shared by every architecture
|
||||
// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).
|
||||
var commonGeneratedInstrs = []string{
|
||||
"CALL",
|
||||
"DUFFCOPY",
|
||||
"DUFFZERO",
|
||||
"END",
|
||||
"FUNCDATA",
|
||||
"GETCALLERPC",
|
||||
"JMP",
|
||||
"NOP",
|
||||
"PCALIGN",
|
||||
"PCALIGNMAX",
|
||||
"PCDATA",
|
||||
"RET",
|
||||
"TEXT",
|
||||
"UNDEF",
|
||||
}
|
||||
@@ -0,0 +1,81 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package arch
|
||||
|
||||
import "fmt"
|
||||
|
||||
func buildLOONG64() *Table {
|
||||
return newTable(LOONG64, loong64Registers(), relaxCounts(mergedInstrs(loong64Summaries(), commonGeneratedInstrs, loong64GeneratedInstrs, loong64Aliases())))
|
||||
}
|
||||
|
||||
// loong64Aliases are branch spellings the assembler front-end accepts in
|
||||
// addition to the generated opcode table (notably JAL, BFPF and BFPT).
|
||||
func loong64Aliases() []string {
|
||||
return []string{"JAL", "BFPF", "BFPT"}
|
||||
}
|
||||
|
||||
func loong64Summaries() map[string]Instr { return toMap(loong64Curated()) }
|
||||
|
||||
// loong64Registers builds the LoongArch (loong64) register file: 32 integer
|
||||
// (R), 32 floating-point (F), and the LSX/LASX SIMD vector registers (V and X).
|
||||
func loong64Registers() []Register {
|
||||
var regs []Register
|
||||
add := func(name string, class RegClass, desc string) {
|
||||
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
|
||||
}
|
||||
for i := 0; i <= 31; i++ {
|
||||
add(fmt.Sprintf("R%d", i), GPR, "integer register")
|
||||
}
|
||||
for i := 0; i <= 31; i++ {
|
||||
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
|
||||
}
|
||||
for i := 0; i <= 31; i++ {
|
||||
add(fmt.Sprintf("V%d", i), VecARM, "LSX 128-bit vector register")
|
||||
}
|
||||
for i := 0; i <= 31; i++ {
|
||||
add(fmt.Sprintf("X%d", i), VecARM, "LASX 256-bit vector register")
|
||||
}
|
||||
return regs
|
||||
}
|
||||
|
||||
// loong64Curated is a hand-written subset of common LoongArch instructions
|
||||
// carrying summaries. The authoritative, complete set is loong64GeneratedInstrs.
|
||||
func loong64Curated() []Instr {
|
||||
return []Instr{
|
||||
ic("ADD", "Integer add (word)", 3, 3), ic("ADDW", "Add word", 3, 3),
|
||||
ic("ADDD", "Add doubleword", 3, 3), ic("ADDI", "Add immediate", 3, 3),
|
||||
ic("SUB", "Subtract (word)", 3, 3), ic("SUBW", "Subtract word", 3, 3),
|
||||
ic("SUBD", "Subtract doubleword", 3, 3),
|
||||
ic("AND", "Bitwise AND", 3, 3), ic("ANDI", "AND immediate", 3, 3),
|
||||
ic("OR", "Bitwise OR", 3, 3), ic("ORI", "OR immediate", 3, 3),
|
||||
ic("XOR", "Bitwise XOR", 3, 3), ic("XORI", "XOR immediate", 3, 3),
|
||||
ic("NOR", "Bitwise NOR", 3, 3),
|
||||
ic("MUL", "Multiply (word)", 3, 3), ic("MULW", "Multiply word", 3, 3),
|
||||
ic("MULD", "Multiply doubleword", 3, 3),
|
||||
ic("DIV", "Divide (word)", 3, 3), ic("DIVW", "Divide word", 3, 3),
|
||||
ic("DIVD", "Divide doubleword", 3, 3),
|
||||
ic("MOD", "Modulo (word)", 3, 3), ic("MODW", "Modulo word", 3, 3),
|
||||
ic("MODD", "Modulo doubleword", 3, 3),
|
||||
ic("SLL", "Shift left logical", 3, 3), ic("SRL", "Shift right logical", 3, 3),
|
||||
ic("SRA", "Shift right arithmetic", 3, 3), ic("ROTR", "Rotate right", 3, 3),
|
||||
ic("SLT", "Set if less than", 3, 3), ic("SLTU", "Set if less than unsigned", 3, 3),
|
||||
ic("SLTI", "Set if less than immediate", 3, 3),
|
||||
ic("LD", "Load doubleword", 2, 2), ic("LDW", "Load word", 2, 2),
|
||||
ic("LDH", "Load halfword", 2, 2), ic("LDB", "Load byte", 2, 2),
|
||||
ic("ST", "Store doubleword", 2, 2), ic("STW", "Store word", 2, 2),
|
||||
ic("STH", "Store halfword", 2, 2), ic("STB", "Store byte", 2, 2),
|
||||
ic("BEQ", "Branch if equal", 3, 3), ic("BNE", "Branch if not equal", 3, 3),
|
||||
ic("BLT", "Branch if less than", 3, 3), ic("BGE", "Branch if greater or equal", 3, 3),
|
||||
ic("BLTU", "Branch if less than unsigned", 3, 3), ic("BGEU", "Branch if greater or equal unsigned", 3, 3),
|
||||
ic("B", "Unconditional branch", 1, 1), ic("BL", "Branch with link", 1, 1),
|
||||
ic("JIRL", "Jump indirect with link", 1, 3),
|
||||
ic("RET", "Return", 0, 1), ic("NOP", "No operation", 0, 1),
|
||||
i("SYSCALL", "System call"), i("BREAK", "Breakpoint"), i("DBAR", "Barrier"),
|
||||
ic("FADDS", "FP add (single)", 3, 3), ic("FADDD", "FP add (double)", 3, 3),
|
||||
ic("FSUBS", "FP subtract (single)", 3, 3), ic("FSUBD", "FP subtract (double)", 3, 3),
|
||||
ic("FMULS", "FP multiply (single)", 3, 3), ic("FMULD", "FP multiply (double)", 3, 3),
|
||||
ic("FDIVS", "FP divide (single)", 3, 3), ic("FDIVD", "FP divide (double)", 3, 3),
|
||||
ic("MOV", "Move register", 2, 2),
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,808 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
|
||||
|
||||
package arch
|
||||
|
||||
// loong64GeneratedInstrs is the complete set of loong64 mnemonics accepted by
|
||||
// Go's Plan 9 assembler.
|
||||
var loong64GeneratedInstrs = []string{
|
||||
"ABSD",
|
||||
"ABSF",
|
||||
"ADD",
|
||||
"ADDD",
|
||||
"ADDF",
|
||||
"ADDV",
|
||||
"ADDV16",
|
||||
"ADDVU",
|
||||
"ADDW",
|
||||
"ALSLV",
|
||||
"ALSLW",
|
||||
"ALSLWU",
|
||||
"AMADDDBV",
|
||||
"AMADDDBW",
|
||||
"AMADDV",
|
||||
"AMADDW",
|
||||
"AMANDDBV",
|
||||
"AMANDDBW",
|
||||
"AMANDV",
|
||||
"AMANDW",
|
||||
"AMCASB",
|
||||
"AMCASDBB",
|
||||
"AMCASDBH",
|
||||
"AMCASDBV",
|
||||
"AMCASDBW",
|
||||
"AMCASH",
|
||||
"AMCASV",
|
||||
"AMCASW",
|
||||
"AMMAXDBV",
|
||||
"AMMAXDBVU",
|
||||
"AMMAXDBW",
|
||||
"AMMAXDBWU",
|
||||
"AMMAXV",
|
||||
"AMMAXVU",
|
||||
"AMMAXW",
|
||||
"AMMAXWU",
|
||||
"AMMINDBV",
|
||||
"AMMINDBVU",
|
||||
"AMMINDBW",
|
||||
"AMMINDBWU",
|
||||
"AMMINV",
|
||||
"AMMINVU",
|
||||
"AMMINW",
|
||||
"AMMINWU",
|
||||
"AMORDBV",
|
||||
"AMORDBW",
|
||||
"AMORV",
|
||||
"AMORW",
|
||||
"AMSWAPB",
|
||||
"AMSWAPDBB",
|
||||
"AMSWAPDBH",
|
||||
"AMSWAPDBV",
|
||||
"AMSWAPDBW",
|
||||
"AMSWAPH",
|
||||
"AMSWAPV",
|
||||
"AMSWAPW",
|
||||
"AMXORDBV",
|
||||
"AMXORDBW",
|
||||
"AMXORV",
|
||||
"AMXORW",
|
||||
"AND",
|
||||
"ANDN",
|
||||
"BEQ",
|
||||
"BFPF",
|
||||
"BFPT",
|
||||
"BGE",
|
||||
"BGEU",
|
||||
"BGEZ",
|
||||
"BGTZ",
|
||||
"BITREV4B",
|
||||
"BITREV8B",
|
||||
"BITREVV",
|
||||
"BITREVW",
|
||||
"BLEZ",
|
||||
"BLT",
|
||||
"BLTU",
|
||||
"BLTZ",
|
||||
"BNE",
|
||||
"BREAK",
|
||||
"BSTRINSV",
|
||||
"BSTRINSW",
|
||||
"BSTRPICKV",
|
||||
"BSTRPICKW",
|
||||
"CLOV",
|
||||
"CLOW",
|
||||
"CLZV",
|
||||
"CLZW",
|
||||
"CMPEQD",
|
||||
"CMPEQF",
|
||||
"CMPGED",
|
||||
"CMPGEF",
|
||||
"CMPGTD",
|
||||
"CMPGTF",
|
||||
"CPUCFG",
|
||||
"CRCCWBW",
|
||||
"CRCCWHW",
|
||||
"CRCCWVW",
|
||||
"CRCCWWW",
|
||||
"CRCWBW",
|
||||
"CRCWHW",
|
||||
"CRCWVW",
|
||||
"CRCWWW",
|
||||
"CTOV",
|
||||
"CTOW",
|
||||
"CTZV",
|
||||
"CTZW",
|
||||
"DBAR",
|
||||
"DIV",
|
||||
"DIVD",
|
||||
"DIVF",
|
||||
"DIVU",
|
||||
"DIVV",
|
||||
"DIVVU",
|
||||
"DIVW",
|
||||
"DIVWU",
|
||||
"EXTWB",
|
||||
"EXTWH",
|
||||
"FCLASSD",
|
||||
"FCLASSF",
|
||||
"FCOPYSGD",
|
||||
"FCOPYSGF",
|
||||
"FFINTDV",
|
||||
"FFINTDW",
|
||||
"FFINTFV",
|
||||
"FFINTFW",
|
||||
"FLOGBD",
|
||||
"FLOGBF",
|
||||
"FMADDD",
|
||||
"FMADDF",
|
||||
"FMAXAD",
|
||||
"FMAXAF",
|
||||
"FMAXD",
|
||||
"FMAXF",
|
||||
"FMINAD",
|
||||
"FMINAF",
|
||||
"FMIND",
|
||||
"FMINF",
|
||||
"FMSUBD",
|
||||
"FMSUBF",
|
||||
"FNMADDD",
|
||||
"FNMADDF",
|
||||
"FNMSUBD",
|
||||
"FNMSUBF",
|
||||
"FSCALEBD",
|
||||
"FSCALEBF",
|
||||
"FSEL",
|
||||
"FTINTRMVD",
|
||||
"FTINTRMVF",
|
||||
"FTINTRMWD",
|
||||
"FTINTRMWF",
|
||||
"FTINTRNEVD",
|
||||
"FTINTRNEVF",
|
||||
"FTINTRNEWD",
|
||||
"FTINTRNEWF",
|
||||
"FTINTRPVD",
|
||||
"FTINTRPVF",
|
||||
"FTINTRPWD",
|
||||
"FTINTRPWF",
|
||||
"FTINTRZVD",
|
||||
"FTINTRZVF",
|
||||
"FTINTRZWD",
|
||||
"FTINTRZWF",
|
||||
"FTINTVD",
|
||||
"FTINTVF",
|
||||
"FTINTWD",
|
||||
"FTINTWF",
|
||||
"JIRL",
|
||||
"LL",
|
||||
"LLV",
|
||||
"LU12IW",
|
||||
"LU32ID",
|
||||
"LU52ID",
|
||||
"LUI",
|
||||
"MASKEQZ",
|
||||
"MASKNEZ",
|
||||
"MOVB",
|
||||
"MOVBU",
|
||||
"MOVD",
|
||||
"MOVDF",
|
||||
"MOVDV",
|
||||
"MOVDW",
|
||||
"MOVF",
|
||||
"MOVFD",
|
||||
"MOVFV",
|
||||
"MOVFW",
|
||||
"MOVH",
|
||||
"MOVHU",
|
||||
"MOVV",
|
||||
"MOVVD",
|
||||
"MOVVF",
|
||||
"MOVVP",
|
||||
"MOVW",
|
||||
"MOVWD",
|
||||
"MOVWF",
|
||||
"MOVWP",
|
||||
"MOVWU",
|
||||
"MUL",
|
||||
"MULD",
|
||||
"MULF",
|
||||
"MULH",
|
||||
"MULHU",
|
||||
"MULHV",
|
||||
"MULHVU",
|
||||
"MULV",
|
||||
"MULVU",
|
||||
"MULW",
|
||||
"MULWVW",
|
||||
"MULWVWU",
|
||||
"NEGD",
|
||||
"NEGF",
|
||||
"NEGV",
|
||||
"NEGW",
|
||||
"NOOP",
|
||||
"NOR",
|
||||
"OR",
|
||||
"ORN",
|
||||
"PCADDU12I",
|
||||
"PCALAU12I",
|
||||
"PRELD",
|
||||
"PRELDX",
|
||||
"RDTIMED",
|
||||
"RDTIMEHW",
|
||||
"RDTIMELW",
|
||||
"REM",
|
||||
"REMU",
|
||||
"REMV",
|
||||
"REMVU",
|
||||
"REMW",
|
||||
"REMWU",
|
||||
"REVB2H",
|
||||
"REVB2W",
|
||||
"REVB4H",
|
||||
"REVBV",
|
||||
"REVH2W",
|
||||
"REVHV",
|
||||
"RFE",
|
||||
"ROTR",
|
||||
"ROTRV",
|
||||
"SC",
|
||||
"SCV",
|
||||
"SGT",
|
||||
"SGTU",
|
||||
"SLL",
|
||||
"SLLV",
|
||||
"SQRTD",
|
||||
"SQRTF",
|
||||
"SRA",
|
||||
"SRAV",
|
||||
"SRL",
|
||||
"SRLV",
|
||||
"SUB",
|
||||
"SUBD",
|
||||
"SUBF",
|
||||
"SUBV",
|
||||
"SUBVU",
|
||||
"SUBW",
|
||||
"SYSCALL",
|
||||
"TEQ",
|
||||
"TNE",
|
||||
"TRUNCDV",
|
||||
"TRUNCDW",
|
||||
"TRUNCFV",
|
||||
"TRUNCFW",
|
||||
"VADDB",
|
||||
"VADDBU",
|
||||
"VADDD",
|
||||
"VADDF",
|
||||
"VADDH",
|
||||
"VADDHU",
|
||||
"VADDQ",
|
||||
"VADDV",
|
||||
"VADDVU",
|
||||
"VADDW",
|
||||
"VADDWEVHB",
|
||||
"VADDWEVHBU",
|
||||
"VADDWEVQV",
|
||||
"VADDWEVQVU",
|
||||
"VADDWEVVW",
|
||||
"VADDWEVVWU",
|
||||
"VADDWEVWH",
|
||||
"VADDWEVWHU",
|
||||
"VADDWODHB",
|
||||
"VADDWODHBU",
|
||||
"VADDWODQV",
|
||||
"VADDWODQVU",
|
||||
"VADDWODVW",
|
||||
"VADDWODVWU",
|
||||
"VADDWODWH",
|
||||
"VADDWODWHU",
|
||||
"VADDWU",
|
||||
"VANDB",
|
||||
"VANDNV",
|
||||
"VANDV",
|
||||
"VBITCLRB",
|
||||
"VBITCLRH",
|
||||
"VBITCLRV",
|
||||
"VBITCLRW",
|
||||
"VBITREVB",
|
||||
"VBITREVH",
|
||||
"VBITREVV",
|
||||
"VBITREVW",
|
||||
"VBITSETB",
|
||||
"VBITSETH",
|
||||
"VBITSETV",
|
||||
"VBITSETW",
|
||||
"VDIVB",
|
||||
"VDIVBU",
|
||||
"VDIVD",
|
||||
"VDIVF",
|
||||
"VDIVH",
|
||||
"VDIVHU",
|
||||
"VDIVV",
|
||||
"VDIVVU",
|
||||
"VDIVW",
|
||||
"VDIVWU",
|
||||
"VEXTRINSB",
|
||||
"VEXTRINSH",
|
||||
"VEXTRINSV",
|
||||
"VEXTRINSW",
|
||||
"VFCLASSD",
|
||||
"VFCLASSF",
|
||||
"VFRECIPD",
|
||||
"VFRECIPF",
|
||||
"VFRINTD",
|
||||
"VFRINTF",
|
||||
"VFRINTRMD",
|
||||
"VFRINTRMF",
|
||||
"VFRINTRNED",
|
||||
"VFRINTRNEF",
|
||||
"VFRINTRPD",
|
||||
"VFRINTRPF",
|
||||
"VFRINTRZD",
|
||||
"VFRINTRZF",
|
||||
"VFRSQRTD",
|
||||
"VFRSQRTF",
|
||||
"VFSQRTD",
|
||||
"VFSQRTF",
|
||||
"VILVHB",
|
||||
"VILVHH",
|
||||
"VILVHV",
|
||||
"VILVHW",
|
||||
"VILVLB",
|
||||
"VILVLH",
|
||||
"VILVLV",
|
||||
"VILVLW",
|
||||
"VMADDB",
|
||||
"VMADDH",
|
||||
"VMADDV",
|
||||
"VMADDW",
|
||||
"VMADDWEVHB",
|
||||
"VMADDWEVHBU",
|
||||
"VMADDWEVHBUB",
|
||||
"VMADDWEVQV",
|
||||
"VMADDWEVQVU",
|
||||
"VMADDWEVQVUV",
|
||||
"VMADDWEVVW",
|
||||
"VMADDWEVVWU",
|
||||
"VMADDWEVVWUW",
|
||||
"VMADDWEVWH",
|
||||
"VMADDWEVWHU",
|
||||
"VMADDWEVWHUH",
|
||||
"VMADDWODHB",
|
||||
"VMADDWODHBU",
|
||||
"VMADDWODHBUB",
|
||||
"VMADDWODQV",
|
||||
"VMADDWODQVU",
|
||||
"VMADDWODQVUV",
|
||||
"VMADDWODVW",
|
||||
"VMADDWODVWU",
|
||||
"VMADDWODVWUW",
|
||||
"VMADDWODWH",
|
||||
"VMADDWODWHU",
|
||||
"VMADDWODWHUH",
|
||||
"VMODB",
|
||||
"VMODBU",
|
||||
"VMODH",
|
||||
"VMODHU",
|
||||
"VMODV",
|
||||
"VMODVU",
|
||||
"VMODW",
|
||||
"VMODWU",
|
||||
"VMOVQ",
|
||||
"VMSUBB",
|
||||
"VMSUBH",
|
||||
"VMSUBV",
|
||||
"VMSUBW",
|
||||
"VMUHB",
|
||||
"VMUHBU",
|
||||
"VMUHH",
|
||||
"VMUHHU",
|
||||
"VMUHV",
|
||||
"VMUHVU",
|
||||
"VMUHW",
|
||||
"VMUHWU",
|
||||
"VMULB",
|
||||
"VMULD",
|
||||
"VMULF",
|
||||
"VMULH",
|
||||
"VMULV",
|
||||
"VMULW",
|
||||
"VMULWEVHB",
|
||||
"VMULWEVHBU",
|
||||
"VMULWEVHBUB",
|
||||
"VMULWEVQV",
|
||||
"VMULWEVQVU",
|
||||
"VMULWEVQVUV",
|
||||
"VMULWEVVW",
|
||||
"VMULWEVVWU",
|
||||
"VMULWEVVWUW",
|
||||
"VMULWEVWH",
|
||||
"VMULWEVWHU",
|
||||
"VMULWEVWHUH",
|
||||
"VMULWODHB",
|
||||
"VMULWODHBU",
|
||||
"VMULWODHBUB",
|
||||
"VMULWODQV",
|
||||
"VMULWODQVU",
|
||||
"VMULWODQVUV",
|
||||
"VMULWODVW",
|
||||
"VMULWODVWU",
|
||||
"VMULWODVWUW",
|
||||
"VMULWODWH",
|
||||
"VMULWODWHU",
|
||||
"VMULWODWHUH",
|
||||
"VNEGB",
|
||||
"VNEGH",
|
||||
"VNEGV",
|
||||
"VNEGW",
|
||||
"VNORB",
|
||||
"VNORV",
|
||||
"VORB",
|
||||
"VORNV",
|
||||
"VORV",
|
||||
"VPCNTB",
|
||||
"VPCNTH",
|
||||
"VPCNTV",
|
||||
"VPCNTW",
|
||||
"VPERMIW",
|
||||
"VROTRB",
|
||||
"VROTRH",
|
||||
"VROTRV",
|
||||
"VROTRW",
|
||||
"VSADDB",
|
||||
"VSADDBU",
|
||||
"VSADDH",
|
||||
"VSADDHU",
|
||||
"VSADDV",
|
||||
"VSADDVU",
|
||||
"VSADDW",
|
||||
"VSADDWU",
|
||||
"VSEQB",
|
||||
"VSEQH",
|
||||
"VSEQV",
|
||||
"VSEQW",
|
||||
"VSETALLNEB",
|
||||
"VSETALLNEH",
|
||||
"VSETALLNEV",
|
||||
"VSETALLNEW",
|
||||
"VSETANYEQB",
|
||||
"VSETANYEQH",
|
||||
"VSETANYEQV",
|
||||
"VSETANYEQW",
|
||||
"VSETEQV",
|
||||
"VSETNEV",
|
||||
"VSHUF4IB",
|
||||
"VSHUF4IH",
|
||||
"VSHUF4IV",
|
||||
"VSHUF4IW",
|
||||
"VSHUFB",
|
||||
"VSHUFH",
|
||||
"VSHUFV",
|
||||
"VSHUFW",
|
||||
"VSLLB",
|
||||
"VSLLH",
|
||||
"VSLLV",
|
||||
"VSLLW",
|
||||
"VSLTB",
|
||||
"VSLTBU",
|
||||
"VSLTH",
|
||||
"VSLTHU",
|
||||
"VSLTV",
|
||||
"VSLTVU",
|
||||
"VSLTW",
|
||||
"VSLTWU",
|
||||
"VSRAB",
|
||||
"VSRAH",
|
||||
"VSRAV",
|
||||
"VSRAW",
|
||||
"VSRLB",
|
||||
"VSRLH",
|
||||
"VSRLV",
|
||||
"VSRLW",
|
||||
"VSSUBB",
|
||||
"VSSUBBU",
|
||||
"VSSUBH",
|
||||
"VSSUBHU",
|
||||
"VSSUBV",
|
||||
"VSSUBVU",
|
||||
"VSSUBW",
|
||||
"VSSUBWU",
|
||||
"VSUBB",
|
||||
"VSUBBU",
|
||||
"VSUBD",
|
||||
"VSUBF",
|
||||
"VSUBH",
|
||||
"VSUBHU",
|
||||
"VSUBQ",
|
||||
"VSUBV",
|
||||
"VSUBVU",
|
||||
"VSUBW",
|
||||
"VSUBWEVHB",
|
||||
"VSUBWEVHBU",
|
||||
"VSUBWEVQV",
|
||||
"VSUBWEVQVU",
|
||||
"VSUBWEVVW",
|
||||
"VSUBWEVVWU",
|
||||
"VSUBWEVWH",
|
||||
"VSUBWEVWHU",
|
||||
"VSUBWODHB",
|
||||
"VSUBWODHBU",
|
||||
"VSUBWODQV",
|
||||
"VSUBWODQVU",
|
||||
"VSUBWODVW",
|
||||
"VSUBWODVWU",
|
||||
"VSUBWODWH",
|
||||
"VSUBWODWHU",
|
||||
"VSUBWU",
|
||||
"VXORB",
|
||||
"VXORV",
|
||||
"WORD",
|
||||
"XOR",
|
||||
"XVADDB",
|
||||
"XVADDBU",
|
||||
"XVADDD",
|
||||
"XVADDF",
|
||||
"XVADDH",
|
||||
"XVADDHU",
|
||||
"XVADDQ",
|
||||
"XVADDV",
|
||||
"XVADDVU",
|
||||
"XVADDW",
|
||||
"XVADDWEVHB",
|
||||
"XVADDWEVHBU",
|
||||
"XVADDWEVQV",
|
||||
"XVADDWEVQVU",
|
||||
"XVADDWEVVW",
|
||||
"XVADDWEVVWU",
|
||||
"XVADDWEVWH",
|
||||
"XVADDWEVWHU",
|
||||
"XVADDWODHB",
|
||||
"XVADDWODHBU",
|
||||
"XVADDWODQV",
|
||||
"XVADDWODQVU",
|
||||
"XVADDWODVW",
|
||||
"XVADDWODVWU",
|
||||
"XVADDWODWH",
|
||||
"XVADDWODWHU",
|
||||
"XVADDWU",
|
||||
"XVANDB",
|
||||
"XVANDNV",
|
||||
"XVANDV",
|
||||
"XVBITCLRB",
|
||||
"XVBITCLRH",
|
||||
"XVBITCLRV",
|
||||
"XVBITCLRW",
|
||||
"XVBITREVB",
|
||||
"XVBITREVH",
|
||||
"XVBITREVV",
|
||||
"XVBITREVW",
|
||||
"XVBITSETB",
|
||||
"XVBITSETH",
|
||||
"XVBITSETV",
|
||||
"XVBITSETW",
|
||||
"XVDIVB",
|
||||
"XVDIVBU",
|
||||
"XVDIVD",
|
||||
"XVDIVF",
|
||||
"XVDIVH",
|
||||
"XVDIVHU",
|
||||
"XVDIVV",
|
||||
"XVDIVVU",
|
||||
"XVDIVW",
|
||||
"XVDIVWU",
|
||||
"XVEXTRINSB",
|
||||
"XVEXTRINSH",
|
||||
"XVEXTRINSV",
|
||||
"XVEXTRINSW",
|
||||
"XVFCLASSD",
|
||||
"XVFCLASSF",
|
||||
"XVFRECIPD",
|
||||
"XVFRECIPF",
|
||||
"XVFRINTD",
|
||||
"XVFRINTF",
|
||||
"XVFRINTRMD",
|
||||
"XVFRINTRMF",
|
||||
"XVFRINTRNED",
|
||||
"XVFRINTRNEF",
|
||||
"XVFRINTRPD",
|
||||
"XVFRINTRPF",
|
||||
"XVFRINTRZD",
|
||||
"XVFRINTRZF",
|
||||
"XVFRSQRTD",
|
||||
"XVFRSQRTF",
|
||||
"XVFSQRTD",
|
||||
"XVFSQRTF",
|
||||
"XVILVHB",
|
||||
"XVILVHH",
|
||||
"XVILVHV",
|
||||
"XVILVHW",
|
||||
"XVILVLB",
|
||||
"XVILVLH",
|
||||
"XVILVLV",
|
||||
"XVILVLW",
|
||||
"XVMADDB",
|
||||
"XVMADDH",
|
||||
"XVMADDV",
|
||||
"XVMADDW",
|
||||
"XVMADDWEVHB",
|
||||
"XVMADDWEVHBU",
|
||||
"XVMADDWEVHBUB",
|
||||
"XVMADDWEVQV",
|
||||
"XVMADDWEVQVU",
|
||||
"XVMADDWEVQVUV",
|
||||
"XVMADDWEVVW",
|
||||
"XVMADDWEVVWU",
|
||||
"XVMADDWEVVWUW",
|
||||
"XVMADDWEVWH",
|
||||
"XVMADDWEVWHU",
|
||||
"XVMADDWEVWHUH",
|
||||
"XVMADDWODHB",
|
||||
"XVMADDWODHBU",
|
||||
"XVMADDWODHBUB",
|
||||
"XVMADDWODQV",
|
||||
"XVMADDWODQVU",
|
||||
"XVMADDWODQVUV",
|
||||
"XVMADDWODVW",
|
||||
"XVMADDWODVWU",
|
||||
"XVMADDWODVWUW",
|
||||
"XVMADDWODWH",
|
||||
"XVMADDWODWHU",
|
||||
"XVMADDWODWHUH",
|
||||
"XVMODB",
|
||||
"XVMODBU",
|
||||
"XVMODH",
|
||||
"XVMODHU",
|
||||
"XVMODV",
|
||||
"XVMODVU",
|
||||
"XVMODW",
|
||||
"XVMODWU",
|
||||
"XVMOVQ",
|
||||
"XVMSUBB",
|
||||
"XVMSUBH",
|
||||
"XVMSUBV",
|
||||
"XVMSUBW",
|
||||
"XVMUHB",
|
||||
"XVMUHBU",
|
||||
"XVMUHH",
|
||||
"XVMUHHU",
|
||||
"XVMUHV",
|
||||
"XVMUHVU",
|
||||
"XVMUHW",
|
||||
"XVMUHWU",
|
||||
"XVMULB",
|
||||
"XVMULD",
|
||||
"XVMULF",
|
||||
"XVMULH",
|
||||
"XVMULV",
|
||||
"XVMULW",
|
||||
"XVMULWEVHB",
|
||||
"XVMULWEVHBU",
|
||||
"XVMULWEVHBUB",
|
||||
"XVMULWEVQV",
|
||||
"XVMULWEVQVU",
|
||||
"XVMULWEVQVUV",
|
||||
"XVMULWEVVW",
|
||||
"XVMULWEVVWU",
|
||||
"XVMULWEVVWUW",
|
||||
"XVMULWEVWH",
|
||||
"XVMULWEVWHU",
|
||||
"XVMULWEVWHUH",
|
||||
"XVMULWODHB",
|
||||
"XVMULWODHBU",
|
||||
"XVMULWODHBUB",
|
||||
"XVMULWODQV",
|
||||
"XVMULWODQVU",
|
||||
"XVMULWODQVUV",
|
||||
"XVMULWODVW",
|
||||
"XVMULWODVWU",
|
||||
"XVMULWODVWUW",
|
||||
"XVMULWODWH",
|
||||
"XVMULWODWHU",
|
||||
"XVMULWODWHUH",
|
||||
"XVNEGB",
|
||||
"XVNEGH",
|
||||
"XVNEGV",
|
||||
"XVNEGW",
|
||||
"XVNORB",
|
||||
"XVNORV",
|
||||
"XVORB",
|
||||
"XVORNV",
|
||||
"XVORV",
|
||||
"XVPCNTB",
|
||||
"XVPCNTH",
|
||||
"XVPCNTV",
|
||||
"XVPCNTW",
|
||||
"XVPERMIQ",
|
||||
"XVPERMIV",
|
||||
"XVPERMIW",
|
||||
"XVROTRB",
|
||||
"XVROTRH",
|
||||
"XVROTRV",
|
||||
"XVROTRW",
|
||||
"XVSADDB",
|
||||
"XVSADDBU",
|
||||
"XVSADDH",
|
||||
"XVSADDHU",
|
||||
"XVSADDV",
|
||||
"XVSADDVU",
|
||||
"XVSADDW",
|
||||
"XVSADDWU",
|
||||
"XVSEQB",
|
||||
"XVSEQH",
|
||||
"XVSEQV",
|
||||
"XVSEQW",
|
||||
"XVSETALLNEB",
|
||||
"XVSETALLNEH",
|
||||
"XVSETALLNEV",
|
||||
"XVSETALLNEW",
|
||||
"XVSETANYEQB",
|
||||
"XVSETANYEQH",
|
||||
"XVSETANYEQV",
|
||||
"XVSETANYEQW",
|
||||
"XVSETEQV",
|
||||
"XVSETNEV",
|
||||
"XVSHUF4IB",
|
||||
"XVSHUF4IH",
|
||||
"XVSHUF4IV",
|
||||
"XVSHUF4IW",
|
||||
"XVSHUFB",
|
||||
"XVSHUFH",
|
||||
"XVSHUFV",
|
||||
"XVSHUFW",
|
||||
"XVSLLB",
|
||||
"XVSLLH",
|
||||
"XVSLLV",
|
||||
"XVSLLW",
|
||||
"XVSLTB",
|
||||
"XVSLTBU",
|
||||
"XVSLTH",
|
||||
"XVSLTHU",
|
||||
"XVSLTV",
|
||||
"XVSLTVU",
|
||||
"XVSLTW",
|
||||
"XVSLTWU",
|
||||
"XVSRAB",
|
||||
"XVSRAH",
|
||||
"XVSRAV",
|
||||
"XVSRAW",
|
||||
"XVSRLB",
|
||||
"XVSRLH",
|
||||
"XVSRLV",
|
||||
"XVSRLW",
|
||||
"XVSSUBB",
|
||||
"XVSSUBBU",
|
||||
"XVSSUBH",
|
||||
"XVSSUBHU",
|
||||
"XVSSUBV",
|
||||
"XVSSUBVU",
|
||||
"XVSSUBW",
|
||||
"XVSSUBWU",
|
||||
"XVSUBB",
|
||||
"XVSUBBU",
|
||||
"XVSUBD",
|
||||
"XVSUBF",
|
||||
"XVSUBH",
|
||||
"XVSUBHU",
|
||||
"XVSUBQ",
|
||||
"XVSUBV",
|
||||
"XVSUBVU",
|
||||
"XVSUBW",
|
||||
"XVSUBWEVHB",
|
||||
"XVSUBWEVHBU",
|
||||
"XVSUBWEVQV",
|
||||
"XVSUBWEVQVU",
|
||||
"XVSUBWEVVW",
|
||||
"XVSUBWEVVWU",
|
||||
"XVSUBWEVWH",
|
||||
"XVSUBWEVWHU",
|
||||
"XVSUBWODHB",
|
||||
"XVSUBWODHBU",
|
||||
"XVSUBWODQV",
|
||||
"XVSUBWODQVU",
|
||||
"XVSUBWODVW",
|
||||
"XVSUBWODVWU",
|
||||
"XVSUBWODWH",
|
||||
"XVSUBWODWHU",
|
||||
"XVSUBWU",
|
||||
"XVXORB",
|
||||
"XVXORV",
|
||||
}
|
||||
+100
@@ -0,0 +1,100 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package arch
|
||||
|
||||
import "fmt"
|
||||
|
||||
func buildRISCV() *Table {
|
||||
return newTable(RISCV, riscvRegisters(), relaxCounts(mergedInstrs(riscvSummaries(), commonGeneratedInstrs, riscvGeneratedInstrs)))
|
||||
}
|
||||
|
||||
func riscvSummaries() map[string]Instr { return toMap(riscvCurated()) }
|
||||
|
||||
// riscvRegisters builds the RISC-V register file: the numbered integer (X) and
|
||||
// floating-point (F) registers plus their standard ABI aliases.
|
||||
func riscvRegisters() []Register {
|
||||
var regs []Register
|
||||
add := func(name string, class RegClass, desc string) {
|
||||
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
|
||||
}
|
||||
for i := 0; i <= 31; i++ {
|
||||
add(fmt.Sprintf("X%d", i), GPR, "integer register")
|
||||
}
|
||||
for i := 0; i <= 31; i++ {
|
||||
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
|
||||
}
|
||||
// Integer ABI aliases.
|
||||
for _, n := range []string{"ZERO", "RA", "SP", "GP", "TP", "FP", "LR", "TMP"} {
|
||||
add(n, GPR, "integer ABI alias")
|
||||
}
|
||||
for i := 0; i <= 6; i++ {
|
||||
add(fmt.Sprintf("T%d", i), GPR, "temporary")
|
||||
}
|
||||
for i := 0; i <= 11; i++ {
|
||||
add(fmt.Sprintf("S%d", i), GPR, "saved register")
|
||||
}
|
||||
for i := 0; i <= 7; i++ {
|
||||
add(fmt.Sprintf("A%d", i), GPR, "argument/result register")
|
||||
}
|
||||
// Floating-point ABI aliases.
|
||||
for i := 0; i <= 11; i++ {
|
||||
add(fmt.Sprintf("FT%d", i), Float, "FP temporary")
|
||||
}
|
||||
for i := 0; i <= 11; i++ {
|
||||
add(fmt.Sprintf("FS%d", i), Float, "FP saved register")
|
||||
}
|
||||
for i := 0; i <= 7; i++ {
|
||||
add(fmt.Sprintf("FA%d", i), Float, "FP argument/result register")
|
||||
}
|
||||
return regs
|
||||
}
|
||||
|
||||
// riscvCurated is a hand-written subset of common RISC-V instructions carrying
|
||||
// summaries. The authoritative, complete set is riscvGeneratedInstrs.
|
||||
func riscvCurated() []Instr {
|
||||
return []Instr{
|
||||
ic("ADD", "Integer add", 3, 3), ic("ADDI", "Add immediate", 3, 3),
|
||||
ic("ADDIW", "Add immediate (32-bit)", 3, 3), ic("ADDW", "Add (32-bit)", 3, 3),
|
||||
ic("SUB", "Integer subtract", 3, 3), ic("SUBW", "Subtract (32-bit)", 3, 3),
|
||||
ic("AND", "Bitwise AND", 3, 3), ic("ANDI", "AND immediate", 3, 3),
|
||||
ic("OR", "Bitwise OR", 3, 3), ic("ORI", "OR immediate", 3, 3),
|
||||
ic("XOR", "Bitwise XOR", 3, 3), ic("XORI", "XOR immediate", 3, 3),
|
||||
ic("SLL", "Shift left logical", 3, 3), ic("SLLI", "Shift left logical immediate", 3, 3),
|
||||
ic("SRL", "Shift right logical", 3, 3), ic("SRLI", "Shift right logical immediate", 3, 3),
|
||||
ic("SRA", "Shift right arithmetic", 3, 3), ic("SRAI", "Shift right arithmetic immediate", 3, 3),
|
||||
ic("SLT", "Set if less than", 3, 3), ic("SLTI", "Set if less than immediate", 3, 3),
|
||||
ic("SLTU", "Set if less than unsigned", 3, 3), ic("SLTIU", "Set if less than unsigned immediate", 3, 3),
|
||||
ic("MUL", "Multiply", 3, 3), ic("MULH", "Multiply high", 3, 3),
|
||||
ic("MULHU", "Multiply high unsigned", 3, 3), ic("MULHSU", "Multiply high signed/unsigned", 3, 3),
|
||||
ic("DIV", "Divide", 3, 3), ic("DIVU", "Divide unsigned", 3, 3),
|
||||
ic("REM", "Remainder", 3, 3), ic("REMU", "Remainder unsigned", 3, 3),
|
||||
ic("MULW", "Multiply (32-bit)", 3, 3), ic("DIVW", "Divide (32-bit)", 3, 3),
|
||||
ic("LB", "Load byte", 2, 2), ic("LBU", "Load byte unsigned", 2, 2),
|
||||
ic("LH", "Load halfword", 2, 2), ic("LHU", "Load halfword unsigned", 2, 2),
|
||||
ic("LW", "Load word", 2, 2), ic("LWU", "Load word unsigned", 2, 2),
|
||||
ic("LD", "Load doubleword", 2, 2),
|
||||
ic("SB", "Store byte", 2, 2), ic("SH", "Store halfword", 2, 2),
|
||||
ic("SW", "Store word", 2, 2), ic("SD", "Store doubleword", 2, 2),
|
||||
ic("LUI", "Load upper immediate", 2, 2), ic("AUIPC", "Add upper immediate to PC", 2, 2),
|
||||
ic("BEQ", "Branch if equal", 3, 3), ic("BNE", "Branch if not equal", 3, 3),
|
||||
ic("BLT", "Branch if less than", 3, 3), ic("BGE", "Branch if greater or equal", 3, 3),
|
||||
ic("BLTU", "Branch if less than unsigned", 3, 3), ic("BGEU", "Branch if greater or equal unsigned", 3, 3),
|
||||
ic("JAL", "Jump and link", 1, 2), ic("JALR", "Jump and link register", 1, 3),
|
||||
i("JMP", "Unconditional jump"), i("CALL", "Call subroutine"),
|
||||
ic("RET", "Return", 0, 1), i("ECALL", "Environment call"), i("EBREAK", "Breakpoint"),
|
||||
i("FENCE", "Memory barrier"), i("CSR", "Control/status register access"),
|
||||
// Floating point.
|
||||
ic("FADDS", "FP add (single)", 3, 3), ic("FADDD", "FP add (double)", 3, 3),
|
||||
ic("FSUBS", "FP subtract (single)", 3, 3), ic("FSUBD", "FP subtract (double)", 3, 3),
|
||||
ic("FMULS", "FP multiply (single)", 3, 3), ic("FMULD", "FP multiply (double)", 3, 3),
|
||||
ic("FDIVS", "FP divide (single)", 3, 3), ic("FDIVD", "FP divide (double)", 3, 3),
|
||||
ic("FLW", "FP load word", 2, 2), ic("FLD", "FP load doubleword", 2, 2),
|
||||
ic("FSW", "FP store word", 2, 2), ic("FSD", "FP store doubleword", 2, 2),
|
||||
// Atomics.
|
||||
i("LRW", "Load-reserved word"), i("LRD", "Load-reserved doubleword"),
|
||||
i("SCW", "Store-conditional word"), i("SCD", "Store-conditional doubleword"),
|
||||
i("AMOSWAPW", "Atomic swap word"), i("AMOSWAPD", "Atomic swap doubleword"),
|
||||
i("AMOADDW", "Atomic add word"), i("AMOADDD", "Atomic add doubleword"),
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,970 @@
|
||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
||||
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
|
||||
|
||||
package arch
|
||||
|
||||
// riscvGeneratedInstrs is the complete set of riscv mnemonics accepted by
|
||||
// Go's Plan 9 assembler.
|
||||
var riscvGeneratedInstrs = []string{
|
||||
"ADD",
|
||||
"ADDI",
|
||||
"ADDIW",
|
||||
"ADDUW",
|
||||
"ADDW",
|
||||
"AMOADDD",
|
||||
"AMOADDW",
|
||||
"AMOANDD",
|
||||
"AMOANDW",
|
||||
"AMOMAXD",
|
||||
"AMOMAXUD",
|
||||
"AMOMAXUW",
|
||||
"AMOMAXW",
|
||||
"AMOMIND",
|
||||
"AMOMINUD",
|
||||
"AMOMINUW",
|
||||
"AMOMINW",
|
||||
"AMOORD",
|
||||
"AMOORW",
|
||||
"AMOSWAPD",
|
||||
"AMOSWAPW",
|
||||
"AMOXORD",
|
||||
"AMOXORW",
|
||||
"AND",
|
||||
"ANDI",
|
||||
"ANDN",
|
||||
"AUIPC",
|
||||
"BCLR",
|
||||
"BCLRI",
|
||||
"BEQ",
|
||||
"BEQZ",
|
||||
"BEXT",
|
||||
"BEXTI",
|
||||
"BGE",
|
||||
"BGEU",
|
||||
"BGEZ",
|
||||
"BGT",
|
||||
"BGTU",
|
||||
"BGTZ",
|
||||
"BINV",
|
||||
"BINVI",
|
||||
"BLE",
|
||||
"BLEU",
|
||||
"BLEZ",
|
||||
"BLT",
|
||||
"BLTU",
|
||||
"BLTZ",
|
||||
"BNE",
|
||||
"BNEZ",
|
||||
"BSET",
|
||||
"BSETI",
|
||||
"CADD",
|
||||
"CADDI",
|
||||
"CADDI16SP",
|
||||
"CADDI4SPN",
|
||||
"CADDIW",
|
||||
"CADDW",
|
||||
"CAND",
|
||||
"CANDI",
|
||||
"CBEQZ",
|
||||
"CBNEZ",
|
||||
"CEBREAK",
|
||||
"CFLD",
|
||||
"CFLDSP",
|
||||
"CFSD",
|
||||
"CFSDSP",
|
||||
"CJ",
|
||||
"CJALR",
|
||||
"CJR",
|
||||
"CLD",
|
||||
"CLDSP",
|
||||
"CLI",
|
||||
"CLUI",
|
||||
"CLW",
|
||||
"CLWSP",
|
||||
"CLZ",
|
||||
"CLZW",
|
||||
"CMV",
|
||||
"CNOP",
|
||||
"COR",
|
||||
"CPOP",
|
||||
"CPOPW",
|
||||
"CSD",
|
||||
"CSDSP",
|
||||
"CSLLI",
|
||||
"CSRAI",
|
||||
"CSRLI",
|
||||
"CSRRC",
|
||||
"CSRRCI",
|
||||
"CSRRS",
|
||||
"CSRRSI",
|
||||
"CSRRW",
|
||||
"CSRRWI",
|
||||
"CSUB",
|
||||
"CSUBW",
|
||||
"CSW",
|
||||
"CSWSP",
|
||||
"CTZ",
|
||||
"CTZW",
|
||||
"CXOR",
|
||||
"CZEROEQZ",
|
||||
"CZERONEZ",
|
||||
"DIV",
|
||||
"DIVU",
|
||||
"DIVUW",
|
||||
"DIVW",
|
||||
"DRET",
|
||||
"EBREAK",
|
||||
"ECALL",
|
||||
"FABSD",
|
||||
"FABSS",
|
||||
"FADDD",
|
||||
"FADDQ",
|
||||
"FADDS",
|
||||
"FCLASSD",
|
||||
"FCLASSQ",
|
||||
"FCLASSS",
|
||||
"FCVTDL",
|
||||
"FCVTDLU",
|
||||
"FCVTDQ",
|
||||
"FCVTDS",
|
||||
"FCVTDW",
|
||||
"FCVTDWU",
|
||||
"FCVTLD",
|
||||
"FCVTLQ",
|
||||
"FCVTLS",
|
||||
"FCVTLUD",
|
||||
"FCVTLUQ",
|
||||
"FCVTLUS",
|
||||
"FCVTQD",
|
||||
"FCVTQL",
|
||||
"FCVTQLU",
|
||||
"FCVTQS",
|
||||
"FCVTQW",
|
||||
"FCVTQWU",
|
||||
"FCVTSD",
|
||||
"FCVTSL",
|
||||
"FCVTSLU",
|
||||
"FCVTSQ",
|
||||
"FCVTSW",
|
||||
"FCVTSWU",
|
||||
"FCVTWD",
|
||||
"FCVTWQ",
|
||||
"FCVTWS",
|
||||
"FCVTWUD",
|
||||
"FCVTWUQ",
|
||||
"FCVTWUS",
|
||||
"FDIVD",
|
||||
"FDIVQ",
|
||||
"FDIVS",
|
||||
"FENCE",
|
||||
"FEQD",
|
||||
"FEQQ",
|
||||
"FEQS",
|
||||
"FLD",
|
||||
"FLED",
|
||||
"FLEQ",
|
||||
"FLES",
|
||||
"FLQ",
|
||||
"FLTD",
|
||||
"FLTQ",
|
||||
"FLTS",
|
||||
"FLW",
|
||||
"FMADDD",
|
||||
"FMADDQ",
|
||||
"FMADDS",
|
||||
"FMAXD",
|
||||
"FMAXQ",
|
||||
"FMAXS",
|
||||
"FMIND",
|
||||
"FMINQ",
|
||||
"FMINS",
|
||||
"FMSUBD",
|
||||
"FMSUBQ",
|
||||
"FMSUBS",
|
||||
"FMULD",
|
||||
"FMULQ",
|
||||
"FMULS",
|
||||
"FMVDX",
|
||||
"FMVSX",
|
||||
"FMVWX",
|
||||
"FMVXD",
|
||||
"FMVXS",
|
||||
"FMVXW",
|
||||
"FNED",
|
||||
"FNEGD",
|
||||
"FNEGS",
|
||||
"FNES",
|
||||
"FNMADDD",
|
||||
"FNMADDQ",
|
||||
"FNMADDS",
|
||||
"FNMSUBD",
|
||||
"FNMSUBQ",
|
||||
"FNMSUBS",
|
||||
"FSD",
|
||||
"FSGNJD",
|
||||
"FSGNJND",
|
||||
"FSGNJNQ",
|
||||
"FSGNJNS",
|
||||
"FSGNJQ",
|
||||
"FSGNJS",
|
||||
"FSGNJXD",
|
||||
"FSGNJXQ",
|
||||
"FSGNJXS",
|
||||
"FSQ",
|
||||
"FSQRTD",
|
||||
"FSQRTQ",
|
||||
"FSQRTS",
|
||||
"FSUBD",
|
||||
"FSUBQ",
|
||||
"FSUBS",
|
||||
"FSW",
|
||||
"JAL",
|
||||
"JALR",
|
||||
"LB",
|
||||
"LBU",
|
||||
"LD",
|
||||
"LH",
|
||||
"LHU",
|
||||
"LRD",
|
||||
"LRW",
|
||||
"LUI",
|
||||
"LW",
|
||||
"LWU",
|
||||
"MAX",
|
||||
"MAXU",
|
||||
"MIN",
|
||||
"MINU",
|
||||
"MOV",
|
||||
"MOVB",
|
||||
"MOVBU",
|
||||
"MOVD",
|
||||
"MOVF",
|
||||
"MOVH",
|
||||
"MOVHU",
|
||||
"MOVW",
|
||||
"MOVWU",
|
||||
"MRET",
|
||||
"MUL",
|
||||
"MULH",
|
||||
"MULHSU",
|
||||
"MULHU",
|
||||
"MULW",
|
||||
"NEG",
|
||||
"NEGW",
|
||||
"NOT",
|
||||
"OR",
|
||||
"ORCB",
|
||||
"ORI",
|
||||
"ORN",
|
||||
"RDCYCLE",
|
||||
"RDINSTRET",
|
||||
"RDTIME",
|
||||
"REM",
|
||||
"REMU",
|
||||
"REMUW",
|
||||
"REMW",
|
||||
"REV8",
|
||||
"ROL",
|
||||
"ROLW",
|
||||
"ROR",
|
||||
"RORI",
|
||||
"RORIW",
|
||||
"RORW",
|
||||
"SB",
|
||||
"SBREAK",
|
||||
"SCALL",
|
||||
"SCD",
|
||||
"SCW",
|
||||
"SD",
|
||||
"SEQZ",
|
||||
"SEXTB",
|
||||
"SEXTH",
|
||||
"SFENCEVMA",
|
||||
"SH",
|
||||
"SH1ADD",
|
||||
"SH1ADDUW",
|
||||
"SH2ADD",
|
||||
"SH2ADDUW",
|
||||
"SH3ADD",
|
||||
"SH3ADDUW",
|
||||
"SLL",
|
||||
"SLLI",
|
||||
"SLLIUW",
|
||||
"SLLIW",
|
||||
"SLLW",
|
||||
"SLT",
|
||||
"SLTI",
|
||||
"SLTIU",
|
||||
"SLTU",
|
||||
"SNEZ",
|
||||
"SRA",
|
||||
"SRAI",
|
||||
"SRAIW",
|
||||
"SRAW",
|
||||
"SRET",
|
||||
"SRL",
|
||||
"SRLI",
|
||||
"SRLIW",
|
||||
"SRLW",
|
||||
"SUB",
|
||||
"SUBW",
|
||||
"SW",
|
||||
"VAADDUVV",
|
||||
"VAADDUVX",
|
||||
"VAADDVV",
|
||||
"VAADDVX",
|
||||
"VADCVIM",
|
||||
"VADCVVM",
|
||||
"VADCVXM",
|
||||
"VADDVI",
|
||||
"VADDVV",
|
||||
"VADDVX",
|
||||
"VANDVI",
|
||||
"VANDVV",
|
||||
"VANDVX",
|
||||
"VASUBUVV",
|
||||
"VASUBUVX",
|
||||
"VASUBVV",
|
||||
"VASUBVX",
|
||||
"VCOMPRESSVM",
|
||||
"VCPOPM",
|
||||
"VDIVUVV",
|
||||
"VDIVUVX",
|
||||
"VDIVVV",
|
||||
"VDIVVX",
|
||||
"VFABSV",
|
||||
"VFADDVF",
|
||||
"VFADDVV",
|
||||
"VFCLASSV",
|
||||
"VFCVTFXUV",
|
||||
"VFCVTFXV",
|
||||
"VFCVTRTZXFV",
|
||||
"VFCVTRTZXUFV",
|
||||
"VFCVTXFV",
|
||||
"VFCVTXUFV",
|
||||
"VFDIVVF",
|
||||
"VFDIVVV",
|
||||
"VFIRSTM",
|
||||
"VFMACCVF",
|
||||
"VFMACCVV",
|
||||
"VFMADDVF",
|
||||
"VFMADDVV",
|
||||
"VFMAXVF",
|
||||
"VFMAXVV",
|
||||
"VFMERGEVFM",
|
||||
"VFMINVF",
|
||||
"VFMINVV",
|
||||
"VFMSACVF",
|
||||
"VFMSACVV",
|
||||
"VFMSUBVF",
|
||||
"VFMSUBVV",
|
||||
"VFMULVF",
|
||||
"VFMULVV",
|
||||
"VFMVFS",
|
||||
"VFMVSF",
|
||||
"VFMVVF",
|
||||
"VFNCVTFFW",
|
||||
"VFNCVTFXUW",
|
||||
"VFNCVTFXW",
|
||||
"VFNCVTRODFFW",
|
||||
"VFNCVTRTZXFW",
|
||||
"VFNCVTRTZXUFW",
|
||||
"VFNCVTXFW",
|
||||
"VFNCVTXUFW",
|
||||
"VFNEGV",
|
||||
"VFNMACCVF",
|
||||
"VFNMACCVV",
|
||||
"VFNMADDVF",
|
||||
"VFNMADDVV",
|
||||
"VFNMSACVF",
|
||||
"VFNMSACVV",
|
||||
"VFNMSUBVF",
|
||||
"VFNMSUBVV",
|
||||
"VFRDIVVF",
|
||||
"VFREC7V",
|
||||
"VFREDMAXVS",
|
||||
"VFREDMINVS",
|
||||
"VFREDOSUMVS",
|
||||
"VFREDUSUMVS",
|
||||
"VFRSQRT7V",
|
||||
"VFRSUBVF",
|
||||
"VFSGNJNVF",
|
||||
"VFSGNJNVV",
|
||||
"VFSGNJVF",
|
||||
"VFSGNJVV",
|
||||
"VFSGNJXVF",
|
||||
"VFSGNJXVV",
|
||||
"VFSLIDE1DOWNVF",
|
||||
"VFSLIDE1UPVF",
|
||||
"VFSQRTV",
|
||||
"VFSUBVF",
|
||||
"VFSUBVV",
|
||||
"VFWADDVF",
|
||||
"VFWADDVV",
|
||||
"VFWADDWF",
|
||||
"VFWADDWV",
|
||||
"VFWCVTFFV",
|
||||
"VFWCVTFXUV",
|
||||
"VFWCVTFXV",
|
||||
"VFWCVTRTZXFV",
|
||||
"VFWCVTRTZXUFV",
|
||||
"VFWCVTXFV",
|
||||
"VFWCVTXUFV",
|
||||
"VFWMACCVF",
|
||||
"VFWMACCVV",
|
||||
"VFWMSACVF",
|
||||
"VFWMSACVV",
|
||||
"VFWMULVF",
|
||||
"VFWMULVV",
|
||||
"VFWNMACCVF",
|
||||
"VFWNMACCVV",
|
||||
"VFWNMSACVF",
|
||||
"VFWNMSACVV",
|
||||
"VFWREDOSUMVS",
|
||||
"VFWREDUSUMVS",
|
||||
"VFWSUBVF",
|
||||
"VFWSUBVV",
|
||||
"VFWSUBWF",
|
||||
"VFWSUBWV",
|
||||
"VIDV",
|
||||
"VIOTAM",
|
||||
"VL1RE16V",
|
||||
"VL1RE32V",
|
||||
"VL1RE64V",
|
||||
"VL1RE8V",
|
||||
"VL1RV",
|
||||
"VL2RE16V",
|
||||
"VL2RE32V",
|
||||
"VL2RE64V",
|
||||
"VL2RE8V",
|
||||
"VL2RV",
|
||||
"VL4RE16V",
|
||||
"VL4RE32V",
|
||||
"VL4RE64V",
|
||||
"VL4RE8V",
|
||||
"VL4RV",
|
||||
"VL8RE16V",
|
||||
"VL8RE32V",
|
||||
"VL8RE64V",
|
||||
"VL8RE8V",
|
||||
"VL8RV",
|
||||
"VLE16FFV",
|
||||
"VLE16V",
|
||||
"VLE32FFV",
|
||||
"VLE32V",
|
||||
"VLE64FFV",
|
||||
"VLE64V",
|
||||
"VLE8FFV",
|
||||
"VLE8V",
|
||||
"VLMV",
|
||||
"VLOXEI16V",
|
||||
"VLOXEI32V",
|
||||
"VLOXEI64V",
|
||||
"VLOXEI8V",
|
||||
"VLOXSEG2EI16V",
|
||||
"VLOXSEG2EI32V",
|
||||
"VLOXSEG2EI64V",
|
||||
"VLOXSEG2EI8V",
|
||||
"VLOXSEG3EI16V",
|
||||
"VLOXSEG3EI32V",
|
||||
"VLOXSEG3EI64V",
|
||||
"VLOXSEG3EI8V",
|
||||
"VLOXSEG4EI16V",
|
||||
"VLOXSEG4EI32V",
|
||||
"VLOXSEG4EI64V",
|
||||
"VLOXSEG4EI8V",
|
||||
"VLOXSEG5EI16V",
|
||||
"VLOXSEG5EI32V",
|
||||
"VLOXSEG5EI64V",
|
||||
"VLOXSEG5EI8V",
|
||||
"VLOXSEG6EI16V",
|
||||
"VLOXSEG6EI32V",
|
||||
"VLOXSEG6EI64V",
|
||||
"VLOXSEG6EI8V",
|
||||
"VLOXSEG7EI16V",
|
||||
"VLOXSEG7EI32V",
|
||||
"VLOXSEG7EI64V",
|
||||
"VLOXSEG7EI8V",
|
||||
"VLOXSEG8EI16V",
|
||||
"VLOXSEG8EI32V",
|
||||
"VLOXSEG8EI64V",
|
||||
"VLOXSEG8EI8V",
|
||||
"VLSE16V",
|
||||
"VLSE32V",
|
||||
"VLSE64V",
|
||||
"VLSE8V",
|
||||
"VLSEG2E16FFV",
|
||||
"VLSEG2E16V",
|
||||
"VLSEG2E32FFV",
|
||||
"VLSEG2E32V",
|
||||
"VLSEG2E64FFV",
|
||||
"VLSEG2E64V",
|
||||
"VLSEG2E8FFV",
|
||||
"VLSEG2E8V",
|
||||
"VLSEG3E16FFV",
|
||||
"VLSEG3E16V",
|
||||
"VLSEG3E32FFV",
|
||||
"VLSEG3E32V",
|
||||
"VLSEG3E64FFV",
|
||||
"VLSEG3E64V",
|
||||
"VLSEG3E8FFV",
|
||||
"VLSEG3E8V",
|
||||
"VLSEG4E16FFV",
|
||||
"VLSEG4E16V",
|
||||
"VLSEG4E32FFV",
|
||||
"VLSEG4E32V",
|
||||
"VLSEG4E64FFV",
|
||||
"VLSEG4E64V",
|
||||
"VLSEG4E8FFV",
|
||||
"VLSEG4E8V",
|
||||
"VLSEG5E16FFV",
|
||||
"VLSEG5E16V",
|
||||
"VLSEG5E32FFV",
|
||||
"VLSEG5E32V",
|
||||
"VLSEG5E64FFV",
|
||||
"VLSEG5E64V",
|
||||
"VLSEG5E8FFV",
|
||||
"VLSEG5E8V",
|
||||
"VLSEG6E16FFV",
|
||||
"VLSEG6E16V",
|
||||
"VLSEG6E32FFV",
|
||||
"VLSEG6E32V",
|
||||
"VLSEG6E64FFV",
|
||||
"VLSEG6E64V",
|
||||
"VLSEG6E8FFV",
|
||||
"VLSEG6E8V",
|
||||
"VLSEG7E16FFV",
|
||||
"VLSEG7E16V",
|
||||
"VLSEG7E32FFV",
|
||||
"VLSEG7E32V",
|
||||
"VLSEG7E64FFV",
|
||||
"VLSEG7E64V",
|
||||
"VLSEG7E8FFV",
|
||||
"VLSEG7E8V",
|
||||
"VLSEG8E16FFV",
|
||||
"VLSEG8E16V",
|
||||
"VLSEG8E32FFV",
|
||||
"VLSEG8E32V",
|
||||
"VLSEG8E64FFV",
|
||||
"VLSEG8E64V",
|
||||
"VLSEG8E8FFV",
|
||||
"VLSEG8E8V",
|
||||
"VLSSEG2E16V",
|
||||
"VLSSEG2E32V",
|
||||
"VLSSEG2E64V",
|
||||
"VLSSEG2E8V",
|
||||
"VLSSEG3E16V",
|
||||
"VLSSEG3E32V",
|
||||
"VLSSEG3E64V",
|
||||
"VLSSEG3E8V",
|
||||
"VLSSEG4E16V",
|
||||
"VLSSEG4E32V",
|
||||
"VLSSEG4E64V",
|
||||
"VLSSEG4E8V",
|
||||
"VLSSEG5E16V",
|
||||
"VLSSEG5E32V",
|
||||
"VLSSEG5E64V",
|
||||
"VLSSEG5E8V",
|
||||
"VLSSEG6E16V",
|
||||
"VLSSEG6E32V",
|
||||
"VLSSEG6E64V",
|
||||
"VLSSEG6E8V",
|
||||
"VLSSEG7E16V",
|
||||
"VLSSEG7E32V",
|
||||
"VLSSEG7E64V",
|
||||
"VLSSEG7E8V",
|
||||
"VLSSEG8E16V",
|
||||
"VLSSEG8E32V",
|
||||
"VLSSEG8E64V",
|
||||
"VLSSEG8E8V",
|
||||
"VLUXEI16V",
|
||||
"VLUXEI32V",
|
||||
"VLUXEI64V",
|
||||
"VLUXEI8V",
|
||||
"VLUXSEG2EI16V",
|
||||
"VLUXSEG2EI32V",
|
||||
"VLUXSEG2EI64V",
|
||||
"VLUXSEG2EI8V",
|
||||
"VLUXSEG3EI16V",
|
||||
"VLUXSEG3EI32V",
|
||||
"VLUXSEG3EI64V",
|
||||
"VLUXSEG3EI8V",
|
||||
"VLUXSEG4EI16V",
|
||||
"VLUXSEG4EI32V",
|
||||
"VLUXSEG4EI64V",
|
||||
"VLUXSEG4EI8V",
|
||||
"VLUXSEG5EI16V",
|
||||
"VLUXSEG5EI32V",
|
||||
"VLUXSEG5EI64V",
|
||||
"VLUXSEG5EI8V",
|
||||
"VLUXSEG6EI16V",
|
||||
"VLUXSEG6EI32V",
|
||||
"VLUXSEG6EI64V",
|
||||
"VLUXSEG6EI8V",
|
||||
"VLUXSEG7EI16V",
|
||||
"VLUXSEG7EI32V",
|
||||
"VLUXSEG7EI64V",
|
||||
"VLUXSEG7EI8V",
|
||||
"VLUXSEG8EI16V",
|
||||
"VLUXSEG8EI32V",
|
||||
"VLUXSEG8EI64V",
|
||||
"VLUXSEG8EI8V",
|
||||
"VMACCVV",
|
||||
"VMACCVX",
|
||||
"VMADCVI",
|
||||
"VMADCVIM",
|
||||
"VMADCVV",
|
||||
"VMADCVVM",
|
||||
"VMADCVX",
|
||||
"VMADCVXM",
|
||||
"VMADDVV",
|
||||
"VMADDVX",
|
||||
"VMANDMM",
|
||||
"VMANDNMM",
|
||||
"VMAXUVV",
|
||||
"VMAXUVX",
|
||||
"VMAXVV",
|
||||
"VMAXVX",
|
||||
"VMCLRM",
|
||||
"VMERGEVIM",
|
||||
"VMERGEVVM",
|
||||
"VMERGEVXM",
|
||||
"VMFEQVF",
|
||||
"VMFEQVV",
|
||||
"VMFGEVF",
|
||||
"VMFGEVV",
|
||||
"VMFGTVF",
|
||||
"VMFGTVV",
|
||||
"VMFLEVF",
|
||||
"VMFLEVV",
|
||||
"VMFLTVF",
|
||||
"VMFLTVV",
|
||||
"VMFNEVF",
|
||||
"VMFNEVV",
|
||||
"VMINUVV",
|
||||
"VMINUVX",
|
||||
"VMINVV",
|
||||
"VMINVX",
|
||||
"VMMVM",
|
||||
"VMNANDMM",
|
||||
"VMNORMM",
|
||||
"VMNOTM",
|
||||
"VMORMM",
|
||||
"VMORNMM",
|
||||
"VMSBCVV",
|
||||
"VMSBCVVM",
|
||||
"VMSBCVX",
|
||||
"VMSBCVXM",
|
||||
"VMSBFM",
|
||||
"VMSEQVI",
|
||||
"VMSEQVV",
|
||||
"VMSEQVX",
|
||||
"VMSETM",
|
||||
"VMSGEUVI",
|
||||
"VMSGEUVV",
|
||||
"VMSGEVI",
|
||||
"VMSGEVV",
|
||||
"VMSGTUVI",
|
||||
"VMSGTUVV",
|
||||
"VMSGTUVX",
|
||||
"VMSGTVI",
|
||||
"VMSGTVV",
|
||||
"VMSGTVX",
|
||||
"VMSIFM",
|
||||
"VMSLEUVI",
|
||||
"VMSLEUVV",
|
||||
"VMSLEUVX",
|
||||
"VMSLEVI",
|
||||
"VMSLEVV",
|
||||
"VMSLEVX",
|
||||
"VMSLTUVI",
|
||||
"VMSLTUVV",
|
||||
"VMSLTUVX",
|
||||
"VMSLTVI",
|
||||
"VMSLTVV",
|
||||
"VMSLTVX",
|
||||
"VMSNEVI",
|
||||
"VMSNEVV",
|
||||
"VMSNEVX",
|
||||
"VMSOFM",
|
||||
"VMULHSUVV",
|
||||
"VMULHSUVX",
|
||||
"VMULHUVV",
|
||||
"VMULHUVX",
|
||||
"VMULHVV",
|
||||
"VMULHVX",
|
||||
"VMULVV",
|
||||
"VMULVX",
|
||||
"VMV1RV",
|
||||
"VMV2RV",
|
||||
"VMV4RV",
|
||||
"VMV8RV",
|
||||
"VMVSX",
|
||||
"VMVVI",
|
||||
"VMVVV",
|
||||
"VMVVX",
|
||||
"VMVXS",
|
||||
"VMXNORMM",
|
||||
"VMXORMM",
|
||||
"VNCLIPUWI",
|
||||
"VNCLIPUWV",
|
||||
"VNCLIPUWX",
|
||||
"VNCLIPWI",
|
||||
"VNCLIPWV",
|
||||
"VNCLIPWX",
|
||||
"VNCVTXXW",
|
||||
"VNEGV",
|
||||
"VNMSACVV",
|
||||
"VNMSACVX",
|
||||
"VNMSUBVV",
|
||||
"VNMSUBVX",
|
||||
"VNOTV",
|
||||
"VNSRAWI",
|
||||
"VNSRAWV",
|
||||
"VNSRAWX",
|
||||
"VNSRLWI",
|
||||
"VNSRLWV",
|
||||
"VNSRLWX",
|
||||
"VORVI",
|
||||
"VORVV",
|
||||
"VORVX",
|
||||
"VREDANDVS",
|
||||
"VREDMAXUVS",
|
||||
"VREDMAXVS",
|
||||
"VREDMINUVS",
|
||||
"VREDMINVS",
|
||||
"VREDORVS",
|
||||
"VREDSUMVS",
|
||||
"VREDXORVS",
|
||||
"VREMUVV",
|
||||
"VREMUVX",
|
||||
"VREMVV",
|
||||
"VREMVX",
|
||||
"VRGATHEREI16VV",
|
||||
"VRGATHERVI",
|
||||
"VRGATHERVV",
|
||||
"VRGATHERVX",
|
||||
"VRSUBVI",
|
||||
"VRSUBVX",
|
||||
"VS1RV",
|
||||
"VS2RV",
|
||||
"VS4RV",
|
||||
"VS8RV",
|
||||
"VSADDUVI",
|
||||
"VSADDUVV",
|
||||
"VSADDUVX",
|
||||
"VSADDVI",
|
||||
"VSADDVV",
|
||||
"VSADDVX",
|
||||
"VSBCVVM",
|
||||
"VSBCVXM",
|
||||
"VSE16V",
|
||||
"VSE32V",
|
||||
"VSE64V",
|
||||
"VSE8V",
|
||||
"VSETIVLI",
|
||||
"VSETVL",
|
||||
"VSETVLI",
|
||||
"VSEXTVF2",
|
||||
"VSEXTVF4",
|
||||
"VSEXTVF8",
|
||||
"VSLIDE1DOWNVX",
|
||||
"VSLIDE1UPVX",
|
||||
"VSLIDEDOWNVI",
|
||||
"VSLIDEDOWNVX",
|
||||
"VSLIDEUPVI",
|
||||
"VSLIDEUPVX",
|
||||
"VSLLVI",
|
||||
"VSLLVV",
|
||||
"VSLLVX",
|
||||
"VSMULVV",
|
||||
"VSMULVX",
|
||||
"VSMV",
|
||||
"VSOXEI16V",
|
||||
"VSOXEI32V",
|
||||
"VSOXEI64V",
|
||||
"VSOXEI8V",
|
||||
"VSOXSEG2EI16V",
|
||||
"VSOXSEG2EI32V",
|
||||
"VSOXSEG2EI64V",
|
||||
"VSOXSEG2EI8V",
|
||||
"VSOXSEG3EI16V",
|
||||
"VSOXSEG3EI32V",
|
||||
"VSOXSEG3EI64V",
|
||||
"VSOXSEG3EI8V",
|
||||
"VSOXSEG4EI16V",
|
||||
"VSOXSEG4EI32V",
|
||||
"VSOXSEG4EI64V",
|
||||
"VSOXSEG4EI8V",
|
||||
"VSOXSEG5EI16V",
|
||||
"VSOXSEG5EI32V",
|
||||
"VSOXSEG5EI64V",
|
||||
"VSOXSEG5EI8V",
|
||||
"VSOXSEG6EI16V",
|
||||
"VSOXSEG6EI32V",
|
||||
"VSOXSEG6EI64V",
|
||||
"VSOXSEG6EI8V",
|
||||
"VSOXSEG7EI16V",
|
||||
"VSOXSEG7EI32V",
|
||||
"VSOXSEG7EI64V",
|
||||
"VSOXSEG7EI8V",
|
||||
"VSOXSEG8EI16V",
|
||||
"VSOXSEG8EI32V",
|
||||
"VSOXSEG8EI64V",
|
||||
"VSOXSEG8EI8V",
|
||||
"VSRAVI",
|
||||
"VSRAVV",
|
||||
"VSRAVX",
|
||||
"VSRLVI",
|
||||
"VSRLVV",
|
||||
"VSRLVX",
|
||||
"VSSE16V",
|
||||
"VSSE32V",
|
||||
"VSSE64V",
|
||||
"VSSE8V",
|
||||
"VSSEG2E16V",
|
||||
"VSSEG2E32V",
|
||||
"VSSEG2E64V",
|
||||
"VSSEG2E8V",
|
||||
"VSSEG3E16V",
|
||||
"VSSEG3E32V",
|
||||
"VSSEG3E64V",
|
||||
"VSSEG3E8V",
|
||||
"VSSEG4E16V",
|
||||
"VSSEG4E32V",
|
||||
"VSSEG4E64V",
|
||||
"VSSEG4E8V",
|
||||
"VSSEG5E16V",
|
||||
"VSSEG5E32V",
|
||||
"VSSEG5E64V",
|
||||
"VSSEG5E8V",
|
||||
"VSSEG6E16V",
|
||||
"VSSEG6E32V",
|
||||
"VSSEG6E64V",
|
||||
"VSSEG6E8V",
|
||||
"VSSEG7E16V",
|
||||
"VSSEG7E32V",
|
||||
"VSSEG7E64V",
|
||||
"VSSEG7E8V",
|
||||
"VSSEG8E16V",
|
||||
"VSSEG8E32V",
|
||||
"VSSEG8E64V",
|
||||
"VSSEG8E8V",
|
||||
"VSSRAVI",
|
||||
"VSSRAVV",
|
||||
"VSSRAVX",
|
||||
"VSSRLVI",
|
||||
"VSSRLVV",
|
||||
"VSSRLVX",
|
||||
"VSSSEG2E16V",
|
||||
"VSSSEG2E32V",
|
||||
"VSSSEG2E64V",
|
||||
"VSSSEG2E8V",
|
||||
"VSSSEG3E16V",
|
||||
"VSSSEG3E32V",
|
||||
"VSSSEG3E64V",
|
||||
"VSSSEG3E8V",
|
||||
"VSSSEG4E16V",
|
||||
"VSSSEG4E32V",
|
||||
"VSSSEG4E64V",
|
||||
"VSSSEG4E8V",
|
||||
"VSSSEG5E16V",
|
||||
"VSSSEG5E32V",
|
||||
"VSSSEG5E64V",
|
||||
"VSSSEG5E8V",
|
||||
"VSSSEG6E16V",
|
||||
"VSSSEG6E32V",
|
||||
"VSSSEG6E64V",
|
||||
"VSSSEG6E8V",
|
||||
"VSSSEG7E16V",
|
||||
"VSSSEG7E32V",
|
||||
"VSSSEG7E64V",
|
||||
"VSSSEG7E8V",
|
||||
"VSSSEG8E16V",
|
||||
"VSSSEG8E32V",
|
||||
"VSSSEG8E64V",
|
||||
"VSSSEG8E8V",
|
||||
"VSSUBUVV",
|
||||
"VSSUBUVX",
|
||||
"VSSUBVV",
|
||||
"VSSUBVX",
|
||||
"VSUBVV",
|
||||
"VSUBVX",
|
||||
"VSUXEI16V",
|
||||
"VSUXEI32V",
|
||||
"VSUXEI64V",
|
||||
"VSUXEI8V",
|
||||
"VSUXSEG2EI16V",
|
||||
"VSUXSEG2EI32V",
|
||||
"VSUXSEG2EI64V",
|
||||
"VSUXSEG2EI8V",
|
||||
"VSUXSEG3EI16V",
|
||||
"VSUXSEG3EI32V",
|
||||
"VSUXSEG3EI64V",
|
||||
"VSUXSEG3EI8V",
|
||||
"VSUXSEG4EI16V",
|
||||
"VSUXSEG4EI32V",
|
||||
"VSUXSEG4EI64V",
|
||||
"VSUXSEG4EI8V",
|
||||
"VSUXSEG5EI16V",
|
||||
"VSUXSEG5EI32V",
|
||||
"VSUXSEG5EI64V",
|
||||
"VSUXSEG5EI8V",
|
||||
"VSUXSEG6EI16V",
|
||||
"VSUXSEG6EI32V",
|
||||
"VSUXSEG6EI64V",
|
||||
"VSUXSEG6EI8V",
|
||||
"VSUXSEG7EI16V",
|
||||
"VSUXSEG7EI32V",
|
||||
"VSUXSEG7EI64V",
|
||||
"VSUXSEG7EI8V",
|
||||
"VSUXSEG8EI16V",
|
||||
"VSUXSEG8EI32V",
|
||||
"VSUXSEG8EI64V",
|
||||
"VSUXSEG8EI8V",
|
||||
"VWADDUVV",
|
||||
"VWADDUVX",
|
||||
"VWADDUWV",
|
||||
"VWADDUWX",
|
||||
"VWADDVV",
|
||||
"VWADDVX",
|
||||
"VWADDWV",
|
||||
"VWADDWX",
|
||||
"VWCVTUXXV",
|
||||
"VWCVTXXV",
|
||||
"VWMACCSUVV",
|
||||
"VWMACCSUVX",
|
||||
"VWMACCUSVX",
|
||||
"VWMACCUVV",
|
||||
"VWMACCUVX",
|
||||
"VWMACCVV",
|
||||
"VWMACCVX",
|
||||
"VWMULSUVV",
|
||||
"VWMULSUVX",
|
||||
"VWMULUVV",
|
||||
"VWMULUVX",
|
||||
"VWMULVV",
|
||||
"VWMULVX",
|
||||
"VWREDSUMUVS",
|
||||
"VWREDSUMVS",
|
||||
"VWSUBUVV",
|
||||
"VWSUBUVX",
|
||||
"VWSUBUWV",
|
||||
"VWSUBUWX",
|
||||
"VWSUBVV",
|
||||
"VWSUBVX",
|
||||
"VWSUBWV",
|
||||
"VWSUBWX",
|
||||
"VXORVI",
|
||||
"VXORVV",
|
||||
"VXORVX",
|
||||
"VZEXTVF2",
|
||||
"VZEXTVF4",
|
||||
"VZEXTVF8",
|
||||
"WFI",
|
||||
"WORD",
|
||||
"XNOR",
|
||||
"XOR",
|
||||
"XORI",
|
||||
"ZEXTH",
|
||||
}
|
||||
+292
@@ -0,0 +1,292 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
// Assemble encodes the body of a TEXT function into x86-64 machine code,
|
||||
// resolving local labels to relative jump offsets and translating the FP/SP
|
||||
// pseudo-registers onto the hardware stack pointer (matching the Go
|
||||
// assembler's default frame-pointer behaviour). Jumps always use the 32-bit
|
||||
// relative form so instruction sizes are fixed and offsets resolve in a single
|
||||
// layout pass.
|
||||
//
|
||||
// Supported operands: registers, memory (real base register), immediates,
|
||||
// FP/SP frame-relative operands, and local-label jumps. SB (global symbol)
|
||||
// operands require relocations and are not yet supported; SIMD (VEX/EVEX)
|
||||
// instructions are pending.
|
||||
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
||||
fi := computeFrame(t)
|
||||
|
||||
// Pass 1: lay out instructions (including prologue/epilogue) to fix label
|
||||
// offsets.
|
||||
offsets := map[string]int{}
|
||||
sizes := make([]int, len(t.Body))
|
||||
pos := len(fi.prologue)
|
||||
for i, stmt := range t.Body {
|
||||
switch s := stmt.(type) {
|
||||
case *ast.Label:
|
||||
offsets[s.Name.Text] = pos
|
||||
case *ast.Instr:
|
||||
sz, err := instrSize(s, fi)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
}
|
||||
sizes[i] = sz
|
||||
pos += sz
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2: emit.
|
||||
out := append([]byte(nil), fi.prologue...)
|
||||
pos = len(fi.prologue)
|
||||
for i, stmt := range t.Body {
|
||||
s, ok := stmt.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, err := encodeInstr(s, pos, offsets, fi)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||
}
|
||||
if len(code) != sizes[i] {
|
||||
return nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||
}
|
||||
out = append(out, code...)
|
||||
pos += len(code)
|
||||
}
|
||||
return out, offsets, nil
|
||||
}
|
||||
|
||||
// frameInfo carries the frame layout derived from the TEXT directive.
|
||||
type frameInfo struct {
|
||||
size int // local frame size ($framesize)
|
||||
useFP bool // a frame pointer (BP) is set up
|
||||
fpAdjust int64 // added to x+N(FP) to reach the hardware SP-relative offset
|
||||
spAdjust int64 // x-N(SP) becomes (spAdjust - N)(SP)
|
||||
prologue []byte
|
||||
epilogue []byte
|
||||
}
|
||||
|
||||
// computeFrame derives the frame layout, matching the Go assembler's default
|
||||
// (a frame pointer is used whenever the function has a non-zero frame).
|
||||
func computeFrame(t *ast.Text) frameInfo {
|
||||
fi := frameInfo{}
|
||||
if t.Frame != nil && t.Frame.Imm.HasVal {
|
||||
fi.size = int(t.Frame.Imm.Val)
|
||||
}
|
||||
if fi.size > 0 {
|
||||
fi.useFP = true
|
||||
fi.fpAdjust = int64(fi.size) + 16 // frame + saved BP + return address
|
||||
fi.spAdjust = int64(fi.size)
|
||||
fi.prologue = prologueBytes(fi.size)
|
||||
fi.epilogue = epilogueBytes(fi.size)
|
||||
} else {
|
||||
fi.fpAdjust = 8 // return address only
|
||||
}
|
||||
return fi
|
||||
}
|
||||
|
||||
// prologueBytes emits: PUSHQ BP; MOVQ SP, BP; SUBQ $size, SP.
|
||||
func prologueBytes(size int) []byte {
|
||||
out := []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
|
||||
return append(out, subSP(size)...)
|
||||
}
|
||||
|
||||
// epilogueBytes emits: ADDQ $size, SP; POPQ BP.
|
||||
func epilogueBytes(size int) []byte {
|
||||
out := addSP(size)
|
||||
return append(out, 0x5D) // POPQ BP
|
||||
}
|
||||
|
||||
func subSP(size int) []byte { // SUBQ $size, SP
|
||||
if size >= -128 && size <= 127 {
|
||||
return []byte{0x48, 0x83, 0xEC, byte(int8(size))}
|
||||
}
|
||||
return append([]byte{0x48, 0x81, 0xEC}, le32(int64(size))...)
|
||||
}
|
||||
|
||||
func addSP(size int) []byte { // ADDQ $size, SP
|
||||
if size >= -128 && size <= 127 {
|
||||
return []byte{0x48, 0x83, 0xC4, byte(int8(size))}
|
||||
}
|
||||
return append([]byte{0x48, 0x81, 0xC4}, le32(int64(size))...)
|
||||
}
|
||||
|
||||
// instrSize returns the encoded length of an instruction (pass 1). encodeInstr
|
||||
// already includes the epilogue for a RET in a frame-pointer function; jumps use
|
||||
// a fixed rel32 size (no epilogue).
|
||||
func instrSize(s *ast.Instr, fi frameInfo) (int, error) {
|
||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||
if isJumpMnemonic(mnem) {
|
||||
return jumpSize(mnem), nil
|
||||
}
|
||||
code, err := encodeInstr(s, 0, nil, fi)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(code), nil
|
||||
}
|
||||
|
||||
func isJumpMnemonic(mnem string) bool {
|
||||
if mnem == "JMP" || mnem == "CALL" {
|
||||
return true
|
||||
}
|
||||
_, ok := condCode(mnem)
|
||||
return ok
|
||||
}
|
||||
|
||||
// jumpSize returns the fixed length of a rel32 jump instruction.
|
||||
func jumpSize(mnem string) int {
|
||||
if mnem == "JMP" || mnem == "CALL" {
|
||||
return 5 // opcode + rel32
|
||||
}
|
||||
return 6 // 0x0F 0x8x + rel32
|
||||
}
|
||||
|
||||
// encodeInstr encodes one instruction, resolving jump targets against offsets
|
||||
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
|
||||
// function is prefixed with the epilogue.
|
||||
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo) ([]byte, error) {
|
||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||
|
||||
var prefix []byte
|
||||
if mnem == "RET" && fi.useFP {
|
||||
prefix = fi.epilogue
|
||||
}
|
||||
|
||||
var code []byte
|
||||
var err error
|
||||
if isJumpMnemonic(mnem) {
|
||||
code, err = encodeJump(s, mnem, pc+len(prefix), offsets)
|
||||
} else {
|
||||
code, err = encodeNormal(s, fi)
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return append(prefix, code...), nil
|
||||
}
|
||||
|
||||
func encodeNormal(s *ast.Instr, fi frameInfo) ([]byte, error) {
|
||||
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
|
||||
if size == 0 {
|
||||
size = 8
|
||||
}
|
||||
ops := make([]Operand, len(s.Operands))
|
||||
for i, op := range s.Operands {
|
||||
o, err := operandFromAST(op, size, fi)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
ops[i] = o
|
||||
}
|
||||
return Encode(s.Mnemonic.Text, ops...)
|
||||
}
|
||||
|
||||
// encodeJump encodes a JMP/CALL/Jcc with a rel32 offset resolved from the
|
||||
// target label.
|
||||
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int) ([]byte, error) {
|
||||
if len(s.Operands) != 1 {
|
||||
return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands))
|
||||
}
|
||||
name, ok := labelName(s.Operands[0])
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("jump target must be a local label")
|
||||
}
|
||||
target, ok := offsets[name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q", name)
|
||||
}
|
||||
rel := int64(target - (pc + jumpSize(mnem)))
|
||||
|
||||
switch mnem {
|
||||
case "JMP":
|
||||
return append([]byte{0xE9}, le32(rel)...), nil
|
||||
case "CALL":
|
||||
return append([]byte{0xE8}, le32(rel)...), nil
|
||||
default:
|
||||
cc, _ := condCode(mnem)
|
||||
return append([]byte{0x0F, 0x80 + byte(cc)}, le32(rel)...), nil
|
||||
}
|
||||
}
|
||||
|
||||
// labelName extracts a local-label name from a jump operand.
|
||||
func labelName(op *ast.Operand) (string, bool) {
|
||||
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
|
||||
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
|
||||
return op.Addr.Sym.Name, true
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// spReg is the hardware stack pointer used to realise FP/SP pseudo-operands.
|
||||
var spReg = Reg{idx: 4, size: 8}
|
||||
|
||||
// operandFromAST converts a parsed operand into an encoder Operand, applying
|
||||
// the frame translation to FP/SP pseudo-register operands.
|
||||
func operandFromAST(op *ast.Operand, size int, fi frameInfo) (Operand, error) {
|
||||
switch op.Kind {
|
||||
case ast.OpImmediate:
|
||||
if op.Imm.HasVal {
|
||||
v := op.Imm.Val
|
||||
if op.Imm.Neg {
|
||||
v = -v
|
||||
}
|
||||
return Imm(v), nil
|
||||
}
|
||||
return nil, fmt.Errorf("non-integer immediate not supported")
|
||||
|
||||
case ast.OpAddr:
|
||||
a := op.Addr
|
||||
|
||||
// FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the
|
||||
// symbol, not the address displacement.
|
||||
if a.Sym != nil && a.Sym.Pseudo == "FP" {
|
||||
off := a.Sym.Offset + fi.fpAdjust
|
||||
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
|
||||
}
|
||||
// SP-relative local: x-N(SP) → (spAdjust + offset)(SP).
|
||||
if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" {
|
||||
off := fi.spAdjust + a.Sym.Offset
|
||||
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
|
||||
}
|
||||
// SB (global symbol) needs a relocation — not yet supported.
|
||||
if a.Sym != nil && a.Sym.Pseudo == "SB" {
|
||||
return nil, fmt.Errorf("SB (global symbol) operands need relocation support (pending)")
|
||||
}
|
||||
|
||||
// Memory with a real base register: (base), off(base), (base)(index*scale).
|
||||
if a.Base != "" {
|
||||
base, ok := ParseReg(a.Base)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("unknown base register %q", a.Base)
|
||||
}
|
||||
m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size}
|
||||
if a.Index != "" {
|
||||
idx, ok := ParseReg(a.Index)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("unknown index register %q", a.Index)
|
||||
}
|
||||
m.Index = idx
|
||||
m.Scale = a.Scale
|
||||
m.HasIndex = true
|
||||
}
|
||||
return m, nil
|
||||
}
|
||||
// Bare register.
|
||||
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
|
||||
if r, ok := ParseReg(a.Sym.Name); ok {
|
||||
return r, nil
|
||||
}
|
||||
}
|
||||
return nil, fmt.Errorf("operand form not yet supported")
|
||||
}
|
||||
return nil, fmt.Errorf("unsupported operand")
|
||||
}
|
||||
@@ -0,0 +1,201 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// firstText parses src and returns its first TEXT function.
|
||||
func firstText(t *testing.T, src string) *ast.Text {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("f_amd64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
for _, d := range f.Decls {
|
||||
if txt, ok := d.(*ast.Text); ok {
|
||||
return txt
|
||||
}
|
||||
}
|
||||
t.Fatal("no TEXT function found")
|
||||
return nil
|
||||
}
|
||||
|
||||
// disasm decodes a machine-code blob into Intel-syntax instruction strings.
|
||||
func disasm(t *testing.T, code []byte) []string {
|
||||
t.Helper()
|
||||
var out []string
|
||||
for len(code) > 0 {
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Fatalf("decode %x: %v", code, err)
|
||||
}
|
||||
out = append(out, x86asm.IntelSyntax(inst, 0, nil))
|
||||
code = code[inst.Len:]
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func hexBytes(b []byte) string {
|
||||
var sb strings.Builder
|
||||
for _, x := range b {
|
||||
sb.WriteString(" ")
|
||||
const hexdig = "0123456789abcdef"
|
||||
sb.WriteByte(hexdig[x>>4])
|
||||
sb.WriteByte(hexdig[x&0xf])
|
||||
}
|
||||
return strings.TrimSpace(sb.String())
|
||||
}
|
||||
|
||||
func TestAssembleLoop(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
XORQ AX, AX
|
||||
loop:
|
||||
ADDQ $1, AX
|
||||
CMPQ $10, AX
|
||||
JLT loop
|
||||
RET
|
||||
`)
|
||||
code, labels, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
if _, ok := labels["loop"]; !ok {
|
||||
t.Fatalf("label 'loop' not recorded: %v", labels)
|
||||
}
|
||||
got := strings.Join(disasm(t, code), "\n")
|
||||
want := strings.Join([]string{
|
||||
"xor rax, rax",
|
||||
"add rax, 0x1",
|
||||
"cmp rax, 0xa",
|
||||
"jl 0x0",
|
||||
"ret",
|
||||
}, "\n")
|
||||
gotLines := strings.Split(got, "\n")
|
||||
wantLines := strings.Split(want, "\n")
|
||||
if len(gotLines) != len(wantLines) {
|
||||
t.Fatalf("instruction count mismatch:\n got:\n%s\n want:\n%s", got, want)
|
||||
}
|
||||
for i := range wantLines {
|
||||
if strings.HasPrefix(wantLines[i], "jl") {
|
||||
if !strings.HasPrefix(gotLines[i], "jl") {
|
||||
t.Errorf("line %d: got %q, want a jl", i, gotLines[i])
|
||||
}
|
||||
continue
|
||||
}
|
||||
if gotLines[i] != wantLines[i] {
|
||||
t.Errorf("line %d: got %q, want %q", i, gotLines[i], wantLines[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssembleMemory(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·g(SB), NOSPLIT, $0
|
||||
MOVQ (AX), BX
|
||||
MOVQ 8(AX), CX
|
||||
LEAQ (AX)(BX*4), DX
|
||||
RET
|
||||
`)
|
||||
code, _, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
got := strings.Join(disasm(t, code), "\n")
|
||||
want := strings.Join([]string{
|
||||
"mov rbx, qword ptr [rax]",
|
||||
"mov rcx, qword ptr [rax+0x8]",
|
||||
"lea rdx, ptr [rax+4*rbx]",
|
||||
"ret",
|
||||
}, "\n")
|
||||
if got != want {
|
||||
t.Errorf("assemble memory:\n got:\n%s\n want:\n%s", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleFP verifies the FP pseudo-register translation for a NOSPLIT $0
|
||||
// function against the exact bytes the Go assembler produces (verified via
|
||||
// `go tool objdump`): x+N(FP) maps to (N+8)(SP).
|
||||
func TestAssembleFP(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·loadarg(SB), NOSPLIT, $0-24
|
||||
MOVQ p+0(FP), AX
|
||||
MOVQ n+8(FP), CX
|
||||
ADDQ CX, AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
`)
|
||||
code, _, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
// From `go tool objdump` of the Go-assembled function:
|
||||
// MOVQ 0x8(SP), AX 488b442408
|
||||
// MOVQ 0x10(SP), CX 488b4c2410
|
||||
// ADDQ CX, AX 4801c8
|
||||
// MOVQ AX, 0x18(SP) 4889442418
|
||||
// RET c3
|
||||
want := []byte{
|
||||
0x48, 0x8b, 0x44, 0x24, 0x08,
|
||||
0x48, 0x8b, 0x4c, 0x24, 0x10,
|
||||
0x48, 0x01, 0xc8,
|
||||
0x48, 0x89, 0x44, 0x24, 0x18,
|
||||
0xc3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
|
||||
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
|
||||
// the bytes the Go assembler produces.
|
||||
func TestAssembleFrame(t *testing.T) {
|
||||
fn := firstText(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·withframe(SB), NOSPLIT, $16-16
|
||||
MOVQ a+0(FP), AX
|
||||
MOVQ b+8(FP), CX
|
||||
ADDQ CX, AX
|
||||
MOVQ AX, ret+16(FP)
|
||||
RET
|
||||
`)
|
||||
code, _, err := Assemble(fn)
|
||||
if err != nil {
|
||||
t.Fatalf("Assemble: %v", err)
|
||||
}
|
||||
// From `go tool objdump`:
|
||||
// PUSHQ BP 55
|
||||
// MOVQ SP, BP 4889e5
|
||||
// SUBQ $0x10, SP 4883ec10
|
||||
// MOVQ 0x20(SP), AX 488b442420 (0 + 16 + 16)
|
||||
// MOVQ 0x28(SP), CX 488b4c2428 (8 + 16 + 16)
|
||||
// ADDQ CX, AX 4801c8
|
||||
// MOVQ AX, 0x30(SP) 4889442430 (16 + 16 + 16)
|
||||
// ADDQ $0x10, SP 4883c410
|
||||
// POPQ BP 5d
|
||||
// RET c3
|
||||
want := []byte{
|
||||
0x55, 0x48, 0x89, 0xe5, 0x48, 0x83, 0xec, 0x10,
|
||||
0x48, 0x8b, 0x44, 0x24, 0x20,
|
||||
0x48, 0x8b, 0x4c, 0x24, 0x28,
|
||||
0x48, 0x01, 0xc8,
|
||||
0x48, 0x89, 0x44, 0x24, 0x30,
|
||||
0x48, 0x83, 0xc4, 0x10, 0x5d, 0xc3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("frame translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
+288
@@ -0,0 +1,288 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Encode encodes one Plan 9 instruction (mnemonic plus operands, in source
|
||||
// order) into x86-64 machine code.
|
||||
func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
|
||||
e := &enc{}
|
||||
if err := e.encode(mnemonic, ops); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return e.out, nil
|
||||
}
|
||||
|
||||
type enc struct {
|
||||
out []byte
|
||||
}
|
||||
|
||||
func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
upper := strings.ToUpper(mnem)
|
||||
|
||||
// Fixed-name instructions (no size suffix).
|
||||
switch {
|
||||
case upper == "RET":
|
||||
return e.encodeRet()
|
||||
case upper == "NOP":
|
||||
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
|
||||
case upper == "CALL":
|
||||
return e.encodeJmpRel(ops, []byte{0xE8})
|
||||
case upper == "JMP":
|
||||
return e.encodeJmpRel(ops, []byte{0xE9})
|
||||
}
|
||||
if cc, ok := condCode(upper); ok {
|
||||
return e.encodeJcc(cc, ops)
|
||||
}
|
||||
|
||||
// VEX (AVX/AVX2) instructions: the trailing B/W/L/Q/D is part of the
|
||||
// mnemonic, not a size suffix, so dispatch before splitSize.
|
||||
if isVex(upper) {
|
||||
return e.encodeVex(upper, ops)
|
||||
}
|
||||
|
||||
base, size := splitSize(upper)
|
||||
if size == 0 {
|
||||
size = 8 // default operand size in 64-bit mode (e.g. PUSHQ)
|
||||
}
|
||||
switch base {
|
||||
case "MOV":
|
||||
return e.encodeMov(ops, size)
|
||||
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
|
||||
return e.encodeALU(aluOp[base], ops, size)
|
||||
case "TEST":
|
||||
return e.encodeTest(ops, size)
|
||||
case "LEA":
|
||||
return e.encodeLea(ops, size)
|
||||
case "INC", "DEC", "NEG", "NOT":
|
||||
return e.encodeUnary(unaryOp[base], ops, size)
|
||||
case "SHL", "SHR", "SAR":
|
||||
return e.encodeShift(shiftOp[base], ops, size)
|
||||
case "IMUL":
|
||||
return e.encodeImul(ops, size)
|
||||
case "PUSH":
|
||||
return e.encodePushPop(ops, true)
|
||||
case "POP":
|
||||
return e.encodePushPop(ops, false)
|
||||
}
|
||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||
}
|
||||
|
||||
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
||||
func splitSize(upper string) (base string, size int) {
|
||||
if upper == "" {
|
||||
return upper, 0
|
||||
}
|
||||
switch upper[len(upper)-1] {
|
||||
case 'B':
|
||||
return upper[:len(upper)-1], 1
|
||||
case 'W':
|
||||
return upper[:len(upper)-1], 2
|
||||
case 'L':
|
||||
return upper[:len(upper)-1], 4
|
||||
case 'Q':
|
||||
return upper[:len(upper)-1], 8
|
||||
}
|
||||
return upper, 0
|
||||
}
|
||||
|
||||
// --- instruction components -------------------------------------------------
|
||||
|
||||
type instr struct {
|
||||
opSize16 bool
|
||||
rexW bool
|
||||
rexR bool
|
||||
rexX bool
|
||||
rexB bool
|
||||
rexForced bool // REX needed even with all bits zero (8-bit low registers)
|
||||
opcode []byte
|
||||
modrm int // -1 if absent
|
||||
sib int // -1 if absent
|
||||
disp []byte
|
||||
imm []byte
|
||||
}
|
||||
|
||||
func (e *enc) emit(i *instr) error {
|
||||
if i.opSize16 {
|
||||
e.out = append(e.out, 0x66)
|
||||
}
|
||||
rex := byte(0)
|
||||
if i.rexW {
|
||||
rex |= 0x08
|
||||
}
|
||||
if i.rexR {
|
||||
rex |= 0x04
|
||||
}
|
||||
if i.rexX {
|
||||
rex |= 0x02
|
||||
}
|
||||
if i.rexB {
|
||||
rex |= 0x01
|
||||
}
|
||||
if rex != 0 || i.rexForced {
|
||||
e.out = append(e.out, 0x40|rex)
|
||||
}
|
||||
e.out = append(e.out, i.opcode...)
|
||||
if i.modrm >= 0 {
|
||||
e.out = append(e.out, byte(i.modrm))
|
||||
}
|
||||
if i.sib >= 0 {
|
||||
e.out = append(e.out, byte(i.sib))
|
||||
}
|
||||
e.out = append(e.out, i.disp...)
|
||||
e.out = append(e.out, i.imm...)
|
||||
return nil
|
||||
}
|
||||
|
||||
// newInstr starts an instruction with a size-derived REX.W and 0x66 prefix.
|
||||
func newInstr(opSize int, opcode []byte) *instr {
|
||||
return &instr{
|
||||
opSize16: opSize == 2,
|
||||
rexW: opSize == 8,
|
||||
opcode: opcode,
|
||||
modrm: -1,
|
||||
sib: -1,
|
||||
}
|
||||
}
|
||||
|
||||
// --- ModR/M, SIB, displacement ----------------------------------------------
|
||||
|
||||
// setRM fills in the ModR/M (and SIB, displacement, REX bits) for an
|
||||
// instruction whose reg field holds a real register `reg` and whose r/m field
|
||||
// holds `rm`.
|
||||
func setRM(i *instr, reg Reg, rm Operand, opSize int) error {
|
||||
return setRMReg(i, reg.idx&7, reg.idx >= 8, reg.needsREX(opSize), rm, opSize)
|
||||
}
|
||||
|
||||
// setRMDigit fills in the ModR/M for an instruction whose reg field is an
|
||||
// opcode /digit extension (0–7), which carries none of the register REX rules.
|
||||
func setRMDigit(i *instr, digit int, rm Operand, opSize int) error {
|
||||
return setRMReg(i, digit, false, false, rm, opSize)
|
||||
}
|
||||
|
||||
func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize int) error {
|
||||
i.rexR = rexR
|
||||
if regForced {
|
||||
i.rexForced = true
|
||||
}
|
||||
|
||||
switch r := rm.(type) {
|
||||
case Reg:
|
||||
i.rexB = r.idx >= 8
|
||||
if r.needsREX(opSize) {
|
||||
i.rexForced = true
|
||||
}
|
||||
i.modrm = 0xC0 | regField<<3 | (r.idx & 7)
|
||||
return nil
|
||||
case Mem:
|
||||
return setMem(i, regField, r)
|
||||
default:
|
||||
return fmt.Errorf("invalid r/m operand %T", rm)
|
||||
}
|
||||
}
|
||||
|
||||
func setMem(i *instr, regField int, m Mem) error {
|
||||
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i.modrm = modrm
|
||||
i.sib = sib
|
||||
i.disp = disp
|
||||
i.rexX = xBit == 1
|
||||
i.rexB = bBit == 1
|
||||
return nil
|
||||
}
|
||||
|
||||
// memComponents computes the ModR/M byte (with the given reg field), the SIB
|
||||
// byte (-1 if none), the displacement bytes, and the high index/base bits, for
|
||||
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
||||
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
|
||||
sib = -1
|
||||
// RIP-relative: neither base nor index.
|
||||
if !m.HasBase && !m.HasIndex {
|
||||
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
|
||||
}
|
||||
|
||||
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
|
||||
|
||||
var mod int
|
||||
switch {
|
||||
case !m.HasBase:
|
||||
mod = 0
|
||||
disp = le32(m.Disp)
|
||||
case m.Base.idx&7 == 5 && m.Disp == 0:
|
||||
mod = 1
|
||||
disp = []byte{0}
|
||||
case m.Disp == 0:
|
||||
mod = 0
|
||||
case fits8(m.Disp):
|
||||
mod = 1
|
||||
disp = []byte{byte(int8(m.Disp))}
|
||||
default:
|
||||
mod = 2
|
||||
disp = le32(m.Disp)
|
||||
}
|
||||
|
||||
if needSIB {
|
||||
idxField := 4 // 100 = no index
|
||||
if m.HasIndex {
|
||||
idxField = m.Index.idx & 7
|
||||
if m.Index.idx >= 8 {
|
||||
xBit = 1
|
||||
}
|
||||
}
|
||||
baseField := 5 // 101 = no base (with mod=00 → disp32)
|
||||
if m.HasBase {
|
||||
baseField = m.Base.idx & 7
|
||||
if m.Base.idx >= 8 {
|
||||
bBit = 1
|
||||
}
|
||||
}
|
||||
return mod<<6 | regField<<3 | 0x04, scaleBits(m.Scale)<<6 | idxField<<3 | baseField, disp, xBit, bBit, nil
|
||||
}
|
||||
|
||||
if m.Base.idx >= 8 {
|
||||
bBit = 1
|
||||
}
|
||||
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 0, bBit, nil
|
||||
}
|
||||
|
||||
func scaleBits(scale int) int {
|
||||
switch scale {
|
||||
case 2:
|
||||
return 1
|
||||
case 4:
|
||||
return 2
|
||||
case 8:
|
||||
return 3
|
||||
default:
|
||||
return 0 // scale 1 (or unset)
|
||||
}
|
||||
}
|
||||
|
||||
func fits8(v int64) bool { return v >= -128 && v <= 127 }
|
||||
|
||||
func le32(v int64) []byte {
|
||||
u := uint32(v)
|
||||
return []byte{byte(u), byte(u >> 8), byte(u >> 16), byte(u >> 24)}
|
||||
}
|
||||
|
||||
func le16(v int64) []byte {
|
||||
u := uint16(v)
|
||||
return []byte{byte(u), byte(u >> 8)}
|
||||
}
|
||||
|
||||
func le64(v int64) []byte {
|
||||
u := uint64(v)
|
||||
b := make([]byte, 8)
|
||||
for i := 0; i < 8; i++ {
|
||||
b[i] = byte(u >> (8 * i))
|
||||
}
|
||||
return b
|
||||
}
|
||||
@@ -0,0 +1,132 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
)
|
||||
|
||||
// decode encodes an instruction and decodes it back, returning the decoded
|
||||
// instruction and its Intel-syntax rendering.
|
||||
func decode(t *testing.T, mnemonic string, ops ...Operand) (x86asm.Inst, string) {
|
||||
t.Helper()
|
||||
code, err := Encode(mnemonic, ops...)
|
||||
if err != nil {
|
||||
t.Fatalf("Encode(%s): %v", mnemonic, err)
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(%x) of %s: %v", code, mnemonic, err)
|
||||
}
|
||||
if inst.Len != len(code) {
|
||||
t.Fatalf("Decode consumed %d of %d bytes for %s (%x)", inst.Len, len(code), mnemonic, code)
|
||||
}
|
||||
return inst, x86asm.IntelSyntax(inst, 0, nil)
|
||||
}
|
||||
|
||||
// checkSyntax asserts an instruction encodes and decodes to the expected
|
||||
// Intel-syntax string.
|
||||
func checkSyntax(t *testing.T, want, mnemonic string, ops ...Operand) {
|
||||
t.Helper()
|
||||
_, got := decode(t, mnemonic, ops...)
|
||||
if got != want {
|
||||
t.Errorf("%s: got %q, want %q", mnemonic, got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// checkOp asserts the decoded opcode (used for relative jumps, whose rendered
|
||||
// target depends on the program counter).
|
||||
func checkOp(t *testing.T, want x86asm.Op, mnemonic string, ops ...Operand) {
|
||||
t.Helper()
|
||||
inst, _ := decode(t, mnemonic, ops...)
|
||||
if inst.Op != want {
|
||||
t.Errorf("%s: got op %v, want %v", mnemonic, inst.Op, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMov(t *testing.T) {
|
||||
checkSyntax(t, "mov rbx, rax", "MOVQ", AX, BX)
|
||||
checkSyntax(t, "mov ebx, eax", "MOVL", AX, BX)
|
||||
checkSyntax(t, "mov bl, al", "MOVB", AL, BL)
|
||||
checkSyntax(t, "mov rax, rbx", "MOVQ", BX, AX)
|
||||
checkSyntax(t, "mov rbx, qword ptr [rax]", "MOVQ", Ptr(AX, 0, 8), BX)
|
||||
checkSyntax(t, "mov qword ptr [rbx], rax", "MOVQ", AX, Ptr(BX, 0, 8))
|
||||
checkSyntax(t, "mov rbx, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), BX)
|
||||
checkSyntax(t, "mov rbx, qword ptr [rsi+4*rbx]", "MOVQ", Idx(SI, BX, 4, 0, 8), BX)
|
||||
checkSyntax(t, "mov rax, 0x5", "MOVQ", Imm(5), AX)
|
||||
checkSyntax(t, "mov r8, 0x5", "MOVQ", Imm(5), Reg{idx: 8, size: 8})
|
||||
checkSyntax(t, "mov qword ptr [rax], 0x5", "MOVQ", Imm(5), Ptr(AX, 0, 8))
|
||||
checkSyntax(t, "mov r12, r13", "MOVQ", Reg{idx: 13, size: 8}, Reg{idx: 12, size: 8})
|
||||
}
|
||||
|
||||
func TestALU(t *testing.T) {
|
||||
checkSyntax(t, "add rbx, rax", "ADDQ", AX, BX)
|
||||
checkSyntax(t, "add rax, 0x1", "ADDQ", Imm(1), AX)
|
||||
checkSyntax(t, "add rax, 0x12c", "ADDQ", Imm(300), AX)
|
||||
checkSyntax(t, "sub rdx, rcx", "SUBQ", CX, DX)
|
||||
checkSyntax(t, "and rbx, 0x7", "ANDQ", Imm(7), BX)
|
||||
checkSyntax(t, "or rcx, rbx", "ORQ", BX, CX)
|
||||
checkSyntax(t, "xor rax, rax", "XORQ", AX, AX)
|
||||
checkSyntax(t, "cmp r10, rsi", "CMPQ", SI, Reg{idx: 10, size: 8})
|
||||
checkSyntax(t, "add rbx, qword ptr [rax]", "ADDQ", Ptr(AX, 0, 8), BX)
|
||||
checkSyntax(t, "add qword ptr [rax], rbx", "ADDQ", BX, Ptr(AX, 0, 8))
|
||||
checkSyntax(t, "cmp rbx, -0x20", "CMPQ", Imm(-32), BX)
|
||||
}
|
||||
|
||||
func TestLea(t *testing.T) {
|
||||
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
|
||||
checkSyntax(t, "lea rax, ptr [rbx+0x8]", "LEAQ", Ptr(BX, 0x8, 8), AX)
|
||||
}
|
||||
|
||||
func TestTest(t *testing.T) {
|
||||
checkSyntax(t, "test rax, rax", "TESTQ", AX, AX)
|
||||
checkSyntax(t, "test rbx, 0x7", "TESTQ", Imm(7), BX)
|
||||
}
|
||||
|
||||
func TestPushPop(t *testing.T) {
|
||||
checkSyntax(t, "push rbx", "PUSHQ", BX)
|
||||
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
|
||||
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
|
||||
}
|
||||
|
||||
func TestUnary(t *testing.T) {
|
||||
checkSyntax(t, "inc rax", "INCQ", AX)
|
||||
checkSyntax(t, "dec rbx", "DECQ", BX)
|
||||
checkSyntax(t, "neg rcx", "NEGQ", CX)
|
||||
checkSyntax(t, "not rdx", "NOTQ", DX)
|
||||
}
|
||||
|
||||
func TestShift(t *testing.T) {
|
||||
checkSyntax(t, "shl rdx, 0x2", "SHLQ", Imm(2), DX)
|
||||
checkSyntax(t, "shl rdx, cl", "SHLQ", CL, DX)
|
||||
checkSyntax(t, "shl rdx, 0x1", "SHLQ", Imm(1), DX)
|
||||
checkSyntax(t, "sar rcx, 0x1f", "SARQ", Imm(31), CX)
|
||||
}
|
||||
|
||||
func TestImul(t *testing.T) {
|
||||
checkSyntax(t, "imul rdx, rcx", "IMULQ", CX, DX)
|
||||
checkSyntax(t, "imul edx, edx, 0x3", "IMULL", Imm(3), DX, DX)
|
||||
checkSyntax(t, "imul rdx, rcx, 0x100", "IMULQ", Imm(256), CX, DX)
|
||||
}
|
||||
|
||||
func TestControl(t *testing.T) {
|
||||
checkSyntax(t, "ret", "RET")
|
||||
checkSyntax(t, "nop", "NOP")
|
||||
checkOp(t, x86asm.JMP, "JMP", Imm(0))
|
||||
checkOp(t, x86asm.CALL, "CALL", Imm(0))
|
||||
checkOp(t, x86asm.JGE, "JGE", Imm(0))
|
||||
checkOp(t, x86asm.JNE, "JNE", Imm(0))
|
||||
checkOp(t, x86asm.JBE, "JLS", Imm(0))
|
||||
}
|
||||
|
||||
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
|
||||
// the encoder handles a realistic instruction sequence.
|
||||
func TestGoFlacScalarTail(t *testing.T) {
|
||||
// MOVQ swin_base+0(FP), SI — modelled as MOVQ disp(reg), reg.
|
||||
checkSyntax(t, "mov rsi, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), SI)
|
||||
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
|
||||
checkSyntax(t, "and r10, -0x8", "ANDQ", Imm(-8), Reg{idx: 10, size: 8})
|
||||
}
|
||||
+488
@@ -0,0 +1,488 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import "fmt"
|
||||
|
||||
// aluOp maps an arithmetic/logic mnemonic to its base "r/m, r" opcode (for
|
||||
// 16/32/64-bit; the 8-bit form is one less) and its /digit for the immediate
|
||||
// forms (0x80/0x81/0x83).
|
||||
var aluOp = map[string]struct {
|
||||
rr byte
|
||||
digit int
|
||||
}{
|
||||
"ADD": {0x01, 0},
|
||||
"OR": {0x09, 1},
|
||||
"AND": {0x21, 4},
|
||||
"SUB": {0x29, 5},
|
||||
"XOR": {0x31, 6},
|
||||
"CMP": {0x39, 7},
|
||||
}
|
||||
|
||||
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
|
||||
// the 0xFE/0xFF group (the short 0x40–0x4F forms are REX prefixes in 64-bit
|
||||
// mode); NEG/NOT use the 0xF6/0xF7 group.
|
||||
var unaryOp = map[string]struct {
|
||||
digit int
|
||||
op byte
|
||||
}{
|
||||
"INC": {0, 0xFF},
|
||||
"DEC": {1, 0xFF},
|
||||
"NOT": {2, 0xF7},
|
||||
"NEG": {3, 0xF7},
|
||||
}
|
||||
|
||||
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0–0xD3 group.
|
||||
var shiftOp = map[string]int{
|
||||
"SHL": 4,
|
||||
"SHR": 5,
|
||||
"SAR": 7,
|
||||
}
|
||||
|
||||
// --- MOV --------------------------------------------------------------------
|
||||
|
||||
func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("MOV expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
|
||||
dstReg, dstIsReg := dst.(Reg)
|
||||
switch src := src.(type) {
|
||||
case Reg:
|
||||
if dstIsReg {
|
||||
// MOV r, r/m: 0x8A/0x8B, reg=dst, rm=src.
|
||||
i := newInstr(size, []byte{movRR(size)})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst(mem).
|
||||
i := newInstr(size, []byte{movRM(size)})
|
||||
if err := setRM(i, src, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
|
||||
case Mem:
|
||||
if !dstIsReg {
|
||||
return fmt.Errorf("MOV: two memory operands")
|
||||
}
|
||||
// MOV r, r/m: reg=dst, rm=src(mem).
|
||||
i := newInstr(size, []byte{movRR(size)})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
|
||||
case Imm:
|
||||
if dstIsReg {
|
||||
// MOV r, imm: 0xB0+reg (8-bit) / 0xB8+reg (16/32/64, imm64 for Q).
|
||||
opBase := byte(0xB8)
|
||||
if size == 1 {
|
||||
opBase = 0xB0
|
||||
}
|
||||
i := newInstr(size, []byte{opBase + byte(dstReg.idx&7)})
|
||||
i.rexB = dstReg.idx >= 8
|
||||
if dstReg.needsREX(size) {
|
||||
i.rexForced = true
|
||||
}
|
||||
i.imm = immediate(int64(src), size, true)
|
||||
return e.emit(i)
|
||||
}
|
||||
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
|
||||
op := byte(0xC7)
|
||||
if size == 1 {
|
||||
op = 0xC6
|
||||
}
|
||||
i := newInstr(size, []byte{op})
|
||||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immediate(int64(src), size, false)
|
||||
return e.emit(i)
|
||||
}
|
||||
return fmt.Errorf("MOV: invalid operands")
|
||||
}
|
||||
|
||||
func movRR(size int) byte { // MOV r, r/m
|
||||
if size == 1 {
|
||||
return 0x8A
|
||||
}
|
||||
return 0x8B
|
||||
}
|
||||
|
||||
func movRM(size int) byte { // MOV r/m, r
|
||||
if size == 1 {
|
||||
return 0x88
|
||||
}
|
||||
return 0x89
|
||||
}
|
||||
|
||||
// --- ALU (ADD/OR/AND/SUB/XOR/CMP) -------------------------------------------
|
||||
|
||||
func (e *enc) encodeALU(op struct {
|
||||
rr byte
|
||||
digit int
|
||||
}, ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("ALU instruction expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
|
||||
if imm, ok := src.(Imm); ok {
|
||||
return e.encodeALUImm(op.digit, dst, int64(imm), size)
|
||||
}
|
||||
|
||||
dstReg, dstIsReg := dst.(Reg)
|
||||
srcReg, srcIsReg := src.(Reg)
|
||||
switch {
|
||||
case srcIsReg:
|
||||
// OP r/m, r: reg=src, rm=dst (dst is a register or memory). This is the
|
||||
// form the Go assembler prefers when the source is a register.
|
||||
opc := op.rr
|
||||
if size == 1 {
|
||||
opc = op.rr - 1
|
||||
}
|
||||
i := newInstr(size, []byte{opc})
|
||||
if err := setRM(i, srcReg, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case dstIsReg:
|
||||
// OP r, r/m: reg=dst, rm=src(memory).
|
||||
opc := op.rr + 2
|
||||
if size == 1 {
|
||||
opc = op.rr + 1
|
||||
}
|
||||
i := newInstr(size, []byte{opc})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
return fmt.Errorf("two memory operands")
|
||||
}
|
||||
|
||||
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
||||
if size == 1 {
|
||||
i := newInstr(1, []byte{0x80})
|
||||
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{byte(int8(imm))}
|
||||
return e.emit(i)
|
||||
}
|
||||
if fits8(imm) {
|
||||
// 0x83 /digit, sign-extended imm8.
|
||||
i := newInstr(size, []byte{0x83})
|
||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{byte(int8(imm))}
|
||||
return e.emit(i)
|
||||
}
|
||||
// 0x81 /digit, imm16/imm32.
|
||||
i := newInstr(size, []byte{0x81})
|
||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immediate(imm, size, false)
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- TEST -------------------------------------------------------------------
|
||||
|
||||
func (e *enc) encodeTest(ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("TEST expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
if imm, ok := src.(Imm); ok {
|
||||
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0.
|
||||
op := byte(0xF7)
|
||||
if size == 1 {
|
||||
op = 0xF6
|
||||
}
|
||||
i := newInstr(size, []byte{op})
|
||||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immediate(int64(imm), size, false)
|
||||
return e.emit(i)
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("TEST: source must be a register or immediate")
|
||||
}
|
||||
// TEST r/m, r: 0x84 (8-bit) / 0x85.
|
||||
op := byte(0x85)
|
||||
if size == 1 {
|
||||
op = 0x84
|
||||
}
|
||||
i := newInstr(size, []byte{op})
|
||||
if err := setRM(i, srcReg, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- LEA --------------------------------------------------------------------
|
||||
|
||||
func (e *enc) encodeLea(ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("LEA expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1] // LEAQ addr, reg
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("LEA: destination must be a register")
|
||||
}
|
||||
mem, ok := src.(Mem)
|
||||
if !ok {
|
||||
return fmt.Errorf("LEA: source must be a memory operand")
|
||||
}
|
||||
i := newInstr(size, []byte{0x8D})
|
||||
if err := setRM(i, dstReg, mem, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- INC/DEC/NEG/NOT --------------------------------------------------------
|
||||
|
||||
func (e *enc) encodeUnary(op struct {
|
||||
digit int
|
||||
op byte
|
||||
}, ops []Operand, size int) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("unary instruction expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
base := op.op
|
||||
if size == 1 {
|
||||
base-- // 0xFF→0xFE, 0xF7→0xF6
|
||||
}
|
||||
i := newInstr(size, []byte{base})
|
||||
if err := setRMDigit(i, op.digit, ops[0], size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- SHL/SHR/SAR ------------------------------------------------------------
|
||||
|
||||
func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("shift expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
count, dst := ops[0], ops[1]
|
||||
// Count is $1, %CL, or an imm8.
|
||||
if reg, ok := count.(Reg); ok && reg.idx == 1 && reg.size <= 1 {
|
||||
// CL: 0xD2 (8-bit) / 0xD3.
|
||||
op := byte(0xD3)
|
||||
if size == 1 {
|
||||
op = 0xD2
|
||||
}
|
||||
i := newInstr(size, []byte{op})
|
||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
imm, ok := count.(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("shift count must be $1, CL or an immediate")
|
||||
}
|
||||
if imm == 1 {
|
||||
// 0xD0 (8-bit) / 0xD1.
|
||||
op := byte(0xD1)
|
||||
if size == 1 {
|
||||
op = 0xD0
|
||||
}
|
||||
i := newInstr(size, []byte{op})
|
||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
// 0xC0 (8-bit) / 0xC1, imm8.
|
||||
op := byte(0xC1)
|
||||
if size == 1 {
|
||||
op = 0xC0
|
||||
}
|
||||
i := newInstr(size, []byte{op})
|
||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{byte(int8(imm))}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- IMUL -------------------------------------------------------------------
|
||||
|
||||
func (e *enc) encodeImul(ops []Operand, size int) error {
|
||||
switch len(ops) {
|
||||
case 2:
|
||||
// IMUL r, r/m: 0x0F 0xAF.
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("IMUL: destination must be a register")
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, 0xAF})
|
||||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case 3:
|
||||
// IMUL r, r/m, imm: 0x6B (imm8) / 0x69 (imm16/32).
|
||||
dstReg, ok := ops[2].(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("IMUL: destination must be a register")
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("IMUL: immediate operand expected first")
|
||||
}
|
||||
// Plan 9 order: IMUL $imm, src, dst.
|
||||
if fits8(int64(imm)) {
|
||||
i := newInstr(size, []byte{0x6B})
|
||||
if err := setRM(i, dstReg, ops[1], size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{byte(int8(imm))}
|
||||
return e.emit(i)
|
||||
}
|
||||
i := newInstr(size, []byte{0x69})
|
||||
if err := setRM(i, dstReg, ops[1], size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immediate(int64(imm), size, false)
|
||||
return e.emit(i)
|
||||
}
|
||||
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
|
||||
}
|
||||
|
||||
// --- PUSH / POP -------------------------------------------------------------
|
||||
|
||||
func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
switch op := ops[0].(type) {
|
||||
case Reg:
|
||||
base := byte(0x50) // PUSH r; POP is 0x58
|
||||
if !push {
|
||||
base = 0x58
|
||||
}
|
||||
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
|
||||
i := &instr{opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
|
||||
i.rexB = op.idx >= 8
|
||||
return e.emit(i)
|
||||
case Mem:
|
||||
opc := byte(0xFF) // PUSH r/m: /6
|
||||
digit := 6
|
||||
if !push {
|
||||
opc = 0x8F // POP r/m: /0
|
||||
digit = 0
|
||||
}
|
||||
i := &instr{opcode: []byte{opc}, modrm: -1, sib: -1}
|
||||
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case Imm:
|
||||
if !push {
|
||||
return fmt.Errorf("POP does not take an immediate")
|
||||
}
|
||||
if fits8(int64(op)) {
|
||||
i := &instr{opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
|
||||
return e.emit(i)
|
||||
}
|
||||
i := &instr{opSize16: false, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: le32(int64(op))}
|
||||
return e.emit(i)
|
||||
}
|
||||
return fmt.Errorf("PUSH/POP: invalid operand")
|
||||
}
|
||||
|
||||
// --- RET / JMP / CALL / Jcc -------------------------------------------------
|
||||
|
||||
func (e *enc) encodeRet() error {
|
||||
return e.emit(&instr{opcode: []byte{0xC3}, modrm: -1, sib: -1})
|
||||
}
|
||||
|
||||
// encodeJmpRel encodes JMP/CALL with a relative displacement (the operand is an
|
||||
// Imm holding the already-computed rel32 offset).
|
||||
func (e *enc) encodeJmpRel(ops []Operand, opcode []byte) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("JMP/CALL expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("JMP/CALL: relative offset must be an immediate (labels are resolved by the assembler)")
|
||||
}
|
||||
return e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, imm: le32(int64(imm))})
|
||||
}
|
||||
|
||||
// condCode maps a Plan 9 conditional-jump mnemonic to its x86 condition code.
|
||||
func condCode(upper string) (int, bool) {
|
||||
if len(upper) < 2 || upper[0] != 'J' || upper == "JMP" {
|
||||
return 0, false
|
||||
}
|
||||
cc, ok := jccMap[upper[1:]]
|
||||
return cc, ok
|
||||
}
|
||||
|
||||
var jccMap = map[string]int{
|
||||
"O": 0x0, "NO": 0x1, "OS": 0x0, "OC": 0x1,
|
||||
"B": 0x2, "C": 0x2, "NAE": 0x2, "CS": 0x2,
|
||||
"NB": 0x3, "NC": 0x3, "AE": 0x3, "CC": 0x3,
|
||||
"E": 0x4, "Z": 0x4, "EQ": 0x4,
|
||||
"NE": 0x5, "NZ": 0x5,
|
||||
"BE": 0x6, "NA": 0x6, "LS": 0x6,
|
||||
"NBE": 0x7, "A": 0x7, "HI": 0x7,
|
||||
"S": 0x8, "MI": 0x8,
|
||||
"NS": 0x9, "PL": 0x9,
|
||||
"P": 0xA, "PE": 0xA, "PS": 0xA,
|
||||
"NP": 0xB, "PO": 0xB, "PC": 0xB,
|
||||
"L": 0xC, "NGE": 0xC, "LT": 0xC,
|
||||
"NL": 0xD, "GE": 0xD,
|
||||
"LE": 0xE, "NG": 0xE,
|
||||
"NLE": 0xF, "G": 0xF, "GT": 0xF,
|
||||
}
|
||||
|
||||
func (e *enc) encodeJcc(cc int, ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("conditional jump expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("conditional jump: relative offset must be an immediate")
|
||||
}
|
||||
if fits8(int64(imm)) {
|
||||
// Short form: 0x70+cc, rel8.
|
||||
return e.emit(&instr{opcode: []byte{0x70 + byte(cc)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}})
|
||||
}
|
||||
// Near form: 0x0F 0x80+cc, rel32.
|
||||
return e.emit(&instr{opcode: []byte{0x0F, 0x80 + byte(cc)}, modrm: -1, sib: -1, imm: le32(int64(imm))})
|
||||
}
|
||||
|
||||
// immediate encodes an immediate of the given operand size. full64 selects the
|
||||
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
|
||||
// sign-extended immediate is used for 64-bit operands.
|
||||
func immediate(v int64, size int, full64 bool) []byte {
|
||||
switch size {
|
||||
case 1:
|
||||
return []byte{byte(int8(v))}
|
||||
case 2:
|
||||
return le16(v)
|
||||
case 4:
|
||||
return le32(v)
|
||||
default: // 8
|
||||
if full64 {
|
||||
return le64(v)
|
||||
}
|
||||
return le32(v) // sign-extended imm32
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
// Operand is an instruction operand: a Reg, a Mem reference or an Imm value.
|
||||
type Operand interface {
|
||||
isOperand()
|
||||
}
|
||||
|
||||
// Imm is an immediate value. Its encoded width is chosen by the instruction
|
||||
// (sign-extended imm8 where possible, otherwise imm32, imm64 for MOV).
|
||||
type Imm int64
|
||||
|
||||
func (Imm) isOperand() {}
|
||||
|
||||
// Mem is a memory operand of the form disp(base)(index*scale).
|
||||
type Mem struct {
|
||||
Base Reg
|
||||
Index Reg
|
||||
Scale int // 1, 2, 4 or 8; 0 means no index
|
||||
Disp int64
|
||||
Size int // operand width in bytes
|
||||
HasBase bool
|
||||
HasIndex bool
|
||||
}
|
||||
|
||||
func (Mem) isOperand() {}
|
||||
|
||||
// Ptr builds a plain displaced memory operand (base)+disp of the given size.
|
||||
func Ptr(base Reg, disp int64, size int) Mem {
|
||||
return Mem{Base: base, Disp: disp, Size: size, HasBase: true}
|
||||
}
|
||||
|
||||
// Idx builds an indexed memory operand disp(base)(index*scale).
|
||||
func Idx(base, index Reg, scale int, disp int64, size int) Mem {
|
||||
return Mem{Base: base, Index: index, Scale: scale, Disp: disp, Size: size, HasBase: true, HasIndex: true}
|
||||
}
|
||||
|
||||
// Rip builds a RIP-relative memory operand (RIP)+disp.
|
||||
func Rip(disp int64, size int) Mem {
|
||||
return Mem{Disp: disp, Size: size}
|
||||
}
|
||||
+169
@@ -0,0 +1,169 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package asm is a standalone assembler: it encodes Plan 9 assembly
|
||||
// instructions into machine code without the Go toolchain. Phase 2 begins
|
||||
// with an amd64 (x86-64) scalar instruction encoder; the encoding is validated
|
||||
// by round-tripping through golang.org/x/arch's decoder in the tests.
|
||||
package asm
|
||||
|
||||
import "strings"
|
||||
|
||||
// Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …)
|
||||
// are size-agnostic — the instruction suffix (MOVQ vs MOVL) fixes the width —
|
||||
// so the encoder keys off the register's index and lets the mnemonic supply the
|
||||
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
||||
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
||||
// those indices but require one.
|
||||
type Reg struct {
|
||||
idx int
|
||||
size int // informational width implied by the name; the mnemonic decides
|
||||
high bool // AH/CH/DH/BH
|
||||
}
|
||||
|
||||
// Index returns the register number (0–15).
|
||||
func (r Reg) Index() int { return r.idx }
|
||||
|
||||
// Size returns the width in bytes implied by the register's name.
|
||||
func (r Reg) Size() int { return r.size }
|
||||
|
||||
func (r Reg) isOperand() {}
|
||||
|
||||
// needsREX reports whether this register forces a REX prefix at the given
|
||||
// operand size: the extended registers R8–R15 always do, and at byte size the
|
||||
// low registers SPL/BPL/SIL/DIL (indices 4–7, not high) do as well.
|
||||
func (r Reg) needsREX(opSize int) bool {
|
||||
if r.idx >= 8 {
|
||||
return true
|
||||
}
|
||||
return opSize == 1 && r.idx >= 4 && !r.high
|
||||
}
|
||||
|
||||
// Register constants (the size is the width the name implies).
|
||||
var (
|
||||
AL = Reg{0, 1, false}
|
||||
CL = Reg{1, 1, false}
|
||||
DL = Reg{2, 1, false}
|
||||
BL = Reg{3, 1, false}
|
||||
AH = Reg{4, 1, true}
|
||||
CH = Reg{5, 1, true}
|
||||
DH = Reg{6, 1, true}
|
||||
BH = Reg{7, 1, true}
|
||||
SPL = Reg{4, 1, false}
|
||||
BPL = Reg{5, 1, false}
|
||||
SIL = Reg{6, 1, false}
|
||||
DIL = Reg{7, 1, false}
|
||||
|
||||
AX = Reg{0, 2, false}
|
||||
CX = Reg{1, 2, false}
|
||||
DX = Reg{2, 2, false}
|
||||
BX = Reg{3, 2, false}
|
||||
SP = Reg{4, 2, false}
|
||||
BP = Reg{5, 2, false}
|
||||
SI = Reg{6, 2, false}
|
||||
DI = Reg{7, 2, false}
|
||||
|
||||
EAX = Reg{0, 4, false}
|
||||
ECX = Reg{1, 4, false}
|
||||
EDX = Reg{2, 4, false}
|
||||
EBX = Reg{3, 4, false}
|
||||
ESP = Reg{4, 4, false}
|
||||
EBP = Reg{5, 4, false}
|
||||
ESI = Reg{6, 4, false}
|
||||
EDI = Reg{7, 4, false}
|
||||
|
||||
RAX = Reg{0, 8, false}
|
||||
RCX = Reg{1, 8, false}
|
||||
RDX = Reg{2, 8, false}
|
||||
RBX = Reg{3, 8, false}
|
||||
RSP = Reg{4, 8, false}
|
||||
RBP = Reg{5, 8, false}
|
||||
RSI = Reg{6, 8, false}
|
||||
RDI = Reg{7, 8, false}
|
||||
)
|
||||
|
||||
// regByName maps an assembly register name (case-insensitive) to a Reg.
|
||||
var regByName = buildRegByName()
|
||||
|
||||
func buildRegByName() map[string]Reg {
|
||||
m := map[string]Reg{}
|
||||
|
||||
// 64-bit: RAX..RDI, R8..R15.
|
||||
r64 := []string{"RAX", "RCX", "RDX", "RBX", "RSP", "RBP", "RSI", "RDI"}
|
||||
for i, n := range r64 {
|
||||
m[n] = Reg{i, 8, false}
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)] = Reg{i, 8, false}
|
||||
}
|
||||
|
||||
// 32-bit: EAX..EDI, R8D..R15D.
|
||||
e32 := []string{"EAX", "ECX", "EDX", "EBX", "ESP", "EBP", "ESI", "EDI"}
|
||||
for i, n := range e32 {
|
||||
m[n] = Reg{i, 4, false}
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)+"D"] = Reg{i, 4, false}
|
||||
}
|
||||
|
||||
// 16-bit: AX..DI, R8W..R15W.
|
||||
w16 := []string{"AX", "CX", "DX", "BX", "SP", "BP", "SI", "DI"}
|
||||
for i, n := range w16 {
|
||||
m[n] = Reg{i, 2, false}
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)+"W"] = Reg{i, 2, false}
|
||||
}
|
||||
|
||||
// 8-bit: AL..BH, SPL..DIL, R8B..R15B.
|
||||
for n, r := range map[string]Reg{
|
||||
"AL": AL, "CL": CL, "DL": DL, "BL": BL,
|
||||
"AH": AH, "CH": CH, "DH": DH, "BH": BH,
|
||||
"SPL": SPL, "BPL": BPL, "SIL": SIL, "DIL": DIL,
|
||||
} {
|
||||
m[n] = r
|
||||
}
|
||||
for i := 8; i <= 15; i++ {
|
||||
m["R"+itoa(i)+"B"] = Reg{i, 1, false}
|
||||
}
|
||||
|
||||
// Vector: X0..X15 (128-bit, encoded size 16), Y0..Y15 (256-bit, size 32).
|
||||
// Z (512-bit) and K (mask) registers arrive with EVEX/AVX-512 support.
|
||||
for i := 0; i <= 15; i++ {
|
||||
m["X"+itoa(i)] = Reg{i, 16, false}
|
||||
m["Y"+itoa(i)] = Reg{i, 32, false}
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// isVec reports whether r is an XMM/YMM vector register.
|
||||
func (r Reg) isVec() bool { return r.size == 16 || r.size == 32 }
|
||||
|
||||
// vecLenBit returns the VEX.L bit for a vector register (X=0/128-bit,
|
||||
// Y=1/256-bit).
|
||||
func (r Reg) vecLenBit() int {
|
||||
if r.size == 32 {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// ParseReg resolves an assembly register name to a Reg.
|
||||
func ParseReg(name string) (Reg, bool) {
|
||||
r, ok := regByName[strings.ToUpper(name)]
|
||||
return r, ok
|
||||
}
|
||||
|
||||
func itoa(n int) string {
|
||||
if n == 0 {
|
||||
return "0"
|
||||
}
|
||||
var buf [3]byte
|
||||
i := len(buf)
|
||||
for n > 0 {
|
||||
i--
|
||||
buf[i] = byte('0' + n%10)
|
||||
n /= 10
|
||||
}
|
||||
return string(buf[i:])
|
||||
}
|
||||
+234
@@ -0,0 +1,234 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import "fmt"
|
||||
|
||||
// This file implements VEX (AVX/AVX2) instruction encoding. EVEX (AVX-512)
|
||||
// support is a later increment.
|
||||
|
||||
// vexForm selects how an instruction's operands map onto the VEX.vvvv,
|
||||
// ModRM.reg and ModRM.rm fields.
|
||||
type vexForm int
|
||||
|
||||
const (
|
||||
// vexNDS3 is the three-operand form `OP src2, src1, dst` (Plan 9 order):
|
||||
// ModRM.reg = dst (op2), VEX.vvvv = src1 (op1), ModRM.rm = src2 (op0).
|
||||
vexNDS3 vexForm = iota
|
||||
// vexRM is the two-operand form `OP src, dst` with no vvvv source:
|
||||
// ModRM.reg = dst (op1), ModRM.rm = src (op0), VEX.vvvv = 1111 (unused).
|
||||
vexRM
|
||||
// vexShiftImm is the immediate-shift form `OP $imm, src, dst`: ModRM.reg =
|
||||
// /digit, ModRM.rm = src (op1), VEX.vvvv = dst (op2), imm8 = op0.
|
||||
vexShiftImm
|
||||
)
|
||||
|
||||
// vexSpec describes one VEX instruction's encoding parameters.
|
||||
type vexSpec struct {
|
||||
mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A
|
||||
opcode byte
|
||||
w int // VEX.W (0 for WIG)
|
||||
pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2
|
||||
opdigit int // ModRM.reg /digit, or -1 when reg is a register
|
||||
form vexForm
|
||||
}
|
||||
|
||||
// vexTable maps an upper-case mnemonic to its VEX encoding. It covers the
|
||||
// AVX2 instructions used by the go-flac kernels in the three-operand NDS form;
|
||||
// it is extended incrementally.
|
||||
var vexTable = map[string]vexSpec{
|
||||
// VEX.128/256.66.0F.WIG — integer arithmetic / logic / compare.
|
||||
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
|
||||
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
|
||||
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
|
||||
"VPSUBQ": {1, 0xFB, 0, 1, -1, vexNDS3},
|
||||
"VPXOR": {1, 0xEF, 0, 1, -1, vexNDS3},
|
||||
"VPOR": {1, 0xEB, 0, 1, -1, vexNDS3},
|
||||
"VPAND": {1, 0xDB, 0, 1, -1, vexNDS3},
|
||||
"VPANDN": {1, 0xDF, 0, 1, -1, vexNDS3},
|
||||
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
|
||||
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
|
||||
// VEX.128/256.66.0F38.WIG.
|
||||
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
|
||||
"VPMULDQ": {2, 0x28, 0, 1, -1, vexNDS3},
|
||||
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F38.WIG — sign/zero extend and broadcast (reg=dst, rm=src,
|
||||
// no vvvv).
|
||||
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
||||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
||||
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
||||
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
||||
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
|
||||
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
|
||||
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
|
||||
|
||||
// VEX.128/256.66.0F.WIG — immediate shifts (opdigit selects the shift).
|
||||
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
|
||||
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
|
||||
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
|
||||
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
|
||||
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
|
||||
}
|
||||
|
||||
// isVex reports whether the mnemonic is a VEX-encoded instruction we handle.
|
||||
func isVex(mnemUpper string) bool {
|
||||
_, ok := vexTable[mnemUpper]
|
||||
return ok
|
||||
}
|
||||
|
||||
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
|
||||
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
spec := vexTable[mnemUpper]
|
||||
switch spec.form {
|
||||
case vexNDS3:
|
||||
return e.encodeVexNDS3(spec, ops)
|
||||
case vexRM:
|
||||
return e.encodeVexRM(spec, ops)
|
||||
case vexShiftImm:
|
||||
return e.encodeVexShiftImm(spec, ops)
|
||||
}
|
||||
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
||||
}
|
||||
|
||||
// encodeVexNDS3 encodes the three-operand NDS form: OP src2, src1, dst.
|
||||
func (e *enc) encodeVexNDS3(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
||||
}
|
||||
src2, src1, dst := ops[0], ops[1], ops[2]
|
||||
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a vector register")
|
||||
}
|
||||
vvvvReg, ok := src1.(Reg)
|
||||
if !ok || !vvvvReg.isVec() {
|
||||
return fmt.Errorf("VEX vvvv operand must be a vector register")
|
||||
}
|
||||
|
||||
regField := dstReg.idx & 7
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
vvvvBar := 15 - (vvvvReg.idx & 15)
|
||||
return e.emitVexFields(spec, dstReg.vecLenBit(), regField, rBit, vvvvBar, src2)
|
||||
}
|
||||
|
||||
// encodeVexRM encodes the two-operand form: OP src, dst (no vvvv source).
|
||||
// ModRM.reg = dst, ModRM.rm = src; the vector length comes from whichever
|
||||
// operand is a vector register (the destination for extends/broadcasts, the
|
||||
// source for the move-mask instructions whose destination is a GPR).
|
||||
func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("VEX two-operand instruction expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("VEX destination must be a register")
|
||||
}
|
||||
regField := dstReg.idx & 7
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
|
||||
// Vector length: from the destination if it is a vector, otherwise from the
|
||||
// source (move-mask instructions have a GPR destination and a vector source).
|
||||
l := 0
|
||||
if dstReg.isVec() {
|
||||
l = dstReg.vecLenBit()
|
||||
} else if srcReg, ok := src.(Reg); ok && srcReg.isVec() {
|
||||
l = srcReg.vecLenBit()
|
||||
}
|
||||
|
||||
return e.emitVexFields(spec, l, regField, rBit, 0, src) // vvvv unused → vvvvBar=0
|
||||
}
|
||||
|
||||
// encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst.
|
||||
// The destination is carried in VEX.vvvv, the source in ModRM.rm, and the
|
||||
// shift kind in the ModRM.reg /digit.
|
||||
func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("VEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||
}
|
||||
imm, src, dst := ops[0], ops[1], ops[2]
|
||||
immVal, ok := imm.(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("shift count must be an immediate")
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok || !srcReg.isVec() {
|
||||
return fmt.Errorf("shift source must be a vector register")
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("shift destination must be a vector register")
|
||||
}
|
||||
|
||||
vvvvBar := 15 - (dstReg.idx & 15)
|
||||
l := dstReg.vecLenBit()
|
||||
rmField := srcReg.idx & 7
|
||||
bBit := 0
|
||||
if srcReg.idx >= 8 {
|
||||
bBit = 1
|
||||
}
|
||||
modrm := 0xC0 | spec.opdigit<<3 | rmField
|
||||
|
||||
if spec.mapSel == 1 && bBit == 0 && spec.w == 0 {
|
||||
e.out = append(e.out, 0xC5, byte(1<<7|vvvvBar<<3|l<<2|spec.pp))
|
||||
} else {
|
||||
e.out = append(e.out, 0xC4,
|
||||
byte(1<<7|1<<6|(1-bBit)<<5|spec.mapSel),
|
||||
byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp))
|
||||
}
|
||||
e.out = append(e.out, spec.opcode, byte(modrm), byte(int8(immVal)))
|
||||
return nil
|
||||
}
|
||||
|
||||
// emitVexFields emits the VEX prefix, opcode, ModR/M, SIB and displacement for
|
||||
// the given precomputed fields. It is shared by the NDS and RM forms.
|
||||
func (e *enc) emitVexFields(spec vexSpec, l, regField, rBit, vvvvBar int, rm Operand) error {
|
||||
var modrm, sib int
|
||||
var disp []byte
|
||||
var xBit, bBit int
|
||||
switch r := rm.(type) {
|
||||
case Reg:
|
||||
modrm = 0xC0 | regField<<3 | (r.idx & 7)
|
||||
sib = -1
|
||||
if r.idx >= 8 {
|
||||
bBit = 1
|
||||
}
|
||||
case Mem:
|
||||
var err error
|
||||
modrm, sib, disp, xBit, bBit, err = memComponents(regField, r)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
default:
|
||||
return fmt.Errorf("invalid VEX r/m operand")
|
||||
}
|
||||
|
||||
if spec.mapSel == 1 && xBit == 0 && bBit == 0 && spec.w == 0 {
|
||||
e.out = append(e.out, 0xC5, byte((1-rBit)<<7|vvvvBar<<3|l<<2|spec.pp))
|
||||
} else {
|
||||
e.out = append(e.out, 0xC4,
|
||||
byte((1-rBit)<<7|(1-xBit)<<6|(1-bBit)<<5|spec.mapSel),
|
||||
byte(spec.w<<7|vvvvBar<<3|l<<2|spec.pp))
|
||||
}
|
||||
e.out = append(e.out, spec.opcode, byte(modrm))
|
||||
if sib >= 0 {
|
||||
e.out = append(e.out, byte(sib))
|
||||
}
|
||||
e.out = append(e.out, disp...)
|
||||
return nil
|
||||
}
|
||||
+142
@@ -0,0 +1,142 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"golang.org/x/arch/x86/x86asm"
|
||||
)
|
||||
|
||||
func vreg(t *testing.T, name string) Reg {
|
||||
t.Helper()
|
||||
r, ok := ParseReg(name)
|
||||
if !ok {
|
||||
t.Fatalf("unknown register %s", name)
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
||||
// instruction and verifies it round-trips through the x86 decoder to the same
|
||||
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
|
||||
func TestVexNDS3(t *testing.T) {
|
||||
for mnem, spec := range vexTable {
|
||||
if spec.form != vexNDS3 {
|
||||
continue
|
||||
}
|
||||
code, err := Encode(mnem, vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2"))
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", mnem, err)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(% x): %v", mnem, code, err)
|
||||
continue
|
||||
}
|
||||
if inst.Op.String() != mnem {
|
||||
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestVexGoFlac checks a representative go-flac instruction sequence encodes
|
||||
// and decodes as expected.
|
||||
func TestVexGoFlac(t *testing.T) {
|
||||
// VPADDD Y5, Y8, Y8 → vpaddd ymm8, ymm8, ymm5.
|
||||
code, err := Encode("VPADDD", vreg(t, "Y5"), vreg(t, "Y8"), vreg(t, "Y8"))
|
||||
if err != nil {
|
||||
t.Fatalf("Encode: %v", err)
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(% x): %v", code, err)
|
||||
}
|
||||
if inst.Op != x86asm.VPADDD {
|
||||
t.Fatalf("decoded %s, want VPADDD", inst.Op)
|
||||
}
|
||||
}
|
||||
|
||||
// TestVexXMM checks the 128-bit (XMM) form selects VEX.L=0.
|
||||
func TestVexXMM(t *testing.T) {
|
||||
code, err := Encode("VPXOR", vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7"))
|
||||
if err != nil {
|
||||
t.Fatalf("Encode: %v", err)
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(% x): %v", code, err)
|
||||
}
|
||||
if inst.Op != x86asm.VPXOR {
|
||||
t.Fatalf("decoded %s, want VPXOR", inst.Op)
|
||||
}
|
||||
// vpxor xmm7, xmm7, xmm7 → C5 C9 EF FF (2-byte VEX, L=0).
|
||||
if code[0] != 0xC5 {
|
||||
t.Errorf("expected 2-byte VEX (C5), got % x", code)
|
||||
}
|
||||
}
|
||||
|
||||
// TestVexRM validates the two-operand (reg=dst, rm=src, no vvvv) forms by
|
||||
// round-tripping through the decoder.
|
||||
func TestVexRM(t *testing.T) {
|
||||
cases := []struct {
|
||||
mnem string
|
||||
ops []Operand
|
||||
want x86asm.Op
|
||||
}{
|
||||
{"VPMOVSXWD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y0")}, x86asm.VPMOVSXWD},
|
||||
{"VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, x86asm.VPMOVSXDQ},
|
||||
{"VPMOVZXDQ", []Operand{vreg(t, "X4"), vreg(t, "Y4")}, x86asm.VPMOVZXDQ},
|
||||
{"VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, x86asm.VPBROADCASTD},
|
||||
{"VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, x86asm.VPMOVMSKB},
|
||||
{"VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, x86asm.VMOVMSKPS},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.mnem, err)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(% x): %v", c.mnem, code, err)
|
||||
continue
|
||||
}
|
||||
if inst.Op != c.want {
|
||||
t.Errorf("%s: decoded as %s (% x)", c.mnem, inst.Op, code)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestVexShiftImm validates the immediate-shift form, checking the destination
|
||||
// (VEX.vvvv) and source (ModRM.rm) land in the right places.
|
||||
func TestVexShiftImm(t *testing.T) {
|
||||
// VPSLLD $1, Y3, Y4 → vpslld ymm4, ymm3, 1.
|
||||
code, err := Encode("VPSLLD", Imm(1), vreg(t, "Y3"), vreg(t, "Y4"))
|
||||
if err != nil {
|
||||
t.Fatalf("Encode: %v", err)
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(% x): %v", code, err)
|
||||
}
|
||||
if inst.Op != x86asm.VPSLLD {
|
||||
t.Fatalf("decoded %s, want VPSLLD (% x)", inst.Op, code)
|
||||
}
|
||||
// Intel order: dst, src, imm → "vpslld ymm4, ymm3, 0x1".
|
||||
if got := x86asm.IntelSyntax(inst, 0, nil); got != "vpslld ymm4, ymm3, 0x1" {
|
||||
t.Errorf("VPSLLD syntax = %q, want \"vpslld ymm4, ymm3, 0x1\" (% x)", got, code)
|
||||
}
|
||||
|
||||
// VPSRAD $31, Y3, Y3 → vpsrad ymm3, ymm3, 31.
|
||||
code, err = Encode("VPSRAD", Imm(31), vreg(t, "Y3"), vreg(t, "Y3"))
|
||||
if err != nil {
|
||||
t.Fatalf("Encode VPSRAD: %v", err)
|
||||
}
|
||||
inst, err = x86asm.Decode(code, 64)
|
||||
if err != nil || inst.Op != x86asm.VPSRAD {
|
||||
t.Fatalf("VPSRAD decoded %v (err %v), want VPSRAD", inst.Op, err)
|
||||
}
|
||||
}
|
||||
+162
@@ -0,0 +1,162 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package ast defines the abstract syntax tree of a GAsm source file. The
|
||||
// tree is produced by the parser and consumed by the formatter, linter and
|
||||
// language server. Operand classification that depends on the target
|
||||
// architecture (is this bare name a register or a label?) is deliberately left
|
||||
// to the arch package; the AST records syntax only.
|
||||
package ast
|
||||
|
||||
import "sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
|
||||
// File is the parsed representation of one .s source file.
|
||||
type File struct {
|
||||
Path string
|
||||
Decls []Decl
|
||||
Orphans []Stmt // labels/instructions seen before any TEXT directive
|
||||
// Macros holds the names introduced by #define directives in this file.
|
||||
// The linter uses it to avoid flagging macro invocations as unknown
|
||||
// instructions (macro expansion itself is out of scope — see the docs).
|
||||
Macros map[string]bool
|
||||
}
|
||||
|
||||
// Decl is a top-level declaration.
|
||||
type Decl interface {
|
||||
declNode()
|
||||
// Pos returns the position of the declaration's first token.
|
||||
Pos() token.Position
|
||||
}
|
||||
|
||||
// Stmt is a statement inside a TEXT body.
|
||||
type Stmt interface {
|
||||
stmtNode()
|
||||
Pos() token.Position
|
||||
}
|
||||
|
||||
// Include is a #include "header" line.
|
||||
type Include struct {
|
||||
Hash token.Token
|
||||
Name token.Token // the directive name, usually "include"
|
||||
Header token.Token // the string literal, quotes included
|
||||
}
|
||||
|
||||
func (*Include) declNode() {}
|
||||
func (d *Include) Pos() token.Position { return d.Hash.Pos }
|
||||
|
||||
// Preproc is any other preprocessor line (#define, #undef, …) captured loosely.
|
||||
type Preproc struct {
|
||||
Hash token.Token
|
||||
Raw string // verbatim text after the '#'
|
||||
}
|
||||
|
||||
func (*Preproc) declNode() {}
|
||||
func (d *Preproc) Pos() token.Position { return d.Hash.Pos }
|
||||
|
||||
// Text is a TEXT function definition and its body.
|
||||
type Text struct {
|
||||
Keyword token.Token // the TEXT token
|
||||
Name *Symbol // ·funcName(SB)
|
||||
Flags []string // NOSPLIT, DUPOK, …
|
||||
Frame *Operand // $0
|
||||
Args *Operand // the -65 part; nil when absent
|
||||
Body []Stmt
|
||||
Doc string // preceding comment block (typically the Go signature)
|
||||
}
|
||||
|
||||
func (*Text) declNode() {}
|
||||
func (d *Text) Pos() token.Position { return d.Keyword.Pos }
|
||||
|
||||
// Globl is a GLOBL symbol declaration.
|
||||
type Globl struct {
|
||||
Keyword token.Token
|
||||
Name *Symbol
|
||||
Flags []string
|
||||
Size *Operand
|
||||
}
|
||||
|
||||
func (*Globl) declNode() {}
|
||||
func (d *Globl) Pos() token.Position { return d.Keyword.Pos }
|
||||
|
||||
// Data is a DATA symbol initialiser.
|
||||
type Data struct {
|
||||
Keyword token.Token
|
||||
Name *Symbol // includes any +offset
|
||||
Width int // the /width suffix; 0 when absent
|
||||
Value *Operand
|
||||
}
|
||||
|
||||
func (*Data) declNode() {}
|
||||
func (d *Data) Pos() token.Position { return d.Keyword.Pos }
|
||||
|
||||
// Label is a local label definition such as vec1:.
|
||||
type Label struct {
|
||||
Name token.Token
|
||||
Colon token.Token
|
||||
}
|
||||
|
||||
func (*Label) stmtNode() {}
|
||||
func (s *Label) Pos() token.Position { return s.Name.Pos }
|
||||
|
||||
// Instr is a single machine instruction with zero or more operands.
|
||||
type Instr struct {
|
||||
Mnemonic token.Token
|
||||
Operands []*Operand
|
||||
Comment string // trailing comment text, without the leading //
|
||||
}
|
||||
|
||||
func (*Instr) stmtNode() {}
|
||||
func (s *Instr) Pos() token.Position { return s.Mnemonic.Pos }
|
||||
|
||||
// Symbol is a symbol reference: ·name(SB), name<>(SB), name+8(FP), …
|
||||
type Symbol struct {
|
||||
Raw string // verbatim text
|
||||
Pkg string // package prefix before the middle dot ("" = current package)
|
||||
Name string // identifier without the middle dot or <>
|
||||
Static bool // the <> marker is present
|
||||
Pseudo string // FP, SP, SB or PC ("" for a bare name)
|
||||
Offset int64
|
||||
HasOff bool
|
||||
Pos token.Position
|
||||
}
|
||||
|
||||
// OpKind classifies an operand syntactically.
|
||||
type OpKind int
|
||||
|
||||
// Operand kinds.
|
||||
const (
|
||||
OpInvalid OpKind = iota
|
||||
OpImmediate // $value
|
||||
OpAddr // register, memory reference, symbol or label
|
||||
)
|
||||
|
||||
// Operand is one instruction operand.
|
||||
type Operand struct {
|
||||
Raw string
|
||||
Kind OpKind
|
||||
Imm Immediate
|
||||
Addr Address
|
||||
Pos token.Position
|
||||
}
|
||||
|
||||
// Immediate is a $ value.
|
||||
type Immediate struct {
|
||||
Neg bool
|
||||
Val int64
|
||||
HasVal bool // a simple integer immediate was parsed
|
||||
Float string // non-empty for a floating-point immediate
|
||||
Str string // non-empty for a string/rune immediate
|
||||
Sym *Symbol // non-nil for $sym(…)
|
||||
}
|
||||
|
||||
// Address is a non-immediate operand: a register, a memory reference, a symbol
|
||||
// reference or a label. Fields are populated best-effort from the syntax.
|
||||
type Address struct {
|
||||
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
|
||||
Base string // base register, from (base)
|
||||
Index string // index register, from (index*scale)
|
||||
Scale int // index scale; 0 when absent
|
||||
Offset int64 // leading displacement, from off(base)
|
||||
HasOff bool // a leading displacement is present
|
||||
Shift string // verbatim arm64 shift suffix, e.g. "<<2"
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package ast
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} }
|
||||
|
||||
func TestDeclPositions(t *testing.T) {
|
||||
hash := token.Token{Kind: token.Hash, Pos: pos(1, 1)}
|
||||
inc := &Include{Hash: hash}
|
||||
if inc.Pos() != hash.Pos {
|
||||
t.Errorf("Include.Pos() = %v, want %v", inc.Pos(), hash.Pos)
|
||||
}
|
||||
|
||||
pre := &Preproc{Hash: hash}
|
||||
if pre.Pos() != hash.Pos {
|
||||
t.Errorf("Preproc.Pos() = %v", pre.Pos())
|
||||
}
|
||||
|
||||
kw := token.Token{Kind: token.Ident, Text: "TEXT", Pos: pos(5, 1)}
|
||||
text := &Text{Keyword: kw}
|
||||
if text.Pos() != kw.Pos {
|
||||
t.Errorf("Text.Pos() = %v", text.Pos())
|
||||
}
|
||||
|
||||
gkw := token.Token{Kind: token.Ident, Text: "GLOBL", Pos: pos(7, 1)}
|
||||
if (&Globl{Keyword: gkw}).Pos() != gkw.Pos {
|
||||
t.Error("Globl.Pos()")
|
||||
}
|
||||
|
||||
dkw := token.Token{Kind: token.Ident, Text: "DATA", Pos: pos(8, 1)}
|
||||
if (&Data{Keyword: dkw}).Pos() != dkw.Pos {
|
||||
t.Error("Data.Pos()")
|
||||
}
|
||||
}
|
||||
|
||||
func TestStmtPositions(t *testing.T) {
|
||||
name := token.Token{Kind: token.Ident, Text: "loop", Pos: pos(3, 1)}
|
||||
if (&Label{Name: name}).Pos() != name.Pos {
|
||||
t.Error("Label.Pos()")
|
||||
}
|
||||
mnem := token.Token{Kind: token.Ident, Text: "RET", Pos: pos(4, 2)}
|
||||
if (&Instr{Mnemonic: mnem}).Pos() != mnem.Pos {
|
||||
t.Error("Instr.Pos()")
|
||||
}
|
||||
}
|
||||
|
||||
// TestInterfaces confirms the node types satisfy their interfaces, so callers
|
||||
// can range over Decls and Stmts.
|
||||
func TestInterfaces(t *testing.T) {
|
||||
var decls []Decl = []Decl{&Include{}, &Preproc{}, &Text{}, &Globl{}, &Data{}}
|
||||
if len(decls) != 5 {
|
||||
t.Fatal("decl interface set")
|
||||
}
|
||||
var stmts []Stmt = []Stmt{&Label{}, &Instr{}}
|
||||
if len(stmts) != 2 {
|
||||
t.Fatal("stmt interface set")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,280 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Command gasm is the developer frontend for GAsm — Go's Plan 9 assembler.
|
||||
// It bundles a token dumper, a parser, a formatter, a linter and a language
|
||||
// server into one binary. Every subcommand works headlessly so it can be
|
||||
// driven from scripts and CI as well as from an editor.
|
||||
package main
|
||||
|
||||
import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/format"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// version is the release version, stamped at build time via
|
||||
// -ldflags "-X main.version=…" (defaulting to the current release).
|
||||
var version = "0.1.0"
|
||||
|
||||
func main() {
|
||||
if len(os.Args) < 2 {
|
||||
usage(os.Stderr)
|
||||
os.Exit(2)
|
||||
}
|
||||
switch os.Args[1] {
|
||||
case "tokens":
|
||||
os.Exit(cmdTokens(os.Args[2:]))
|
||||
case "parse":
|
||||
os.Exit(cmdParse(os.Args[2:]))
|
||||
case "fmt":
|
||||
os.Exit(cmdFmt(os.Args[2:]))
|
||||
case "lint":
|
||||
os.Exit(cmdLint(os.Args[2:]))
|
||||
case "asm":
|
||||
os.Exit(cmdAsm(os.Args[2:]))
|
||||
case "lsp":
|
||||
os.Exit(cmdLSP(os.Args[2:]))
|
||||
case "version", "--version", "-V":
|
||||
fmt.Printf("gasm %s\n", version)
|
||||
case "help", "-h", "--help":
|
||||
usage(os.Stdout)
|
||||
default:
|
||||
fmt.Fprintf(os.Stderr, "gasm: unknown command %q\n\n", os.Args[1])
|
||||
usage(os.Stderr)
|
||||
os.Exit(2)
|
||||
}
|
||||
}
|
||||
|
||||
func usage(w io.Writer) {
|
||||
fmt.Fprintf(w, `gasm %s — developer tooling for Go's Plan 9 assembler
|
||||
|
||||
Usage:
|
||||
gasm tokens <file> print the lexical token stream
|
||||
gasm parse <file> parse and report syntax errors
|
||||
gasm fmt [-w] <file...> canonicalise formatting (-w writes in place)
|
||||
gasm lint <file...> run static checks
|
||||
gasm asm [-o out.bin] <file> assemble to machine code (amd64, Phase 2)
|
||||
gasm lsp run the language server over stdio
|
||||
gasm version print the version
|
||||
`, version)
|
||||
}
|
||||
|
||||
// readSource returns the contents of path, or stdin when path is "-".
|
||||
func readSource(path string) (string, error) {
|
||||
if path == "-" {
|
||||
b, err := io.ReadAll(os.Stdin)
|
||||
return string(b), err
|
||||
}
|
||||
b, err := os.ReadFile(path)
|
||||
return string(b), err
|
||||
}
|
||||
|
||||
func cmdTokens(args []string) int {
|
||||
fs := flag.NewFlagSet("tokens", flag.ExitOnError)
|
||||
fs.Parse(args)
|
||||
if fs.NArg() != 1 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm tokens <file>")
|
||||
return 2
|
||||
}
|
||||
src, err := readSource(fs.Arg(0))
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
return 1
|
||||
}
|
||||
for _, tok := range lexer.Tokenize(src) {
|
||||
fmt.Printf("%s\t%s\t%q\n", tok.Pos, tok.Kind, tok.Text)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func cmdParse(args []string) int {
|
||||
fs := flag.NewFlagSet("parse", flag.ExitOnError)
|
||||
fs.Parse(args)
|
||||
if fs.NArg() != 1 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm parse <file>")
|
||||
return 2
|
||||
}
|
||||
path := fs.Arg(0)
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
return 1
|
||||
}
|
||||
file, errs := parser.Parse(path, src)
|
||||
for _, e := range errs {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||
}
|
||||
if len(errs) > 0 {
|
||||
return 1
|
||||
}
|
||||
funcs := 0
|
||||
for _, d := range file.Decls {
|
||||
if _, ok := d.(*ast.Text); ok {
|
||||
funcs++
|
||||
}
|
||||
}
|
||||
fmt.Printf("%s: OK — %d declarations, %d functions\n", path, len(file.Decls), funcs)
|
||||
return 0
|
||||
}
|
||||
|
||||
func cmdFmt(args []string) int {
|
||||
fs := flag.NewFlagSet("fmt", flag.ExitOnError)
|
||||
write := fs.Bool("w", false, "write result to the source file")
|
||||
fs.Parse(args)
|
||||
if fs.NArg() == 0 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm fmt [-w] <file...>")
|
||||
return 2
|
||||
}
|
||||
rc := 0
|
||||
for _, path := range fs.Args() {
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
rc = 1
|
||||
continue
|
||||
}
|
||||
out := format.Source(path, src)
|
||||
if *write {
|
||||
if out != src {
|
||||
if err := os.WriteFile(path, []byte(out), 0o644); err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
rc = 1
|
||||
}
|
||||
}
|
||||
continue
|
||||
}
|
||||
fmt.Print(out)
|
||||
}
|
||||
return rc
|
||||
}
|
||||
|
||||
func cmdLint(args []string) int {
|
||||
fs := flag.NewFlagSet("lint", flag.ExitOnError)
|
||||
disable := fs.String("disable", "", "comma-separated rule codes to disable")
|
||||
fs.Parse(args)
|
||||
if fs.NArg() == 0 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm lint <file...>")
|
||||
return 2
|
||||
}
|
||||
disabled := map[string]bool{}
|
||||
for _, code := range strings.Split(*disable, ",") {
|
||||
if code = strings.TrimSpace(code); code != "" {
|
||||
disabled[code] = true
|
||||
}
|
||||
}
|
||||
hadError := false
|
||||
for _, path := range fs.Args() {
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
hadError = true
|
||||
continue
|
||||
}
|
||||
file, errs := parser.Parse(path, src)
|
||||
for _, e := range errs {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||
hadError = true
|
||||
}
|
||||
diags := lint.File(file, lint.Config{Arch: arch.FromFilename(path), Disable: disabled})
|
||||
for _, d := range diags {
|
||||
fmt.Printf("%s:%d:%d: %s: %s [%s]\n", path, d.Pos.Line, d.Pos.Column, d.Severity, d.Message, d.Code)
|
||||
if d.Severity == lint.Error {
|
||||
hadError = true
|
||||
}
|
||||
}
|
||||
}
|
||||
if hadError {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func cmdLSP(args []string) int {
|
||||
fs := flag.NewFlagSet("lsp", flag.ExitOnError)
|
||||
fs.Parse(args)
|
||||
srv := lsp.New(os.Stdin, os.Stdout)
|
||||
if err := srv.Run(); err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm lsp:", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func cmdAsm(args []string) int {
|
||||
fs := flag.NewFlagSet("asm", flag.ExitOnError)
|
||||
out := fs.String("o", "", "write the concatenated machine code to this file")
|
||||
fs.Parse(args)
|
||||
if fs.NArg() != 1 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gasm asm [-o out.bin] <file>")
|
||||
return 2
|
||||
}
|
||||
path := fs.Arg(0)
|
||||
if arch.FromFilename(path) != arch.AMD64 {
|
||||
fmt.Fprintln(os.Stderr, "gasm asm: only amd64 is supported in this Phase 2 increment")
|
||||
return 1
|
||||
}
|
||||
src, err := readSource(path)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||
return 1
|
||||
}
|
||||
f, errs := parser.Parse(path, src)
|
||||
for _, e := range errs {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||
}
|
||||
if len(errs) > 0 {
|
||||
return 1
|
||||
}
|
||||
|
||||
var all []byte
|
||||
functions := 0
|
||||
for _, d := range f.Decls {
|
||||
txt, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
code, _, err := asm.Assemble(txt)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "%s: %s: %v\n", path, txt.Name.Name, err)
|
||||
return 1
|
||||
}
|
||||
functions++
|
||||
fmt.Printf("%s: %d bytes\n", txt.Name.Name, len(code))
|
||||
for i := 0; i < len(code); i += 16 {
|
||||
end := i + 16
|
||||
if end > len(code) {
|
||||
end = len(code)
|
||||
}
|
||||
fmt.Printf(" %04x:", i)
|
||||
for _, b := range code[i:end] {
|
||||
fmt.Printf(" %02x", b)
|
||||
}
|
||||
fmt.Println()
|
||||
}
|
||||
all = append(all, code...)
|
||||
}
|
||||
if functions == 0 {
|
||||
fmt.Fprintln(os.Stderr, "gasm asm: no assemblable TEXT functions found")
|
||||
return 1
|
||||
}
|
||||
if *out != "" {
|
||||
if err := os.WriteFile(*out, all, 0o644); err != nil {
|
||||
fmt.Fprintln(os.Stderr, "gasm asm:", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Printf("wrote %d bytes to %s\n", len(all), *out)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -0,0 +1,174 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
const clean = "#include \"textflag.h\"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"\tMOVQ AX, BX\n" +
|
||||
"loop:\n" +
|
||||
"\tJMP loop\n" +
|
||||
"\tRET\n"
|
||||
|
||||
const buggy = "#include \"textflag.h\"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"\tBOGUS AX, BX\n" +
|
||||
"\tJMP nowhere\n" +
|
||||
"\tRET\n"
|
||||
|
||||
func writeTemp(t *testing.T, name, content string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), name)
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
// capture runs fn with stdout and stderr redirected and returns both plus the
|
||||
// exit code fn produced.
|
||||
func capture(fn func() int) (stdout, stderr string, code int) {
|
||||
oldOut, oldErr := os.Stdout, os.Stderr
|
||||
rOut, wOut, _ := os.Pipe()
|
||||
rErr, wErr, _ := os.Pipe()
|
||||
os.Stdout, os.Stderr = wOut, wErr
|
||||
|
||||
code = fn()
|
||||
|
||||
wOut.Close()
|
||||
wErr.Close()
|
||||
os.Stdout, os.Stderr = oldOut, oldErr
|
||||
ob, _ := io.ReadAll(rOut)
|
||||
eb, _ := io.ReadAll(rErr)
|
||||
return string(ob), string(eb), code
|
||||
}
|
||||
|
||||
func TestCmdTokens(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", clean)
|
||||
out, _, code := capture(func() int { return cmdTokens([]string{path}) })
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d", code)
|
||||
}
|
||||
if !strings.Contains(out, "IDENT") || !strings.Contains(out, "TEXT") {
|
||||
t.Errorf("token dump missing expected tokens:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdParseOK(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", clean)
|
||||
out, _, code := capture(func() int { return cmdParse([]string{path}) })
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d", code)
|
||||
}
|
||||
if !strings.Contains(out, "OK") || !strings.Contains(out, "1 functions") {
|
||||
t.Errorf("parse output = %q", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdParseError(t *testing.T) {
|
||||
path := writeTemp(t, "bad_amd64.s", "TEXT ·f(SB), NOSPLIT, $0\n) (\n\tRET\n")
|
||||
_, errOut, code := capture(func() int { return cmdParse([]string{path}) })
|
||||
if code != 1 {
|
||||
t.Fatalf("code = %d, want 1", code)
|
||||
}
|
||||
if errOut == "" {
|
||||
t.Error("expected a parse error on stderr")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdParseMissingFile(t *testing.T) {
|
||||
_, _, code := capture(func() int { return cmdParse([]string{"/nonexistent/file.s"}) })
|
||||
if code != 1 {
|
||||
t.Fatalf("code = %d, want 1", code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdLintClean(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", clean)
|
||||
_, _, code := capture(func() int { return cmdLint([]string{path}) })
|
||||
if code != 0 {
|
||||
t.Fatalf("clean file should lint with code 0, got %d", code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdLintErrors(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", buggy)
|
||||
out, _, code := capture(func() int { return cmdLint([]string{path}) })
|
||||
if code != 1 {
|
||||
t.Fatalf("code = %d, want 1", code)
|
||||
}
|
||||
if !strings.Contains(out, "unknown-instruction") || !strings.Contains(out, "undefined-label") {
|
||||
t.Errorf("lint output missing expected codes:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdLintDisable(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", buggy)
|
||||
args := []string{"-disable", "unknown-instruction,undefined-label", path}
|
||||
_, _, code := capture(func() int { return cmdLint(args) })
|
||||
if code != 0 {
|
||||
t.Fatalf("disabling both rules should yield code 0, got %d", code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdFmtStdout(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", "TEXT ·f(SB),NOSPLIT,$0\nMOVQ AX,BX\nRET\n")
|
||||
out, _, code := capture(func() int { return cmdFmt([]string{path}) })
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d", code)
|
||||
}
|
||||
if !strings.Contains(out, "TEXT ·f(SB), NOSPLIT, $0") || !strings.Contains(out, "\tMOVQ AX, BX") {
|
||||
t.Errorf("formatted output unexpected:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdFmtWrite(t *testing.T) {
|
||||
path := writeTemp(t, "f_amd64.s", "TEXT ·f(SB),NOSPLIT,$0\nMOVQ AX,BX\nRET\n")
|
||||
_, _, code := capture(func() int { return cmdFmt([]string{"-w", path}) })
|
||||
if code != 0 {
|
||||
t.Fatalf("code = %d", code)
|
||||
}
|
||||
b, _ := os.ReadFile(path)
|
||||
if !strings.Contains(string(b), "\tMOVQ AX, BX") {
|
||||
t.Errorf("file not rewritten:\n%s", b)
|
||||
}
|
||||
// Idempotent: a second -w pass leaves the file unchanged.
|
||||
_, _, _ = capture(func() int { return cmdFmt([]string{"-w", path}) })
|
||||
b2, _ := os.ReadFile(path)
|
||||
if string(b) != string(b2) {
|
||||
t.Error("fmt -w is not idempotent")
|
||||
}
|
||||
}
|
||||
|
||||
func TestUsage(t *testing.T) {
|
||||
var b bytes.Buffer
|
||||
usage(&b)
|
||||
if !strings.Contains(b.String(), "gasm") {
|
||||
t.Errorf("usage text unexpected:\n%s", b.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdArgErrors(t *testing.T) {
|
||||
// Missing file arguments produce a usage error (code 2).
|
||||
if _, _, code := capture(func() int { return cmdFmt(nil) }); code != 2 {
|
||||
t.Errorf("cmdFmt() code = %d, want 2", code)
|
||||
}
|
||||
if _, _, code := capture(func() int { return cmdLint(nil) }); code != 2 {
|
||||
t.Errorf("cmdLint() code = %d, want 2", code)
|
||||
}
|
||||
if _, _, code := capture(func() int { return cmdTokens(nil) }); code != 2 {
|
||||
t.Errorf("cmdTokens() code = %d, want 2", code)
|
||||
}
|
||||
if _, _, code := capture(func() int { return cmdParse(nil) }); code != 2 {
|
||||
t.Errorf("cmdParse() code = %d, want 2", code)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,212 @@
|
||||
# Architecture
|
||||
|
||||
How gasm-devkit is put together and why.
|
||||
|
||||
## Design goals
|
||||
|
||||
1. **A real AST, not a grammar hack.** The linter, analyser, assembler and
|
||||
language server all need to *reason* about assembly — not just colour it.
|
||||
So the centre of the toolkit is a hand-written lexer and a parser that
|
||||
produce a typed AST with source positions on every node.
|
||||
2. **Architecture as data, not code.** Per-architecture differences (amd64,
|
||||
arm64, riscv64, loong64) live in register and instruction *tables* (`arch`),
|
||||
never in `if arch == …` branches scattered through the logic. The
|
||||
instruction tables are generated from the Go toolchain's own assembler
|
||||
source (`just gen`), so adding or refreshing an architecture is a data
|
||||
operation, not a coding one.
|
||||
3. **Open integration surface.** Everything the toolkit can do is reachable
|
||||
through two vendor-neutral interfaces: a CLI and an LSP server. No editor
|
||||
owns the toolkit; the toolkit is offered to editors on standard terms.
|
||||
|
||||
## Pipeline
|
||||
|
||||
```mermaid
|
||||
graph TD
|
||||
SRC["source .s"] --> LEX["lexer<br/>token stream"]
|
||||
LEX --> PAR["parser<br/>AST + diagnostics"]
|
||||
LEX --> FMT["format<br/>re-space tokens"]
|
||||
PAR --> LINT["lint<br/>static checks"]
|
||||
PAR --> LSP["lsp server"]
|
||||
LEX --> LSP
|
||||
ARCH["arch tables<br/>amd64 / arm64 / riscv64 / loong64"] --> LINT
|
||||
ARCH --> LSP
|
||||
LINT --> LSP
|
||||
FMT --> CLI["gasm CLI"]
|
||||
LINT --> CLI
|
||||
PAR --> CLI
|
||||
LEX --> CLI
|
||||
LSP --> EDITOR["any LSP editor"]
|
||||
```
|
||||
|
||||
The lexer is the shared foundation: the parser builds the AST from it, the
|
||||
formatter re-spaces its tokens directly, and the language server uses it for
|
||||
semantic highlighting.
|
||||
|
||||
## Components
|
||||
|
||||
### `token` and `lexer`
|
||||
|
||||
The scanner is hand-written and permissive: it never panics and maps anything
|
||||
it cannot classify to an `Illegal` token, so every downstream tool still works
|
||||
on malformed input. Newlines are significant tokens, because Plan 9 assembly
|
||||
is line-oriented and the parser relies on line structure.
|
||||
|
||||
The middle dot (`·`, U+00B7) is treated as an identifier character so that
|
||||
`·funcName(SB)` lexes as one symbol. Multi-character operators (`<<`, `>>`,
|
||||
`->`) are recognised so arm64 shift operands scan correctly. A backslash
|
||||
immediately before a newline is a C-preprocessor line continuation (used by
|
||||
`#define` macros in the runtime `.s` files); the lexer splices the lines
|
||||
together so a multi-line macro becomes one logical line the parser treats as an
|
||||
opaque preprocessor directive.
|
||||
|
||||
### `ast` and `parser`
|
||||
|
||||
The parser is **line-oriented**, matching how the Plan 9 assembler reads a
|
||||
file: it groups tokens into lines, classifies each line (directive, label,
|
||||
instruction, comment, preprocessor) and dispatches. A malformed line is
|
||||
reported and skipped; it never aborts the file.
|
||||
|
||||
Operands are parsed into a faithful, flat representation. The amd64
|
||||
addressing modes — `reg`, `$imm`, `(base)`, `off(base)`, `(base)(index*scale)`,
|
||||
`name+off(FP)`, `name<>(SB)` — are all captured structurally, and the original
|
||||
token text is retained for fidelity.
|
||||
|
||||
A deliberate boundary: the AST records **syntax only**. Whether a bare
|
||||
identifier is a register or a label is an *architecture* question, so it is
|
||||
left to `arch` and resolved in the lint/lsp layers. This keeps the parser
|
||||
arch-agnostic and its output deterministic.
|
||||
|
||||
### `arch`
|
||||
|
||||
Register files are generated programmatically (the regular `R8`–`R15`,
|
||||
`X0`–`X15`, `Y0`–`Y15`, `Z0`–`Z31`, `K0`–`K7` ranges) plus the irregularly
|
||||
named registers listed explicitly. Instruction names are **generated from the
|
||||
Go toolchain's own assembler source** (`cmd/internal/obj/<arch>/anames.go`,
|
||||
plus the common opcodes and the per-architecture front-end aliases such as the
|
||||
arm64 `B`/`BL` branches and the `.P`/`.W` load-store addressing suffixes) by
|
||||
`just gen`, so the tables always match what the real assembler accepts. Each
|
||||
mnemonic maps to a summary and an optional operand-count range; counts are
|
||||
recorded only where unambiguous (`-1` disables the operand-count lint for that
|
||||
instruction) so the linter stays silent rather than guess. For architectures
|
||||
with highly variable operand forms (arm64, riscv64, loong64) only a few
|
||||
fixed-arity instructions (`RET`, `NOP`, `JMP`, `CALL`) carry counts at all.
|
||||
|
||||
### `lint`
|
||||
|
||||
Rules are conservative by design — silence beats a false positive. The rules
|
||||
are `unknown-instruction`, `operand-count`, `undefined-label`,
|
||||
`duplicate-label`, `missing-ret`, `missing-textflag-include`, `abi-argsize` and
|
||||
`unreachable-code`. Every diagnostic carries a stable code so callers can
|
||||
disable rules individually, and arch-specific rules switch off entirely when
|
||||
the target architecture cannot be inferred from the file name.
|
||||
|
||||
Two things keep the rules honest on real-world code:
|
||||
|
||||
- **Pseudo-ops and macros are not instructions.** `unknown-instruction` knows
|
||||
the assembler pseudo-ops (`BYTE`, `WORD`, `FUNCDATA`, `PCDATA`, …) and
|
||||
recognises macro invocations — an in-file `#define` name, or any identifier
|
||||
containing an underscore (no Plan 9 mnemonic ever does).
|
||||
- **Macro-heavy files get the label/RET heuristics turned off.** Without a
|
||||
preprocessor, labels a macro defines are invisible, so `undefined-label` and
|
||||
`missing-ret` are suppressed for files that use macros (an in-file `#define`
|
||||
or a `#include` of anything other than `textflag.h`). `missing-ret` also
|
||||
treats a trailing unconditional jump and `UNDEF` as valid terminators.
|
||||
|
||||
The result is validated by `TestGoRuntimeCorpus`, which parses and lints every
|
||||
`src/runtime/*.s` file the toolchain ships for all four architectures and
|
||||
asserts zero parse errors and zero error-severity diagnostics.
|
||||
|
||||
Two deeper analyses sit on top of the AST:
|
||||
|
||||
- **`abi-argsize`.** Hand-written kernels document their signature in a
|
||||
`// func …` comment above the `TEXT`. The linter parses that signature with
|
||||
the standard library's Go parser, lays out the parameters and results under
|
||||
Go's ABI0 stack rules (results begin on a word boundary after the
|
||||
parameters), and checks the total against the argument size declared in the
|
||||
`TEXT` directive. It only runs for stack-argument functions (a non-zero
|
||||
declared arg area that is actually addressed through `FP`), and aborts
|
||||
silently on a type whose size it cannot determine — so it never guesses.
|
||||
- **`unreachable-code`.** Code after a `RET` and before the next label is
|
||||
dead. The check is suppressed for any function whose reachability cannot be
|
||||
decided statically: those using PC-relative jumps (`JMP 2(PC)`),
|
||||
register-indirect branches (`JALR`/`JR`/`JIRL`/`BR`/`BLR`), or living in a
|
||||
file with `#ifdef` conditionals. `UNDEF` is deliberately not a terminator —
|
||||
code after it is occasionally intentional metadata.
|
||||
- **`register-clobber` (register liveness).** The linter builds the function's
|
||||
control-flow graph (basic blocks split at labels and after branches, with
|
||||
fall-through and jump-target edges), computes a conservative per-instruction
|
||||
register def/use, and runs the standard backward liveness iteration to a fixed
|
||||
point. On top of that it flags a **callee-saved register that is written but
|
||||
never saved and restored** — the per-architecture callee-saved set is amd64
|
||||
`BX/BP/R12–R15`, arm64 `R19–R30`, riscv64 `X1/X8/X9/X18–X27`, loong64
|
||||
`R1/R22–R31`. This is an *audit*: the runtime's own assembly clobbers these
|
||||
registers freely (it controls both sides of the call), so the rule is
|
||||
advisory there, but in hand-written kernels called from ordinary Go code a
|
||||
clobber is a genuine ABI violation. It runs only on macro-free files, where
|
||||
no opaque macro can perform the save/restore.
|
||||
- **`funcdata-pcdata`.** `FUNCDATA $idx, sym(SB)` and `PCDATA $idx, $val` are
|
||||
checked for well-formed operands (arity, immediate index and value, symbol
|
||||
reference) and a literal index is range-checked; a named index constant such
|
||||
as `$PCDATA_StackMapIndex` is accepted without a range check.
|
||||
|
||||
### `format`
|
||||
|
||||
The formatter works on the **token stream, not the AST**, so it preserves
|
||||
every line — comments and blanks included. It only normalises indentation,
|
||||
operand spacing and per-function mnemonic alignment. It is idempotent and its
|
||||
output always round-trips through the parser.
|
||||
|
||||
### `lsp`
|
||||
|
||||
The server speaks JSON-RPC 2.0 with `Content-Length` framing over any
|
||||
`io.Reader`/`io.Writer` (normally stdin/stdout). It maintains an in-memory
|
||||
document store, republishes diagnostics on every change, and provides:
|
||||
|
||||
- **completion** — instructions, registers, pseudo-registers, textflag macros
|
||||
and local labels;
|
||||
- **hover** — instruction summaries and register descriptions from `arch`;
|
||||
- **document symbols** — `TEXT` functions with their labels, plus `GLOBL`/`DATA`;
|
||||
- **semantic tokens** — syntax highlighting delivered as LSP semantic tokens,
|
||||
classified with the lexer plus `arch` (instructions, registers by class,
|
||||
pseudo-registers, labels, immediates, comments, directives, textflag macros).
|
||||
|
||||
Semantic tokens are the key to editor-agnostic highlighting: the editor renders
|
||||
them from the standard LSP legend, so no editor-specific grammar is needed.
|
||||
|
||||
### `asm`
|
||||
|
||||
The standalone assembler (Phase 2). Its core is an amd64 instruction encoder:
|
||||
a REX/ModR-M/SIB/displacement/immediate engine plus the scalar instruction set,
|
||||
with the Plan 9 operand order (source first) mapped onto the x86 encoding.
|
||||
Every encoding is validated by decoding it again with `golang.org/x/arch` — the
|
||||
one module dependency, used in tests only and never linked into the binary.
|
||||
|
||||
On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each
|
||||
operand to an encoder operand, and lays the instructions out in two passes so
|
||||
local labels resolve to fixed rel32 jump offsets. The `FP`/`SP` pseudo-
|
||||
registers are translated onto the hardware stack pointer — `x+N(FP)` becomes
|
||||
`(N+8)(SP)` for a zero-frame function and `(N+frame+16)(SP)` once a frame
|
||||
pointer is set up, with the matching Go prologue/epilogue generated — so the
|
||||
output is byte-identical to the Go assembler for these cases. SIMD is handled
|
||||
SIMD is handled
|
||||
by a VEX (AVX/AVX2) encoder — the two- and three-byte VEX prefixes with XMM/YMM
|
||||
registers — across three operand forms (the three-operand NDS form, the
|
||||
two-operand reg/rm form, and the immediate-shift form), together covering the
|
||||
bulk of the integer SIMD set; each encoding is validated by round-trip
|
||||
decoding. This increment covers register / memory / immediate / FP-frame
|
||||
operands, local-label jumps and these VEX SIMD forms; the remaining SIMD forms
|
||||
(shuffles, extract/insert, permute, moves), EVEX / AVX-512, `SB` (global
|
||||
symbol) operands (relocations) and object-file emission are the rest of
|
||||
Phase 2.
|
||||
|
||||
## Extension points
|
||||
|
||||
- **New architecture:** add an entry to the generator in `_gen`, run
|
||||
`just gen`, and add a `buildXXX()` register file plus a case in `ForArch`.
|
||||
- **New lint rule:** add a function in `lint` and a rule-code constant.
|
||||
- **New LSP feature:** add a method case in `dispatch` and a handler.
|
||||
|
||||
The phases follow a dependency chain. Phase 1 (static analysis) builds only on
|
||||
the AST; Phase 2 (the standalone assembler) emits object code; Phases 3
|
||||
(dynamic analysis) and 4 (the debugger) both consume the execution substrate
|
||||
that the assembler provides.
|
||||
+79
@@ -0,0 +1,79 @@
|
||||
# Using gasm-devkit with Zed
|
||||
|
||||
This document is deliberately blunt, because the situation is a genuine
|
||||
conflict between two of the project's own commitments, and papering over it
|
||||
would be dishonest.
|
||||
|
||||
## The conflict
|
||||
|
||||
gasm-devkit is **pure Go, no C, no cgo, no JavaScript runtimes, no native
|
||||
binaries, no vendor lock-in, no platform-specific IDE internals.**
|
||||
|
||||
Zed's extension model, as verified against Zed's own documentation, is:
|
||||
|
||||
- Extensions are written in **Rust** and compiled to **WebAssembly**
|
||||
(`wasm32-wasip2`).
|
||||
- Syntax highlighting is provided by **Tree-sitter** grammars, which are
|
||||
**C** compiled to WebAssembly with the wasi-sdk, from a grammar written in a
|
||||
**JavaScript** DSL.
|
||||
- A *new* language cannot be registered through configuration alone. Defining
|
||||
a language requires an extension, and every language extension must name a
|
||||
Tree-sitter grammar. (Zed's `lsp` settings section configures
|
||||
already-registered servers; it does not register an arbitrary external binary
|
||||
for a brand-new language.)
|
||||
|
||||
There is therefore **no pure-Go path into Zed's extension host.** This is a
|
||||
property of Zed, not of gasm-devkit: no language tooling author can feed Zed a
|
||||
pure-Go highlighting grammar, because Zed's highlighting engine is Tree-sitter
|
||||
and its plugin runtime is Rust/WASM.
|
||||
|
||||
## What gasm-devkit gives Zed regardless
|
||||
|
||||
The toolkit's integration surface is the **Language Server Protocol**, an open
|
||||
standard. Through `gasm lsp` it provides, with zero editor-specific code:
|
||||
|
||||
- autocomplete (instructions, registers, pseudo-registers, labels),
|
||||
- hover documentation,
|
||||
- diagnostics (the linter, pushed as you type),
|
||||
- document outline (functions and labels),
|
||||
- **syntax highlighting, delivered as LSP semantic tokens.**
|
||||
|
||||
That last point matters: Zed can render highlighting entirely from LSP semantic
|
||||
tokens (`"semantic_tokens": "full"` replaces Tree-sitter highlighting for a
|
||||
language). So the highlighting *capability* exists in pure Go; what Zed needs
|
||||
is merely to be told that `.s` files are a language served by `gasm lsp`.
|
||||
|
||||
## The honest options
|
||||
|
||||
1. **Use an editor that registers an external LSP by configuration.**
|
||||
Neovim, Helix, VS Code and Sublime all let you associate `.s` with the
|
||||
`gasm lsp` binary and use its semantic tokens — no Rust, no C, no lock-in.
|
||||
This is the option that satisfies every stated constraint with no
|
||||
exception.
|
||||
|
||||
2. **Treat a Zed adapter as one quarantined exception.** A minimal Zed
|
||||
extension — a few lines of Rust that register the language and launch
|
||||
`gasm lsp` — plus either a Tree-sitter grammar or `"full"` semantic tokens
|
||||
for highlighting. Crucially, this adapter is the *editor's plugin format*;
|
||||
it is sandboxed inside Zed and never linked into, compiled into, or shipped
|
||||
with the Go toolkit. gasm-devkit itself stays pure Go. But producing it
|
||||
uses the Rust/wasi-sdk/Tree-sitter toolchain, which the project constraints
|
||||
forbid — so it must be a conscious, explicit decision, not a silent one.
|
||||
|
||||
The author's philosophy — digital sovereignty, no dependency on toolchains he
|
||||
does not control — is the tie-breaker, and it is a value judgement rather than
|
||||
a technical one. gasm-devkit is built so that **either** choice keeps the
|
||||
toolkit itself clean: the pure-Go core and the LSP are the product; a Zed
|
||||
adapter, if ever wanted, is a thin, separable leaf.
|
||||
|
||||
## Wiring the LSP (editor-agnostic)
|
||||
|
||||
Run the server and point an LSP client at it:
|
||||
|
||||
```sh
|
||||
go run ./cmd/gasm lsp # or: go install ./cmd/gasm && gasm lsp
|
||||
```
|
||||
|
||||
Associate the command with `*.s` (and `*_amd64.s` / `*_arm64.s`) in whichever
|
||||
editor you use. The server infers the target architecture from the file-name
|
||||
suffix and selects the amd64 or arm64 instruction tables accordingly.
|
||||
@@ -0,0 +1,213 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package format implements a canonical formatter for GAsm source — the
|
||||
// equivalent of gofmt for Plan 9 assembly. It works on the token stream
|
||||
// rather than the AST so that every line (including comments and blanks) is
|
||||
// preserved; it only normalises indentation, operand spacing and per-function
|
||||
// mnemonic alignment. Formatting is idempotent.
|
||||
package format
|
||||
|
||||
import (
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// Source returns the canonical formatting of src.
|
||||
func Source(path, src string) string {
|
||||
lines := splitLines(lexer.Tokenize(src))
|
||||
|
||||
// First pass: classify each line and record, for every instruction, the
|
||||
// index of the TEXT function it belongs to, so that mnemonic widths can be
|
||||
// aligned per function.
|
||||
type info struct {
|
||||
kind int
|
||||
mnemLen int
|
||||
funcID int
|
||||
}
|
||||
const (
|
||||
kBlank = iota
|
||||
kComment
|
||||
kPreproc
|
||||
kDirective
|
||||
kLabel
|
||||
kInstr
|
||||
)
|
||||
|
||||
infos := make([]info, len(lines))
|
||||
funcID := -1
|
||||
maxWidth := map[int]int{} // funcID -> widest mnemonic
|
||||
for i, line := range lines {
|
||||
inf := info{kind: kBlank, funcID: funcID}
|
||||
if len(line) > 0 {
|
||||
switch {
|
||||
case line[0].Kind == token.Comment:
|
||||
inf.kind = kComment
|
||||
case line[0].Kind == token.Hash:
|
||||
inf.kind = kPreproc
|
||||
case line[0].Kind == token.Ident && isDirective(line[0].Text):
|
||||
inf.kind = kDirective
|
||||
if line[0].Text == "TEXT" {
|
||||
funcID++
|
||||
inf.funcID = funcID
|
||||
} else {
|
||||
funcID = -1
|
||||
inf.funcID = -1
|
||||
}
|
||||
case len(line) >= 2 && line[1].Kind == token.Colon:
|
||||
inf.kind = kLabel
|
||||
default:
|
||||
inf.kind = kInstr
|
||||
inf.funcID = funcID
|
||||
inf.mnemLen = len(line[0].Text)
|
||||
if funcID >= 0 && inf.mnemLen > maxWidth[funcID] {
|
||||
maxWidth[funcID] = inf.mnemLen
|
||||
}
|
||||
}
|
||||
}
|
||||
infos[i] = inf
|
||||
}
|
||||
|
||||
// Second pass: render.
|
||||
var b strings.Builder
|
||||
inBody := false
|
||||
for i, line := range lines {
|
||||
inf := infos[i]
|
||||
var out string
|
||||
switch inf.kind {
|
||||
case kBlank:
|
||||
out = ""
|
||||
case kComment:
|
||||
if inBody {
|
||||
out = "\t" + line[0].Text
|
||||
} else {
|
||||
out = line[0].Text
|
||||
}
|
||||
case kPreproc:
|
||||
out = renderPreproc(line)
|
||||
case kDirective:
|
||||
out = line[0].Text + " " + renderOps(line[1:])
|
||||
inBody = line[0].Text == "TEXT"
|
||||
case kLabel:
|
||||
out = line[0].Text + ":"
|
||||
// A label may share its line with an instruction; emit the
|
||||
// instruction on the following line.
|
||||
if rest := line[2:]; len(rest) > 0 {
|
||||
out += "\n" + renderInstr(rest, maxWidth[inf.funcID])
|
||||
}
|
||||
case kInstr:
|
||||
out = renderInstr(line, maxWidth[inf.funcID])
|
||||
}
|
||||
b.WriteString(strings.TrimRight(out, " \t"))
|
||||
b.WriteByte('\n')
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// renderInstr renders an instruction line: a tab, the mnemonic padded to the
|
||||
// function's alignment width, then the re-spaced operands.
|
||||
func renderInstr(line []token.Token, width int) string {
|
||||
if len(line) == 0 {
|
||||
return ""
|
||||
}
|
||||
mnem := line[0].Text
|
||||
ops := renderOps(line[1:])
|
||||
if ops == "" {
|
||||
return "\t" + mnem
|
||||
}
|
||||
if width < len(mnem) {
|
||||
width = len(mnem)
|
||||
}
|
||||
return "\t" + mnem + strings.Repeat(" ", width-len(mnem)) + " " + ops
|
||||
}
|
||||
|
||||
// renderPreproc renders a preprocessor line such as #include "textflag.h".
|
||||
func renderPreproc(line []token.Token) string {
|
||||
// "#" directive [args]
|
||||
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
|
||||
line[2].Kind == token.String {
|
||||
return "#include " + line[2].Text
|
||||
}
|
||||
parts := make([]string, 0, len(line)-1)
|
||||
for _, t := range line[1:] {
|
||||
parts = append(parts, t.Text)
|
||||
}
|
||||
return "#" + strings.Join(parts, " ")
|
||||
}
|
||||
|
||||
// renderOps re-spaces a run of operand tokens into canonical form. It never
|
||||
// invents or drops token text; it only chooses the whitespace between tokens.
|
||||
func renderOps(toks []token.Token) string {
|
||||
var b strings.Builder
|
||||
for i, t := range toks {
|
||||
if i > 0 && spaceBetween(toks[i-1], t) {
|
||||
b.WriteByte(' ')
|
||||
}
|
||||
b.WriteString(t.Text)
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// spaceBetween decides whether a single space separates prev and cur.
|
||||
func spaceBetween(prev, cur token.Token) bool {
|
||||
switch cur.Kind {
|
||||
case token.RParen:
|
||||
return false
|
||||
case token.Comma:
|
||||
return false
|
||||
case token.Star, token.Plus, token.Minus, token.Slash:
|
||||
return false
|
||||
case token.LShift, token.RShift, token.Arrow, token.At:
|
||||
return false
|
||||
case token.LAngle, token.RAngle:
|
||||
return false
|
||||
case token.LParen:
|
||||
// Attach '(' to a preceding name, number, ')' or '>'.
|
||||
switch prev.Kind {
|
||||
case token.Ident, token.Number, token.RParen, token.RAngle:
|
||||
return false
|
||||
default:
|
||||
return true
|
||||
}
|
||||
}
|
||||
switch prev.Kind {
|
||||
case token.LParen, token.Star, token.Plus, token.Minus, token.Slash:
|
||||
return false
|
||||
case token.Dollar:
|
||||
return false
|
||||
case token.LShift, token.RShift, token.Arrow, token.At:
|
||||
return false
|
||||
case token.LAngle, token.RAngle:
|
||||
return false
|
||||
case token.Comma:
|
||||
return true
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func isDirective(s string) bool {
|
||||
return s == "TEXT" || s == "DATA" || s == "GLOBL"
|
||||
}
|
||||
|
||||
// splitLines groups tokens into lines, dropping Newline and EOF tokens.
|
||||
func splitLines(toks []token.Token) [][]token.Token {
|
||||
var lines [][]token.Token
|
||||
var cur []token.Token
|
||||
for _, t := range toks {
|
||||
if t.Kind == token.EOF {
|
||||
break
|
||||
}
|
||||
if t.Kind == token.Newline {
|
||||
lines = append(lines, cur)
|
||||
cur = nil
|
||||
continue
|
||||
}
|
||||
cur = append(cur, t)
|
||||
}
|
||||
if len(cur) > 0 {
|
||||
lines = append(lines, cur)
|
||||
}
|
||||
return lines
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package format
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
func TestGolden(t *testing.T) {
|
||||
in := "#include \"textflag.h\"\n" +
|
||||
"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"MOVQ swin_base+0(FP), SI\n" +
|
||||
"LEAQ (SI)(BX*4), R9\n" +
|
||||
"ANDQ $-8, R10\n" +
|
||||
"VFMADD231PD Z14, Z12, Z10\n" +
|
||||
"RET\n"
|
||||
|
||||
want := "#include \"textflag.h\"\n" +
|
||||
"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"\tMOVQ swin_base+0(FP), SI\n" +
|
||||
"\tLEAQ (SI)(BX*4), R9\n" +
|
||||
"\tANDQ $-8, R10\n" +
|
||||
"\tVFMADD231PD Z14, Z12, Z10\n" +
|
||||
"\tRET\n"
|
||||
|
||||
got := Source("f_amd64.s", in)
|
||||
if got != want {
|
||||
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOperandSpacing(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"4(SI)": "4(SI)",
|
||||
"(SI)(BX*4)": "(SI)(BX*4)",
|
||||
"$-8": "$-8",
|
||||
"$0x80020100": "$0x80020100",
|
||||
"swin_base+0(FP)": "swin_base+0(FP)",
|
||||
"mask24<>(SB)": "mask24<>(SB)",
|
||||
"·idx16+0(SB)/4": "·idx16+0(SB)/4",
|
||||
}
|
||||
for in, want := range cases {
|
||||
toks := lexOperands(in)
|
||||
if got := renderOps(toks); got != want {
|
||||
t.Errorf("renderOps(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// lexOperands lexes a single operand string and drops the EOF token.
|
||||
func lexOperands(s string) []token.Token {
|
||||
toks := lexer.Tokenize(s)
|
||||
return toks[:len(toks)-1] // drop trailing EOF
|
||||
}
|
||||
|
||||
func TestIdempotent(t *testing.T) {
|
||||
src, err := os.ReadFile("../testdata/sample_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
once := Source("sample_amd64.s", string(src))
|
||||
twice := Source("sample_amd64.s", once)
|
||||
if once != twice {
|
||||
t.Fatal("formatting is not idempotent on the fixture")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRoundTrip checks that formatting produces source that still parses
|
||||
// cleanly, on the fixture and on the real go-flac kernels when present.
|
||||
func TestRoundTrip(t *testing.T) {
|
||||
files := []string{"../testdata/sample_amd64.s"}
|
||||
real, _ := filepath.Glob("../../go-libraries/go-*/*.s")
|
||||
files = append(files, real...)
|
||||
for _, path := range files {
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
formatted := Source(path, string(src))
|
||||
if _, errs := parser.Parse(path, formatted); len(errs) > 0 {
|
||||
t.Errorf("formatted %s no longer parses: %v", path, errs)
|
||||
}
|
||||
if strings.TrimSpace(formatted) == "" {
|
||||
t.Errorf("formatted %s is empty", path)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
module sourcedock.dev/petrbalvin/gasm-devkit
|
||||
|
||||
go 1.26
|
||||
|
||||
toolchain go1.26.5
|
||||
|
||||
require golang.org/x/arch v0.29.0
|
||||
@@ -0,0 +1,2 @@
|
||||
golang.org/x/arch v0.29.0 h1:8sSET5wB0+exBm0FGmOtdHMqjlRdV2DRD3/IV6OZgho=
|
||||
golang.org/x/arch v0.29.0/go.mod h1:0X+GdSIP+kL5wPmpK7sdkEVTt2XoYP0cSjQSbZBwOi8=
|
||||
@@ -0,0 +1,45 @@
|
||||
# Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
# SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
# gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm).
|
||||
|
||||
version := "0.1.0"
|
||||
|
||||
default:
|
||||
@just --list
|
||||
|
||||
# Download module dependencies.
|
||||
install:
|
||||
go mod download
|
||||
|
||||
# Vet + gofmt check — zero errors, zero warnings.
|
||||
build:
|
||||
go vet ./...
|
||||
@test -z "$(gofmt -l .)" || { echo "gofmt diff:"; gofmt -l .; exit 1; }
|
||||
|
||||
# Full test suite + race detector + 80 % coverage gate.
|
||||
test:
|
||||
go test -race -count=1 -coverprofile=coverage.out ./...
|
||||
go tool cover -func=coverage.out | awk '/^total:/{gsub("%","",$3);if($3+0<80){print "coverage "$3"% < 80%";exit 1}print "coverage "$3"%"}'
|
||||
|
||||
# Format all Go sources.
|
||||
fmt:
|
||||
gofmt -w .
|
||||
|
||||
# Run the gasm CLI (pass args after --, e.g. `just run -- lint file.s`).
|
||||
run *ARGS:
|
||||
go run -ldflags "-X main.version={{version}}" ./cmd/gasm {{ARGS}}
|
||||
|
||||
# Install the gasm binary into $GOBIN (stamped with the release version).
|
||||
install-bin:
|
||||
go install -ldflags "-X main.version={{version}}" ./cmd/gasm
|
||||
|
||||
# Regenerate the architecture instruction tables from the Go toolchain source.
|
||||
gen:
|
||||
go run _gen/gen.go
|
||||
gofmt -w arch/
|
||||
|
||||
# Remove build artefacts.
|
||||
uninstall:
|
||||
rm -f coverage.out gasm
|
||||
find . -name '*.test' -delete
|
||||
+395
@@ -0,0 +1,395 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package lexer implements a hand-written scanner for Go's Plan 9 assembler
|
||||
// (GAsm). It turns a source string into a flat token stream that the parser,
|
||||
// formatter and language server all build on. The scanner is deliberately
|
||||
// permissive: it never panics and maps anything it cannot classify to an
|
||||
// Illegal token so that downstream tools can still operate on malformed input.
|
||||
package lexer
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"unicode"
|
||||
"unicode/utf8"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// middleDot is the Plan 9 symbol separator (U+00B7), used in ·funcName(SB).
|
||||
const middleDot = '\u00B7'
|
||||
|
||||
// Lexer scans a source string one token at a time.
|
||||
type Lexer struct {
|
||||
src []rune
|
||||
off []int // off[i] is the byte offset of src[i]; off[len(src)] is len(bytes)
|
||||
i int // index of the current rune
|
||||
line int // one-based line of src[i]
|
||||
col int // one-based rune column of src[i]
|
||||
}
|
||||
|
||||
// New returns a Lexer over src.
|
||||
func New(src string) *Lexer {
|
||||
runes := []rune(src)
|
||||
off := make([]int, len(runes)+1)
|
||||
b := 0
|
||||
for i, r := range runes {
|
||||
off[i] = b
|
||||
b += utf8.RuneLen(r)
|
||||
}
|
||||
off[len(runes)] = b
|
||||
return &Lexer{src: runes, off: off, line: 1, col: 1}
|
||||
}
|
||||
|
||||
// Tokenize scans src fully and returns every token up to and including the
|
||||
// trailing EOF token.
|
||||
func Tokenize(src string) []token.Token {
|
||||
l := New(src)
|
||||
var out []token.Token
|
||||
for {
|
||||
tok := l.Next()
|
||||
out = append(out, tok)
|
||||
if tok.Kind == token.EOF {
|
||||
return out
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// cur returns the current rune, or 0 at end of input.
|
||||
func (l *Lexer) cur() rune {
|
||||
if l.i >= len(l.src) {
|
||||
return 0
|
||||
}
|
||||
return l.src[l.i]
|
||||
}
|
||||
|
||||
// peek returns the rune k positions ahead, or 0 past the end.
|
||||
func (l *Lexer) peek(k int) rune {
|
||||
if l.i+k >= len(l.src) || l.i+k < 0 {
|
||||
return 0
|
||||
}
|
||||
return l.src[l.i+k]
|
||||
}
|
||||
|
||||
// pos snapshots the current source position.
|
||||
func (l *Lexer) pos() token.Position {
|
||||
return token.Position{Offset: l.off[l.i], Line: l.line, Column: l.col}
|
||||
}
|
||||
|
||||
// advance consumes one rune, updating line and column bookkeeping.
|
||||
func (l *Lexer) advance() {
|
||||
if l.i >= len(l.src) {
|
||||
return
|
||||
}
|
||||
if l.src[l.i] == '\n' {
|
||||
l.line++
|
||||
l.col = 1
|
||||
} else {
|
||||
l.col++
|
||||
}
|
||||
l.i++
|
||||
}
|
||||
|
||||
// make builds a token of the given kind spanning [start, current position).
|
||||
func (l *Lexer) make(kind token.Kind, start token.Position, text string) token.Token {
|
||||
return token.Token{Kind: kind, Text: text, Pos: start, End: l.pos()}
|
||||
}
|
||||
|
||||
// Next returns the next token, skipping spaces and tabs. Newlines are
|
||||
// significant and returned as Newline tokens so the parser can treat the
|
||||
// stream line by line.
|
||||
func (l *Lexer) Next() token.Token {
|
||||
for {
|
||||
// Skip horizontal whitespace. A backslash immediately before a newline
|
||||
// is a C-preprocessor line continuation (used by #define macros in the
|
||||
// runtime .s files): splice the lines together by consuming both, so
|
||||
// the whole macro becomes one logical line that the parser treats as an
|
||||
// opaque preprocessor directive.
|
||||
for {
|
||||
c := l.cur()
|
||||
if c == ' ' || c == '\t' || c == '\r' {
|
||||
l.advance()
|
||||
continue
|
||||
}
|
||||
if c == '\\' && (l.peek(1) == '\n' || l.peek(1) == '\r') {
|
||||
l.advance() // backslash
|
||||
if l.cur() == '\r' {
|
||||
l.advance()
|
||||
}
|
||||
if l.cur() == '\n' {
|
||||
l.advance()
|
||||
}
|
||||
continue
|
||||
}
|
||||
break
|
||||
}
|
||||
|
||||
start := l.pos()
|
||||
r := l.cur()
|
||||
|
||||
switch {
|
||||
case r == 0:
|
||||
return l.make(token.EOF, start, "")
|
||||
|
||||
case r == '\n':
|
||||
l.advance()
|
||||
return l.make(token.Newline, start, "\n")
|
||||
|
||||
case r == '/':
|
||||
switch l.peek(1) {
|
||||
case '/':
|
||||
return l.lineComment(start)
|
||||
case '*':
|
||||
return l.blockComment(start)
|
||||
default:
|
||||
l.advance()
|
||||
return l.make(token.Slash, start, "/")
|
||||
}
|
||||
|
||||
case r == '"':
|
||||
return l.string(start)
|
||||
|
||||
case r == '\'':
|
||||
return l.runeLit(start)
|
||||
|
||||
case isIdentStart(r):
|
||||
return l.ident(start)
|
||||
|
||||
case isDigit(r):
|
||||
return l.number(start)
|
||||
|
||||
default:
|
||||
return l.punct(start)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// lineComment consumes a // comment up to, but not including, the newline.
|
||||
func (l *Lexer) lineComment(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
for l.cur() != 0 && l.cur() != '\n' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Comment, start, b.String())
|
||||
}
|
||||
|
||||
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
|
||||
func (l *Lexer) blockComment(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
b.WriteRune(l.cur()) // '/'
|
||||
l.advance()
|
||||
b.WriteRune(l.cur()) // '*'
|
||||
l.advance()
|
||||
for l.cur() != 0 {
|
||||
if l.cur() == '*' && l.peek(1) == '/' {
|
||||
b.WriteString("*/")
|
||||
l.advance()
|
||||
l.advance()
|
||||
break
|
||||
}
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Comment, start, b.String())
|
||||
}
|
||||
|
||||
// string consumes a double-quoted string literal, honouring backslash escapes.
|
||||
func (l *Lexer) string(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
b.WriteRune('"')
|
||||
l.advance() // opening quote
|
||||
for l.cur() != 0 && l.cur() != '\n' {
|
||||
r := l.cur()
|
||||
b.WriteRune(r)
|
||||
l.advance()
|
||||
if r == '\\' {
|
||||
if l.cur() != 0 && l.cur() != '\n' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
continue
|
||||
}
|
||||
if r == '"' {
|
||||
return l.make(token.String, start, b.String())
|
||||
}
|
||||
}
|
||||
// Unterminated string: return what we have rather than failing.
|
||||
return l.make(token.String, start, b.String())
|
||||
}
|
||||
|
||||
// runeLit consumes a single-quoted rune literal such as 'a' or '\n'.
|
||||
func (l *Lexer) runeLit(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
b.WriteRune('\'')
|
||||
l.advance() // opening quote
|
||||
for l.cur() != 0 && l.cur() != '\n' {
|
||||
r := l.cur()
|
||||
b.WriteRune(r)
|
||||
l.advance()
|
||||
if r == '\\' {
|
||||
if l.cur() != 0 && l.cur() != '\n' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
continue
|
||||
}
|
||||
if r == '\'' {
|
||||
return l.make(token.Rune, start, b.String())
|
||||
}
|
||||
}
|
||||
return l.make(token.Rune, start, b.String())
|
||||
}
|
||||
|
||||
// ident consumes an identifier: letters, digits, '_', '.', and the middle dot.
|
||||
func (l *Lexer) ident(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
for isIdentChar(l.cur()) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Ident, start, b.String())
|
||||
}
|
||||
|
||||
// number consumes an integer or floating-point literal. The sign is never
|
||||
// part of the literal; it is scanned separately as a Minus or Plus token.
|
||||
func (l *Lexer) number(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
// Base prefixes.
|
||||
if l.cur() == '0' && (l.peek(1) == 'x' || l.peek(1) == 'X') {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
for isHexDigit(l.cur()) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Number, start, b.String())
|
||||
}
|
||||
if l.cur() == '0' && (l.peek(1) == 'b' || l.peek(1) == 'B') {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
for l.cur() == '0' || l.cur() == '1' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Number, start, b.String())
|
||||
}
|
||||
if l.cur() == '0' && (l.peek(1) == 'o' || l.peek(1) == 'O') {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
for l.cur() >= '0' && l.cur() <= '7' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Number, start, b.String())
|
||||
}
|
||||
// Decimal, possibly fractional and/or with an exponent.
|
||||
for isDigit(l.cur()) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
if l.cur() == '.' && isDigit(l.peek(1)) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
for isDigit(l.cur()) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
}
|
||||
if l.cur() == 'e' || l.cur() == 'E' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
if l.cur() == '+' || l.cur() == '-' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
for isDigit(l.cur()) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
}
|
||||
return l.make(token.Number, start, b.String())
|
||||
}
|
||||
|
||||
// punct consumes a single punctuation or operator token, handling the
|
||||
// multi-character operators <<, >> and ->.
|
||||
func (l *Lexer) punct(start token.Position) token.Token {
|
||||
r := l.cur()
|
||||
switch r {
|
||||
case '(':
|
||||
l.advance()
|
||||
return l.make(token.LParen, start, "(")
|
||||
case ')':
|
||||
l.advance()
|
||||
return l.make(token.RParen, start, ")")
|
||||
case ',':
|
||||
l.advance()
|
||||
return l.make(token.Comma, start, ",")
|
||||
case '+':
|
||||
l.advance()
|
||||
return l.make(token.Plus, start, "+")
|
||||
case '-':
|
||||
if l.peek(1) == '>' {
|
||||
l.advance()
|
||||
l.advance()
|
||||
return l.make(token.Arrow, start, "->")
|
||||
}
|
||||
l.advance()
|
||||
return l.make(token.Minus, start, "-")
|
||||
case '*':
|
||||
l.advance()
|
||||
return l.make(token.Star, start, "*")
|
||||
case ':':
|
||||
l.advance()
|
||||
return l.make(token.Colon, start, ":")
|
||||
case '$':
|
||||
l.advance()
|
||||
return l.make(token.Dollar, start, "$")
|
||||
case '<':
|
||||
if l.peek(1) == '<' {
|
||||
l.advance()
|
||||
l.advance()
|
||||
return l.make(token.LShift, start, "<<")
|
||||
}
|
||||
l.advance()
|
||||
return l.make(token.LAngle, start, "<")
|
||||
case '>':
|
||||
if l.peek(1) == '>' {
|
||||
l.advance()
|
||||
l.advance()
|
||||
return l.make(token.RShift, start, ">>")
|
||||
}
|
||||
l.advance()
|
||||
return l.make(token.RAngle, start, ">")
|
||||
case '@':
|
||||
l.advance()
|
||||
return l.make(token.At, start, "@")
|
||||
case '#':
|
||||
l.advance()
|
||||
return l.make(token.Hash, start, "#")
|
||||
default:
|
||||
// Unknown rune: emit it as Illegal and move on.
|
||||
l.advance()
|
||||
return l.make(token.Illegal, start, string(r))
|
||||
}
|
||||
}
|
||||
|
||||
func isDigit(r rune) bool { return r >= '0' && r <= '9' }
|
||||
|
||||
func isHexDigit(r rune) bool {
|
||||
return isDigit(r) || (r >= 'a' && r <= 'f') || (r >= 'A' && r <= 'F')
|
||||
}
|
||||
|
||||
func isIdentStart(r rune) bool {
|
||||
return r == '_' || r == middleDot || unicode.IsLetter(r)
|
||||
}
|
||||
|
||||
func isIdentChar(r rune) bool {
|
||||
return isIdentStart(r) || isDigit(r) || r == '.'
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lexer
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// kinds tokenizes src and returns the kind sequence, dropping Newline/EOF.
|
||||
func kinds(src string) []token.Kind {
|
||||
var out []token.Kind
|
||||
for _, t := range Tokenize(src) {
|
||||
if t.Kind == token.Newline || t.Kind == token.EOF {
|
||||
continue
|
||||
}
|
||||
out = append(out, t.Kind)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// texts tokenizes src and returns the literal text of each significant token.
|
||||
func texts(src string) []string {
|
||||
var out []string
|
||||
for _, t := range Tokenize(src) {
|
||||
if t.Kind == token.Newline || t.Kind == token.EOF {
|
||||
continue
|
||||
}
|
||||
out = append(out, t.Text)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func eq[T comparable](t *testing.T, got, want []T) {
|
||||
t.Helper()
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("length mismatch:\n got %v\n want %v", got, want)
|
||||
}
|
||||
for i := range got {
|
||||
if got[i] != want[i] {
|
||||
t.Fatalf("index %d:\n got %v\n want %v", i, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextDirective(t *testing.T) {
|
||||
eq(t, texts("TEXT ·analyzeO1RangeAVX2(SB), NOSPLIT, $0-65"),
|
||||
[]string{"TEXT", "·analyzeO1RangeAVX2", "(", "SB", ")", ",", "NOSPLIT", ",", "$", "0", "-", "65"})
|
||||
}
|
||||
|
||||
func TestDataAndGlobl(t *testing.T) {
|
||||
eq(t, texts("GLOBL ·idx16(SB), RODATA, $64"),
|
||||
[]string{"GLOBL", "·idx16", "(", "SB", ")", ",", "RODATA", ",", "$", "64"})
|
||||
eq(t, texts("DATA ·idx16+0(SB)/4, $1"),
|
||||
[]string{"DATA", "·idx16", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "1"})
|
||||
}
|
||||
|
||||
func TestStaticSymbol(t *testing.T) {
|
||||
// mask24<> is a file-local symbol; <> must lex as two angle tokens.
|
||||
eq(t, texts("GLOBL mask24<>(SB), RODATA, $16"),
|
||||
[]string{"GLOBL", "mask24", "<", ">", "(", "SB", ")", ",", "RODATA", ",", "$", "16"})
|
||||
}
|
||||
|
||||
func TestNegativeImmediate(t *testing.T) {
|
||||
eq(t, texts("ANDQ $-8, R10"),
|
||||
[]string{"ANDQ", "$", "-", "8", ",", "R10"})
|
||||
}
|
||||
|
||||
func TestHexImmediate(t *testing.T) {
|
||||
eq(t, texts("DATA mask24<>+0(SB)/4, $0x80020100"),
|
||||
[]string{"DATA", "mask24", "<", ">", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "0x80020100"})
|
||||
}
|
||||
|
||||
func TestMemoryAddressing(t *testing.T) {
|
||||
eq(t, texts("LEAQ (SI)(BX*4), R9"),
|
||||
[]string{"LEAQ", "(", "SI", ")", "(", "BX", "*", "4", ")", ",", "R9"})
|
||||
eq(t, texts("VMOVDQU32 Z0, 4(SI)(AX*1)"),
|
||||
[]string{"VMOVDQU32", "Z0", ",", "4", "(", "SI", ")", "(", "AX", "*", "1", ")"})
|
||||
}
|
||||
|
||||
func TestLabelAndComment(t *testing.T) {
|
||||
eq(t, kinds("vec1:\n\tJMP vec1 // loop"),
|
||||
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
|
||||
}
|
||||
|
||||
func TestAVX512Mnemonics(t *testing.T) {
|
||||
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
|
||||
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
|
||||
eq(t, texts("KTESTW K1, K1"),
|
||||
[]string{"KTESTW", "K1", ",", "K1"})
|
||||
}
|
||||
|
||||
func TestArm64Shifts(t *testing.T) {
|
||||
eq(t, texts("ADD R0<<2, R1, R2"),
|
||||
[]string{"ADD", "R0", "<<", "2", ",", "R1", ",", "R2"})
|
||||
eq(t, texts("MOVD R3->4, R5"),
|
||||
[]string{"MOVD", "R3", "->", "4", ",", "R5"})
|
||||
}
|
||||
|
||||
func TestInclude(t *testing.T) {
|
||||
eq(t, texts(`#include "textflag.h"`),
|
||||
[]string{"#", "include", `"textflag.h"`})
|
||||
}
|
||||
|
||||
func TestPositions(t *testing.T) {
|
||||
toks := Tokenize("MOVQ AX, BX\nRET")
|
||||
// Find RET and check it landed on line 2.
|
||||
var ret token.Token
|
||||
for _, tok := range toks {
|
||||
if tok.Text == "RET" {
|
||||
ret = tok
|
||||
}
|
||||
}
|
||||
if ret.Pos.Line != 2 || ret.Pos.Column != 1 {
|
||||
t.Fatalf("RET position = %v, want 2:1", ret.Pos)
|
||||
}
|
||||
}
|
||||
|
||||
func TestIllegalNeverPanics(t *testing.T) {
|
||||
// A stray backtick and NUL-ish garbage must not crash the scanner.
|
||||
toks := Tokenize("MOVQ ` , \x01 AX")
|
||||
if len(toks) == 0 {
|
||||
t.Fatal("expected tokens")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBlockComment(t *testing.T) {
|
||||
eq(t, kinds("MOVQ /* inline */ AX"),
|
||||
[]token.Kind{token.Ident, token.Comment, token.Ident})
|
||||
// An unterminated block comment is tolerated.
|
||||
toks := Tokenize("MOVQ /* never closed")
|
||||
if toks[len(toks)-2].Kind != token.Comment {
|
||||
t.Errorf("expected a comment token, got %v", toks)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRuneLiteral(t *testing.T) {
|
||||
eq(t, texts("MOVL $'a', AX"),
|
||||
[]string{"MOVL", "$", "'a'", ",", "AX"})
|
||||
}
|
||||
|
||||
func TestFloatAndBases(t *testing.T) {
|
||||
eq(t, texts("$1.5"), []string{"$", "1.5"})
|
||||
eq(t, texts("$0b1010"), []string{"$", "0b1010"})
|
||||
eq(t, texts("$0o755"), []string{"$", "0o755"})
|
||||
eq(t, texts("$1e3"), []string{"$", "1e3"})
|
||||
}
|
||||
|
||||
func TestOperatorVariants(t *testing.T) {
|
||||
eq(t, texts("R0>>2"), []string{"R0", ">>", "2"})
|
||||
eq(t, texts("@>"), []string{"@", ">"})
|
||||
eq(t, texts("a/b"), []string{"a", "/", "b"})
|
||||
}
|
||||
+175
@@ -0,0 +1,175 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lint
|
||||
|
||||
import (
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// abiExpectedArgSize computes the argument-area size (parameters plus results,
|
||||
// laid out with Go's alignment rules on a 64-bit target) from the `// func …`
|
||||
// signature in a TEXT function's doc comment. It returns ok=false when there
|
||||
// is no parseable signature or it uses a type whose size cannot be determined
|
||||
// (a named type), so the caller can skip the check rather than guess.
|
||||
//
|
||||
// The signature is parsed with the standard library's Go parser, so every
|
||||
// legal signature form (shared names such as `left, right []int32`, nested
|
||||
// pointers, arrays, structs) is handled correctly.
|
||||
func abiExpectedArgSize(doc string) (int64, bool) {
|
||||
sig := signatureLine(doc)
|
||||
if sig == "" {
|
||||
return 0, false
|
||||
}
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "sig.go", "package p\n"+sig+" {}\n", 0)
|
||||
if err != nil || len(f.Decls) == 0 {
|
||||
return 0, false
|
||||
}
|
||||
fn, ok := f.Decls[0].(*ast.FuncDecl)
|
||||
if !ok || fn.Type == nil {
|
||||
return 0, false
|
||||
}
|
||||
return signatureSize(fn.Type.Params, fn.Type.Results)
|
||||
}
|
||||
|
||||
// signatureLine returns the first `func …` line from a doc comment, trimmed.
|
||||
func signatureLine(doc string) string {
|
||||
for _, line := range strings.Split(doc, "\n") {
|
||||
if t := strings.TrimSpace(line); strings.HasPrefix(t, "func ") {
|
||||
return t
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// signatureSize lays out the parameters and results and returns the total byte
|
||||
// size of the argument area, matching Go's ABI0 stack layout: parameters are
|
||||
// laid out first, then the result area begins on a word (8-byte) boundary.
|
||||
func signatureSize(params, results *ast.FieldList) (int64, bool) {
|
||||
paramsSize, _, ok := fieldsSizeAlign(params)
|
||||
if !ok {
|
||||
return 0, false
|
||||
}
|
||||
resultsSize, _, ok := fieldsSizeAlign(results)
|
||||
if !ok {
|
||||
return 0, false
|
||||
}
|
||||
// With no results the argument area is exactly the parameter size. When
|
||||
// there are results, the result area begins on a word (8-byte) boundary
|
||||
// after the parameters (Go's ABI0 stack layout).
|
||||
if resultsSize == 0 {
|
||||
return int64(paramsSize), true
|
||||
}
|
||||
return int64(alignUp(paramsSize, 8) + resultsSize), true
|
||||
}
|
||||
|
||||
// fieldsSizeAlign lays out a field list sequentially (each field aligned to its
|
||||
// own alignment) and returns the total size and the maximum field alignment.
|
||||
func fieldsSizeAlign(list *ast.FieldList) (size, align int, ok bool) {
|
||||
if list == nil {
|
||||
return 0, 1, true
|
||||
}
|
||||
offset, maxAlign := 0, 1
|
||||
for _, field := range list.List {
|
||||
es, ea, fieldOK := typeSizeAlign(field.Type)
|
||||
if !fieldOK {
|
||||
return 0, 0, false
|
||||
}
|
||||
n := len(field.Names)
|
||||
if n == 0 {
|
||||
n = 1
|
||||
}
|
||||
for i := 0; i < n; i++ {
|
||||
offset = alignUp(offset, ea)
|
||||
offset += es
|
||||
}
|
||||
if ea > maxAlign {
|
||||
maxAlign = ea
|
||||
}
|
||||
}
|
||||
return offset, maxAlign, true
|
||||
}
|
||||
|
||||
// basicSizes maps built-in type names to {size, align} on a 64-bit target.
|
||||
var basicSizes = map[string][2]int{
|
||||
"bool": {1, 1}, "byte": {1, 1}, "int8": {1, 1}, "uint8": {1, 1},
|
||||
"int16": {2, 2}, "uint16": {2, 2},
|
||||
"int32": {4, 4}, "uint32": {4, 4}, "float32": {4, 4},
|
||||
"int": {8, 8}, "int64": {8, 8}, "uint": {8, 8}, "uint64": {8, 8},
|
||||
"uintptr": {8, 8}, "float64": {8, 8},
|
||||
"complex64": {8, 4}, "complex128": {16, 8},
|
||||
"string": {16, 8}, "any": {16, 8}, "error": {16, 8},
|
||||
}
|
||||
|
||||
// typeSizeAlign returns the size and alignment in bytes of a type expression,
|
||||
// or ok=false when the size cannot be determined (an unknown named type).
|
||||
func typeSizeAlign(e ast.Expr) (size, align int, ok bool) {
|
||||
switch t := e.(type) {
|
||||
case *ast.Ident:
|
||||
if sa, found := basicSizes[t.Name]; found {
|
||||
return sa[0], sa[1], true
|
||||
}
|
||||
return 0, 0, false // named type of unknown size
|
||||
case *ast.SelectorExpr:
|
||||
if pkg, isIdent := t.X.(*ast.Ident); isIdent && pkg.Name == "unsafe" && t.Sel.Name == "Pointer" {
|
||||
return 8, 8, true
|
||||
}
|
||||
return 0, 0, false
|
||||
case *ast.ParenExpr:
|
||||
return typeSizeAlign(t.X)
|
||||
case *ast.StarExpr:
|
||||
return 8, 8, true // pointer
|
||||
case *ast.MapType, *ast.ChanType, *ast.FuncType:
|
||||
return 8, 8, true // map / chan / func are pointer-sized
|
||||
case *ast.InterfaceType:
|
||||
return 16, 8, true
|
||||
case *ast.Ellipsis:
|
||||
return 24, 8, true // variadic parameter is a slice
|
||||
case *ast.ArrayType:
|
||||
if t.Len == nil {
|
||||
return 24, 8, true // slice header
|
||||
}
|
||||
n, lenOK := arrayLength(t.Len)
|
||||
es, ea, elemOK := typeSizeAlign(t.Elt)
|
||||
if !lenOK || !elemOK {
|
||||
return 0, 0, false
|
||||
}
|
||||
return n * es, ea, true
|
||||
case *ast.StructType:
|
||||
return structSizeAlign(t.Fields)
|
||||
}
|
||||
return 0, 0, false
|
||||
}
|
||||
|
||||
// structSizeAlign lays out a struct's fields and returns its size (rounded up
|
||||
// to its alignment) and alignment.
|
||||
func structSizeAlign(fields *ast.FieldList) (size, align int, ok bool) {
|
||||
size, align, ok = fieldsSizeAlign(fields)
|
||||
if !ok {
|
||||
return 0, 0, false
|
||||
}
|
||||
return alignUp(size, align), align, true
|
||||
}
|
||||
|
||||
// arrayLength evaluates a constant array-length expression (a literal, for the
|
||||
// kernels this toolkit targets).
|
||||
func arrayLength(e ast.Expr) (int, bool) {
|
||||
if lit, ok := e.(*ast.BasicLit); ok && lit.Kind == token.INT {
|
||||
if v, err := strconv.Atoi(lit.Value); err == nil {
|
||||
return v, true
|
||||
}
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
func alignUp(offset, align int) int {
|
||||
if align <= 1 {
|
||||
return offset
|
||||
}
|
||||
return (offset + align - 1) &^ (align - 1)
|
||||
}
|
||||
@@ -0,0 +1,129 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lint
|
||||
|
||||
import "testing"
|
||||
|
||||
// TestABIExpectedArgSize checks the Go ABI0 argument-area size computation
|
||||
// against hand-verified signatures (the same layouts the go-flac kernels use).
|
||||
func TestABIExpectedArgSize(t *testing.T) {
|
||||
cases := []struct {
|
||||
sig string
|
||||
want int64
|
||||
}{
|
||||
{"func f()", 0},
|
||||
{"func f(a int, b int)", 16},
|
||||
{"func f(a int32)", 4}, // no results: no word-boundary padding
|
||||
{"func f(a int32) (r int32)", 12}, // results begin on a word boundary: 4 -> 8, +4
|
||||
{"func f(a int) int", 16},
|
||||
{"func f(s []int32, p *[32]uint16) (x uint64, ok bool)", 41},
|
||||
{"func f(left, right []int32, sums *[4]uint64)", 56},
|
||||
{"func f(src []byte, dst []int32)", 48},
|
||||
{"func f(a bool, b int64)", 16}, // bool at 0, int64 aligned to 8
|
||||
{"func f(x struct{ a int32; b int64 })", 16},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, ok := abiExpectedArgSize(c.sig)
|
||||
if !ok {
|
||||
t.Errorf("%s: could not compute size", c.sig)
|
||||
continue
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("%s: size = %d, want %d", c.sig, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestABIUnknownTypeSkipped(t *testing.T) {
|
||||
// A bare named type of unknown size must abort the check rather than guess.
|
||||
// (A *pointer* to a named type is still 8 bytes and is fine.)
|
||||
if _, ok := abiExpectedArgSize("func f(s Stream)"); ok {
|
||||
t.Error("bare named type should make the size undecidable")
|
||||
}
|
||||
if _, ok := abiExpectedArgSize("func f(s *Stream)"); !ok {
|
||||
t.Error("pointer to a named type is decidable (8 bytes)")
|
||||
}
|
||||
}
|
||||
|
||||
// TestABIArgSizeRule checks the lint rule end to end.
|
||||
func TestABIArgSizeRule(t *testing.T) {
|
||||
// Matching: the declared arg size agrees with the signature.
|
||||
clean := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"// func f(a int, b int)\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0-16\n"+
|
||||
"\tMOVQ a+0(FP), AX\n"+
|
||||
"\tRET\n")
|
||||
if codes(clean)[CodeABIArgSize] != 0 {
|
||||
t.Fatalf("matching arg size should not warn: %+v", clean)
|
||||
}
|
||||
|
||||
// Mismatching: declared 8, signature implies 16.
|
||||
bad := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"// func f(a int, b int)\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0-8\n"+
|
||||
"\tMOVQ a+0(FP), AX\n"+
|
||||
"\tRET\n")
|
||||
if codes(bad)[CodeABIArgSize] != 1 {
|
||||
t.Fatalf("mismatching arg size should warn once: %+v", bad)
|
||||
}
|
||||
}
|
||||
|
||||
// TestABIArgSizeSkipsRegisterABI verifies the check does not fire for functions
|
||||
// that declare a zero arg area (register ABI) or never touch FP.
|
||||
func TestABIArgSizeSkipsRegisterABI(t *testing.T) {
|
||||
diags := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"// func f(a int, b int)\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0-0\n"+
|
||||
"\tMOVQ AX, BX\n"+
|
||||
"\tRET\n")
|
||||
if codes(diags)[CodeABIArgSize] != 0 {
|
||||
t.Fatalf("register-ABI function must not be checked: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUnreachableCode exercises the dead-code detection and its guard rails.
|
||||
func TestUnreachableCode(t *testing.T) {
|
||||
// Code after a RET is unreachable.
|
||||
dead := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tRET\n"+
|
||||
"\tMOVQ AX, BX\n")
|
||||
if codes(dead)[CodeUnreachable] != 1 {
|
||||
t.Fatalf("code after RET should be unreachable: %+v", dead)
|
||||
}
|
||||
|
||||
// A label after the RET makes the following code reachable again.
|
||||
live := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tRET\n"+
|
||||
"again:\n"+
|
||||
"\tJMP again\n")
|
||||
if codes(live)[CodeUnreachable] != 0 {
|
||||
t.Fatalf("code after a label is reachable: %+v", live)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUnreachableGuards verifies the analysis is suppressed where reachability
|
||||
// cannot be determined statically.
|
||||
func TestUnreachableGuards(t *testing.T) {
|
||||
// A PC-relative jump defeats the analysis for the whole function.
|
||||
pcrel := lintSrcArch(t, "f_amd64.s", "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tJCC 2(PC)\n"+
|
||||
"\tRET\n"+
|
||||
"\tMOVQ AX, BX\n")
|
||||
if codes(pcrel)[CodeUnreachable] != 0 {
|
||||
t.Fatalf("PC-relative functions must be skipped: %+v", pcrel)
|
||||
}
|
||||
|
||||
// A register-indirect branch (riscv JALR) defeats the analysis too.
|
||||
indirect := lintSrcArch(t, "f_riscv64.s", "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tJALR X1, X5\n"+
|
||||
"\tRET\n"+
|
||||
"\tMOV X1, X2\n")
|
||||
if codes(indirect)[CodeUnreachable] != 0 {
|
||||
t.Fatalf("indirect-branch functions must be skipped: %+v", indirect)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lint
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
// checkFuncdata validates the structure of FUNCDATA and PCDATA directives,
|
||||
// which carry the GC stack-map information. The checks are deliberately
|
||||
// shallow — they confirm the operands are well formed and that a literal index
|
||||
// is within the small range the runtime uses — and never try to interpret a
|
||||
// named index constant such as $PCDATA_StackMapIndex.
|
||||
func checkFuncdata(t *ast.Text, cfg Config) []Diagnostic {
|
||||
var out []Diagnostic
|
||||
for _, s := range t.Body {
|
||||
in, ok := s.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||
case "FUNCDATA":
|
||||
out = append(out, checkFunCDATA(in, cfg)...)
|
||||
case "PCDATA":
|
||||
out = append(out, checkPCDATA(in, cfg)...)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// checkFunCDATA validates `FUNCDATA $index, symbol(SB)`.
|
||||
func checkFunCDATA(in *ast.Instr, cfg Config) []Diagnostic {
|
||||
if cfg.Disable[CodeFuncdata] {
|
||||
return nil
|
||||
}
|
||||
var out []Diagnostic
|
||||
if len(in.Operands) != 2 {
|
||||
return []Diagnostic{{
|
||||
Pos: in.Mnemonic.Pos, End: in.Mnemonic.End, Severity: Warning, Code: CodeFuncdata,
|
||||
Message: fmt.Sprintf("FUNCDATA expects 2 operands (index, symbol), got %d", len(in.Operands)),
|
||||
}}
|
||||
}
|
||||
out = append(out, checkIndex(in.Operands[0], "FUNCDATA")...)
|
||||
if sym := in.Operands[1].Addr.Sym; in.Operands[1].Kind != ast.OpAddr || sym == nil {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: in.Operands[1].Pos, Severity: Warning, Code: CodeFuncdata,
|
||||
Message: "FUNCDATA second operand must be a symbol reference",
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// checkPCDATA validates `PCDATA $index, $value`.
|
||||
func checkPCDATA(in *ast.Instr, cfg Config) []Diagnostic {
|
||||
if cfg.Disable[CodeFuncdata] {
|
||||
return nil
|
||||
}
|
||||
if len(in.Operands) != 2 {
|
||||
return []Diagnostic{{
|
||||
Pos: in.Mnemonic.Pos, End: in.Mnemonic.End, Severity: Warning, Code: CodeFuncdata,
|
||||
Message: fmt.Sprintf("PCDATA expects 2 operands (index, value), got %d", len(in.Operands)),
|
||||
}}
|
||||
}
|
||||
var out []Diagnostic
|
||||
out = append(out, checkIndex(in.Operands[0], "PCDATA")...)
|
||||
if in.Operands[1].Kind != ast.OpImmediate {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: in.Operands[1].Pos, Severity: Warning, Code: CodeFuncdata,
|
||||
Message: "PCDATA value must be an immediate",
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// checkIndex validates an immediate index operand. A literal index must lie in
|
||||
// the small range the runtime uses; a named constant (e.g. $PCDATA_StackMapIndex)
|
||||
// cannot be evaluated and is accepted without a range check.
|
||||
func checkIndex(op *ast.Operand, directive string) []Diagnostic {
|
||||
if op.Kind != ast.OpImmediate {
|
||||
return []Diagnostic{{
|
||||
Pos: op.Pos, Severity: Warning, Code: CodeFuncdata,
|
||||
Message: directive + " index must be an immediate",
|
||||
}}
|
||||
}
|
||||
if op.Imm.HasVal && (op.Imm.Val < 0 || op.Imm.Val > 10) {
|
||||
return []Diagnostic{{
|
||||
Pos: op.Pos, Severity: Warning, Code: CodeFuncdata,
|
||||
Message: fmt.Sprintf("%s index %d is outside the valid range 0–10", directive, op.Imm.Val),
|
||||
}}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lint
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
// TestGoRuntimeCorpus parses and lints every runtime .s file the local Go
|
||||
// toolchain ships for all four supported architectures. This is the
|
||||
// real-world regression net: it exercises the full breadth of each
|
||||
// architecture's syntax (macros, addressing modes, branch aliases) against
|
||||
// production assembly. It is skipped when the toolchain source is absent.
|
||||
//
|
||||
// The bar is zero parse errors and zero error-severity diagnostics — i.e. no
|
||||
// false "unknown instruction" / "undefined label" findings on code the real
|
||||
// assembler accepts. Advisory warnings are reported but not fatal, since they
|
||||
// are heuristics that may legitimately differ across Go versions.
|
||||
func TestGoRuntimeCorpus(t *testing.T) {
|
||||
dir := filepath.Join(runtime.GOROOT(), "src", "runtime")
|
||||
var files []string
|
||||
for _, suffix := range []string{"_amd64.s", "_arm64.s", "_riscv64.s", "_loong64.s"} {
|
||||
matches, _ := filepath.Glob(filepath.Join(dir, "*"+suffix))
|
||||
files = append(files, matches...)
|
||||
}
|
||||
if len(files) == 0 {
|
||||
t.Skip("Go toolchain source (src/runtime/*.s) not present")
|
||||
}
|
||||
|
||||
warnings := 0
|
||||
for _, path := range files {
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("read %s: %v", path, err)
|
||||
}
|
||||
f, errs := parser.Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Errorf("parse %s: %v", filepath.Base(path), errs)
|
||||
continue
|
||||
}
|
||||
diags := File(f, Config{Arch: arch.FromFilename(path)})
|
||||
for _, d := range diags {
|
||||
if d.Severity == Error {
|
||||
t.Errorf("%s:%d: error %s: %s", filepath.Base(path), d.Pos.Line, d.Code, d.Message)
|
||||
} else {
|
||||
warnings++
|
||||
}
|
||||
}
|
||||
}
|
||||
if warnings > 0 {
|
||||
t.Logf("%d advisory warnings across %d files (non-fatal)", warnings, len(files))
|
||||
}
|
||||
}
|
||||
+510
@@ -0,0 +1,510 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package lint runs static checks over a parsed GAsm file. The rules are
|
||||
// deliberately conservative: where a check cannot be certain (for example an
|
||||
// instruction whose operand count varies), it stays silent rather than emit a
|
||||
// false positive. Every diagnostic carries a stable rule code so callers can
|
||||
// disable individual rules.
|
||||
package lint
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// Severity ranks a diagnostic.
|
||||
type Severity int
|
||||
|
||||
// Diagnostic severities, mirroring the language-server protocol ordering.
|
||||
const (
|
||||
Error Severity = iota
|
||||
Warning
|
||||
Information
|
||||
Hint
|
||||
)
|
||||
|
||||
// String returns a lower-case label for the severity.
|
||||
func (s Severity) String() string {
|
||||
switch s {
|
||||
case Error:
|
||||
return "error"
|
||||
case Warning:
|
||||
return "warning"
|
||||
case Information:
|
||||
return "information"
|
||||
default:
|
||||
return "hint"
|
||||
}
|
||||
}
|
||||
|
||||
// Diagnostic is one lint finding.
|
||||
type Diagnostic struct {
|
||||
Pos token.Position
|
||||
End token.Position
|
||||
Severity Severity
|
||||
Code string
|
||||
Message string
|
||||
}
|
||||
|
||||
// Config controls a lint run.
|
||||
type Config struct {
|
||||
// Arch is the target architecture. When it is arch.Unknown the
|
||||
// architecture-specific rules (unknown instruction, operand count) are
|
||||
// skipped because no instruction table can be selected.
|
||||
Arch arch.Arch
|
||||
// Disable lists rule codes to suppress.
|
||||
Disable map[string]bool
|
||||
}
|
||||
|
||||
// Rule codes.
|
||||
const (
|
||||
CodeUnknownInstr = "unknown-instruction"
|
||||
CodeOperandCount = "operand-count"
|
||||
CodeUndefinedLabel = "undefined-label"
|
||||
CodeDuplicateLabel = "duplicate-label"
|
||||
CodeMissingRet = "missing-ret"
|
||||
CodeMissingTextflag = "missing-textflag-include"
|
||||
CodeUnreachable = "unreachable-code"
|
||||
CodeABIArgSize = "abi-argsize"
|
||||
CodeNosplitFrame = "nosplit-frame"
|
||||
CodeRegisterClobber = "register-clobber"
|
||||
CodeFuncdata = "funcdata-pcdata"
|
||||
)
|
||||
|
||||
// pseudoOps are assembler pseudo-operations that are valid instruction-position
|
||||
// tokens but are not machine instructions and so absent from the arch tables.
|
||||
var pseudoOps = map[string]bool{
|
||||
"BYTE": true, "WORD": true, "LONG": true, "QUAD": true, "FLOAT": true,
|
||||
"PCALIGN": true, "FUNCDATA": true, "PCDATA": true, "GO_ARGS": true,
|
||||
}
|
||||
|
||||
// File lints a parsed file and returns the diagnostics in source order.
|
||||
func File(f *ast.File, cfg Config) []Diagnostic {
|
||||
var out []Diagnostic
|
||||
tab := arch.ForArch(cfg.Arch)
|
||||
archKnown := cfg.Arch != arch.Unknown
|
||||
|
||||
hasTextflag := false
|
||||
usesFlags := false
|
||||
var firstFlagPos token.Position
|
||||
|
||||
// Macros (in-file #define, or any #include other than textflag.h, which
|
||||
// only defines flag constants) make label resolution unreliable.
|
||||
macrosInPlay := len(f.Macros) > 0
|
||||
// Preprocessor conditionals (#ifdef …) make control-flow analysis
|
||||
// unreliable, since mutually exclusive branches look sequential.
|
||||
hasConditionals := false
|
||||
for _, d := range f.Decls {
|
||||
switch dd := d.(type) {
|
||||
case *ast.Include:
|
||||
if !strings.Contains(dd.Header.Text, "textflag.h") {
|
||||
macrosInPlay = true
|
||||
}
|
||||
case *ast.Preproc:
|
||||
if isConditionalDirective(dd.Raw) {
|
||||
hasConditionals = true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for _, d := range f.Decls {
|
||||
switch dd := d.(type) {
|
||||
case *ast.Include:
|
||||
if strings.Contains(dd.Header.Text, "textflag.h") {
|
||||
hasTextflag = true
|
||||
}
|
||||
case *ast.Text:
|
||||
out = append(out, lintText(dd, tab, archKnown, cfg, f.Macros, !macrosInPlay, !hasConditionals)...)
|
||||
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
|
||||
usesFlags = true
|
||||
firstFlagPos = dd.Pos()
|
||||
}
|
||||
case *ast.Globl:
|
||||
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
|
||||
usesFlags = true
|
||||
firstFlagPos = dd.Pos()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if !cfg.Disable[CodeMissingTextflag] && usesFlags && !hasTextflag {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: firstFlagPos,
|
||||
Severity: Warning,
|
||||
Code: CodeMissingTextflag,
|
||||
Message: "TEXT/GLOBL flags are used but textflag.h is not #included",
|
||||
})
|
||||
}
|
||||
|
||||
sortDiagnostics(out)
|
||||
return out
|
||||
}
|
||||
|
||||
// lintText lints one TEXT function body. doLabelChecks is false for files
|
||||
// that use macros (an in-file #define or a non-textflag #include): without a
|
||||
// preprocessor we cannot resolve labels that macros define or reference, so the
|
||||
// label and RET heuristics are suppressed there to avoid false positives.
|
||||
func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros map[string]bool, doLabelChecks bool, doUnreachable bool) []Diagnostic {
|
||||
var out []Diagnostic
|
||||
|
||||
defined := map[string]token.Position{}
|
||||
referenced := map[string]token.Position{}
|
||||
hasRet := false
|
||||
lastTerminal := false
|
||||
hasMacro := false
|
||||
instrCount := 0
|
||||
dead := false // inside a region unreachable from above
|
||||
reportedDead := false // the current dead region has already been reported
|
||||
hasPCRel := referencesPC(t) // PC-relative jumps defeat reachability analysis
|
||||
hasIndirect := hasIndirectBranch(t) // register-indirect branches do too
|
||||
// Unreachable-code analysis is only sound in functions whose control flow is
|
||||
// fully label-resolvable: no PC-relative jumps, no register-indirect
|
||||
// branches, and (file-level) no preprocessor conditionals.
|
||||
analyzable := doUnreachable && !hasPCRel && !hasIndirect
|
||||
|
||||
for _, s := range t.Body {
|
||||
switch st := s.(type) {
|
||||
case *ast.Label:
|
||||
name := st.Name.Text
|
||||
if prev, dup := defined[name]; dup {
|
||||
if !cfg.Disable[CodeDuplicateLabel] {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: st.Name.Pos,
|
||||
End: st.Name.End,
|
||||
Severity: Error,
|
||||
Code: CodeDuplicateLabel,
|
||||
Message: fmt.Sprintf("label %q already defined at %s", name, prev),
|
||||
})
|
||||
}
|
||||
} else {
|
||||
defined[name] = st.Name.Pos
|
||||
}
|
||||
// A label is a jump target: code after it is reachable again.
|
||||
dead = false
|
||||
reportedDead = false
|
||||
|
||||
case *ast.Instr:
|
||||
instrCount++
|
||||
mnem := st.Mnemonic.Text
|
||||
upper := strings.ToUpper(mnem)
|
||||
|
||||
// Unreachable code: a real instruction following a RET/UNDEF and
|
||||
// before any label, in a function whose control flow is fully
|
||||
// resolvable. Only RET/UNDEF are treated as terminators here — an
|
||||
// unconditional jump may be one entry of a hand-arranged branch
|
||||
// table (e.g. the generated callback tables), so it is not assumed
|
||||
// to make the following code dead. Pseudo-ops and macro invocations
|
||||
// are never flagged.
|
||||
if analyzable && dead && !reportedDead && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) &&
|
||||
!cfg.Disable[CodeUnreachable] {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: st.Mnemonic.Pos,
|
||||
End: st.Mnemonic.End,
|
||||
Severity: Warning,
|
||||
Code: CodeUnreachable,
|
||||
Message: "unreachable code after terminating instruction",
|
||||
})
|
||||
reportedDead = true
|
||||
}
|
||||
|
||||
// RET never falls through. (UNDEF is a trap/marker rather than a
|
||||
// control-flow terminator: code placed after it is occasionally
|
||||
// deliberate metadata, so it is not treated as making the following
|
||||
// code dead.)
|
||||
if upper == "RET" {
|
||||
dead = true
|
||||
}
|
||||
// A function need not RET if it ends in an unconditional jump (tail
|
||||
// call / loop) or in UNDEF (a deliberate trap that never returns).
|
||||
lastTerminal = isUnconditionalJump(cfg.Arch, upper) || upper == "UNDEF"
|
||||
if isMacroInvocation(mnem, macros) {
|
||||
hasMacro = true
|
||||
}
|
||||
|
||||
if upper == "RET" {
|
||||
hasRet = true
|
||||
}
|
||||
|
||||
if archKnown && !cfg.Disable[CodeUnknownInstr] && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) {
|
||||
if _, ok := tab.Lookup(mnem); !ok {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: st.Mnemonic.Pos,
|
||||
End: st.Mnemonic.End,
|
||||
Severity: Error,
|
||||
Code: CodeUnknownInstr,
|
||||
Message: fmt.Sprintf("unknown %s instruction %q", cfg.Arch, mnem),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) {
|
||||
if in, ok := tab.Lookup(mnem); ok && in.MinOps >= 0 {
|
||||
n := len(st.Operands)
|
||||
if n < in.MinOps || n > in.MaxOps {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: st.Mnemonic.Pos,
|
||||
End: st.Mnemonic.End,
|
||||
Severity: Warning,
|
||||
Code: CodeOperandCount,
|
||||
Message: fmt.Sprintf("%s expects %s, got %d operand(s)",
|
||||
mnem, countRange(in.MinOps, in.MaxOps), n),
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if isJump(cfg.Arch, upper) {
|
||||
for _, op := range st.Operands {
|
||||
if name, pos, ok := localLabelRef(op); ok && !tab.IsRegister(name) && !arch.IsPseudoReg(name) {
|
||||
referenced[name] = pos
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Undefined labels.
|
||||
if doLabelChecks && !cfg.Disable[CodeUndefinedLabel] {
|
||||
for name, pos := range referenced {
|
||||
if _, ok := defined[name]; !ok {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: pos,
|
||||
Severity: Error,
|
||||
Code: CodeUndefinedLabel,
|
||||
Message: fmt.Sprintf("jump to undefined label %q", name),
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Missing RET heuristic. Functions that invoke a macro are skipped: the
|
||||
// macro body (opaque to us) may supply the RET.
|
||||
if doLabelChecks && !cfg.Disable[CodeMissingRet] && instrCount > 0 && !hasRet && !lastTerminal && !hasMacro {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: t.Keyword.Pos,
|
||||
Severity: Warning,
|
||||
Code: CodeMissingRet,
|
||||
Message: fmt.Sprintf("function %q has no RET", t.Name.Name),
|
||||
})
|
||||
}
|
||||
|
||||
// ABI conformance: the argument area declared in the TEXT directive should
|
||||
// match the size computed from the // func signature in the doc comment.
|
||||
// Only applies to stack-argument (ABI0) functions, which reference their
|
||||
// arguments through FP; register-ABI functions declare a zero arg area. Also
|
||||
// skipped when there is no parseable signature or it uses an unknown type.
|
||||
if !cfg.Disable[CodeABIArgSize] {
|
||||
got := int64(0)
|
||||
if t.Args != nil && t.Args.Imm.HasVal {
|
||||
got = t.Args.Imm.Val
|
||||
}
|
||||
// Only meaningful for stack-argument (ABI0) functions: a non-zero
|
||||
// declared arg area that is actually addressed through FP.
|
||||
if got > 0 && usesFPArgs(t) {
|
||||
if want, ok := abiExpectedArgSize(t.Doc); ok {
|
||||
if want != got {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: t.Keyword.Pos,
|
||||
Severity: Warning,
|
||||
Code: CodeABIArgSize,
|
||||
Message: fmt.Sprintf("TEXT declares arg size %d but the // func signature implies %d", got, want),
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Register liveness: a callee-saved register that is written but never
|
||||
// saved and restored is clobbered across the call. The check runs over the
|
||||
// control-flow graph and is skipped for macro-using files, where an opaque
|
||||
// macro may perform the save/restore.
|
||||
if doLabelChecks && archKnown && !cfg.Disable[CodeRegisterClobber] {
|
||||
live := analyzeLiveness(t, cfg.Arch)
|
||||
if clobbered := clobberedCalleeSaved(live, cfg.Arch); len(clobbered) > 0 {
|
||||
out = append(out, Diagnostic{
|
||||
Pos: t.Keyword.Pos,
|
||||
Severity: Warning,
|
||||
Code: CodeRegisterClobber,
|
||||
Message: fmt.Sprintf("callee-saved register(s) %s written but never saved/restored", strings.Join(clobbered, ", ")),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// FUNCDATA / PCDATA structural validation.
|
||||
out = append(out, checkFuncdata(t, cfg)...)
|
||||
|
||||
return out
|
||||
}
|
||||
|
||||
// usesFPArgs reports whether a function references its arguments through the FP
|
||||
// pseudo-register — i.e. it uses the stack-based ABI0 layout, where the
|
||||
// declared argument size must match the signature.
|
||||
func usesFPArgs(t *ast.Text) bool {
|
||||
for _, s := range t.Body {
|
||||
in, ok := s.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
for _, op := range in.Operands {
|
||||
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// referencesPC reports whether a function uses a PC-relative operand (e.g.
|
||||
// `JMP 2(PC)`). Such jumps target a computed offset rather than a label, so
|
||||
// reachability cannot be determined statically and the unreachable-code check
|
||||
// is suppressed for the whole function.
|
||||
func referencesPC(t *ast.Text) bool {
|
||||
for _, s := range t.Body {
|
||||
in, ok := s.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
for _, op := range in.Operands {
|
||||
if strings.Contains(strings.ReplaceAll(op.Raw, " ", ""), "(PC)") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// hasIndirectBranch reports whether a function transfers control through a
|
||||
// register (JALR/JR/JIRL/BR/BLR). Such targets are computed at runtime, so
|
||||
// reachability cannot be determined statically and the unreachable-code check is
|
||||
// suppressed for the whole function.
|
||||
func hasIndirectBranch(t *ast.Text) bool {
|
||||
for _, s := range t.Body {
|
||||
in, ok := s.(*ast.Instr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||
case "JALR", "JR", "JIRL", "BR", "BLR":
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// isMacroInvocation reports whether a mnemonic is a macro invocation rather
|
||||
// than a machine instruction. No Plan 9 mnemonic contains an underscore, so an
|
||||
// underscore is a reliable macro marker (the runtime headers define macros such
|
||||
// as get_tls and NO_LOCAL_POINTERS). Names introduced by an in-file #define
|
||||
// are recognised too (CALLFN, DISPATCH, …). Full macro expansion is out of
|
||||
// scope; this only keeps the linter quiet on invocations it cannot expand.
|
||||
func isMacroInvocation(mnem string, macros map[string]bool) bool {
|
||||
return strings.Contains(mnem, "_") || macros[mnem]
|
||||
}
|
||||
|
||||
// isConditionalDirective reports whether a preprocessor directive (the text
|
||||
// after '#') is a conditional-compilation directive whose branches the parser
|
||||
// cannot resolve.
|
||||
func isConditionalDirective(raw string) bool {
|
||||
fields := strings.Fields(raw)
|
||||
if len(fields) == 0 {
|
||||
return false
|
||||
}
|
||||
switch fields[0] {
|
||||
case "if", "ifdef", "ifndef", "else", "elif", "endif":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// localLabelRef returns the name and position of a bare local-label reference
|
||||
// operand (no pseudo-register, no memory base), if op is one.
|
||||
func localLabelRef(op *ast.Operand) (string, token.Position, bool) {
|
||||
if op == nil || op.Kind != ast.OpAddr || op.Addr.Sym == nil {
|
||||
return "", token.Position{}, false
|
||||
}
|
||||
sym := op.Addr.Sym
|
||||
if sym.Pseudo != "" || op.Addr.Base != "" || sym.Name == "" {
|
||||
return "", token.Position{}, false
|
||||
}
|
||||
return sym.Name, op.Pos, true
|
||||
}
|
||||
|
||||
// riscvBranches and loong64Branches are the conditional-branch mnemonics; they
|
||||
// are listed explicitly rather than matched by a "B" prefix so that bit-manip
|
||||
// instructions (BCLR, BSET, …) are never mistaken for branches.
|
||||
var riscvBranches = map[string]bool{
|
||||
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
|
||||
"BEQZ": true, "BNEZ": true, "BLEZ": true, "BGEZ": true, "BLTZ": true, "BGTZ": true,
|
||||
}
|
||||
|
||||
var loong64Branches = map[string]bool{
|
||||
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
|
||||
"BLEZ": true, "BLTZ": true, "BGEZ": true, "BGTZ": true,
|
||||
}
|
||||
|
||||
// isJump reports whether the mnemonic is any branch.
|
||||
func isJump(a arch.Arch, upper string) bool {
|
||||
switch a {
|
||||
case arch.ARM64:
|
||||
return upper == "CALL" || upper == "BR" || upper == "BLR" || upper == "JMP" ||
|
||||
strings.HasPrefix(upper, "B") ||
|
||||
strings.HasPrefix(upper, "CBZ") || strings.HasPrefix(upper, "CBNZ") ||
|
||||
strings.HasPrefix(upper, "TBZ") || strings.HasPrefix(upper, "TBNZ")
|
||||
case arch.RISCV:
|
||||
return upper == "CALL" || riscvBranches[upper] ||
|
||||
upper == "JMP" || upper == "J" || upper == "JAL" || upper == "JALR" ||
|
||||
upper == "JR" || upper == "BR"
|
||||
case arch.LOONG64:
|
||||
return upper == "CALL" || loong64Branches[upper] ||
|
||||
upper == "JIRL" || upper == "JMP" || upper == "BR"
|
||||
default: // amd64
|
||||
return upper == "CALL" || strings.HasPrefix(upper, "J")
|
||||
}
|
||||
}
|
||||
|
||||
// isUnconditionalJump reports whether the mnemonic is an unconditional branch
|
||||
// (used to suppress the missing-RET heuristic for tail calls and loops).
|
||||
func isUnconditionalJump(a arch.Arch, upper string) bool {
|
||||
switch a {
|
||||
case arch.ARM64:
|
||||
return upper == "B" || upper == "BR" || upper == "JMP"
|
||||
case arch.RISCV:
|
||||
return upper == "JMP" || upper == "J" || upper == "JAL" ||
|
||||
upper == "JALR" || upper == "JR" || upper == "BR"
|
||||
case arch.LOONG64:
|
||||
return upper == "JMP" || upper == "JIRL" || upper == "BR"
|
||||
default:
|
||||
return upper == "JMP"
|
||||
}
|
||||
}
|
||||
|
||||
func countRange(min, max int) string {
|
||||
if min == max {
|
||||
return fmt.Sprintf("%d operand(s)", min)
|
||||
}
|
||||
return fmt.Sprintf("%d–%d operands", min, max)
|
||||
}
|
||||
|
||||
// sortDiagnostics orders diagnostics by line, then column, then code.
|
||||
func sortDiagnostics(d []Diagnostic) {
|
||||
for i := 1; i < len(d); i++ {
|
||||
for j := i; j > 0 && lessDiag(d[j], d[j-1]); j-- {
|
||||
d[j], d[j-1] = d[j-1], d[j]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func lessDiag(a, b Diagnostic) bool {
|
||||
if a.Pos.Line != b.Pos.Line {
|
||||
return a.Pos.Line < b.Pos.Line
|
||||
}
|
||||
if a.Pos.Column != b.Pos.Column {
|
||||
return a.Pos.Column < b.Pos.Column
|
||||
}
|
||||
return a.Code < b.Code
|
||||
}
|
||||
@@ -0,0 +1,242 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lint
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
)
|
||||
|
||||
func lintSrc(t *testing.T, src string) []Diagnostic {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse("test_amd64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
return File(f, Config{Arch: arch.AMD64})
|
||||
}
|
||||
|
||||
// lintSrcArch lints src under the architecture inferred from filename.
|
||||
func lintSrcArch(t *testing.T, filename, src string) []Diagnostic {
|
||||
t.Helper()
|
||||
f, errs := parser.Parse(filename, src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
return File(f, Config{Arch: arch.FromFilename(filename)})
|
||||
}
|
||||
|
||||
func codes(diags []Diagnostic) map[string]int {
|
||||
m := map[string]int{}
|
||||
for _, d := range diags {
|
||||
m[d.Code]++
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
func TestFixtureIsClean(t *testing.T) {
|
||||
src, err := os.ReadFile("../testdata/sample_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse("sample_amd64.s", string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
// The fixture mirrors the go-flac kernels, which use callee-saved registers
|
||||
// (BX, R13) without saving them; the register-clobber audit flags that by
|
||||
// design. This test targets the other rules, so the audit is disabled here
|
||||
// (it is covered by TestRegisterClobber).
|
||||
diags := File(f, Config{Arch: arch.AMD64, Disable: map[string]bool{CodeRegisterClobber: true}})
|
||||
if len(diags) != 0 {
|
||||
t.Fatalf("expected no diagnostics on the fixture, got %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnknownInstruction(t *testing.T) {
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
FOOBAR AX, BX
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeUnknownInstr] != 1 {
|
||||
t.Fatalf("want one unknown-instruction, got %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUndefinedLabel(t *testing.T) {
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
JMP nowhere
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeUndefinedLabel] != 1 {
|
||||
t.Fatalf("want one undefined-label, got %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDuplicateLabel(t *testing.T) {
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
loop:
|
||||
ADDQ $1, AX
|
||||
loop:
|
||||
SUBQ $1, AX
|
||||
JMP loop
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeDuplicateLabel] != 1 {
|
||||
t.Fatalf("want one duplicate-label, got %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMissingRet(t *testing.T) {
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
ADDQ $1, AX
|
||||
`)
|
||||
if codes(diags)[CodeMissingRet] != 1 {
|
||||
t.Fatalf("want one missing-ret, got %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOperandCount(t *testing.T) {
|
||||
// RET takes zero operands; JMP takes exactly one.
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RET AX
|
||||
JMP
|
||||
RET
|
||||
`)
|
||||
c := codes(diags)
|
||||
if c[CodeOperandCount] != 2 {
|
||||
t.Fatalf("want two operand-count findings, got %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMissingTextflag(t *testing.T) {
|
||||
diags := lintSrc(t, `
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeMissingTextflag] != 1 {
|
||||
t.Fatalf("want one missing-textflag-include, got %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDisableRule(t *testing.T) {
|
||||
f, _ := parser.Parse("t_amd64.s", `
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
RET
|
||||
`)
|
||||
diags := File(f, Config{Arch: arch.AMD64, Disable: map[string]bool{CodeMissingTextflag: true}})
|
||||
if len(diags) != 0 {
|
||||
t.Fatalf("disabling the rule should silence it, got %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMacroInvocationSkipped(t *testing.T) {
|
||||
// DISPATCH is defined in-file; get_tls carries an underscore. Neither is a
|
||||
// machine instruction, so both must be ignored by the unknown-instruction
|
||||
// rule rather than flagged.
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
#define DISPATCH CALL ·x(SB)
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
DISPATCH
|
||||
get_tls CX
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeUnknownInstr] != 0 {
|
||||
t.Fatalf("macro invocations must not be flagged: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUndefIsTerminal(t *testing.T) {
|
||||
// A function whose body is UNDEF traps and never returns; it needs no RET.
|
||||
diags := lintSrc(t, `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
UNDEF
|
||||
`)
|
||||
if codes(diags)[CodeMissingRet] != 0 {
|
||||
t.Fatalf("UNDEF should count as terminal: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64BranchAlias(t *testing.T) {
|
||||
diags := lintSrcArch(t, "f_arm64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
B done
|
||||
done:
|
||||
RET
|
||||
`)
|
||||
if len(diags) != 0 {
|
||||
t.Fatalf("arm64 B to a defined label should be clean: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestArm64AddressingSuffix(t *testing.T) {
|
||||
// .W (pre-index) and .P (post-index) suffixes must resolve to the base
|
||||
// instruction.
|
||||
diags := lintSrcArch(t, "f_arm64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
LDP.W (R0), (R1, R2)
|
||||
VST1.P (R3), (R4)
|
||||
RET
|
||||
`)
|
||||
if codes(diags)[CodeUnknownInstr] != 0 {
|
||||
t.Fatalf("suffixed load/store should be recognised: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMacrosInPlaySuppressesLabelRules(t *testing.T) {
|
||||
// Including a non-textflag header means macros may define labels and supply
|
||||
// the RET, so undefined-label and missing-ret are suppressed.
|
||||
diags := lintSrcArch(t, "f_arm64.s", `
|
||||
#include "go_asm.h"
|
||||
TEXT ·f(SB), NOSPLIT, $0
|
||||
JMP RARG0
|
||||
`)
|
||||
if codes(diags)[CodeUndefinedLabel] != 0 || codes(diags)[CodeMissingRet] != 0 {
|
||||
t.Fatalf("label rules should be suppressed in macro files: %+v", diags)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRealGoLibrariesHasNoErrors asserts that the production go-flac kernels
|
||||
// lint free of errors. Skipped when the sibling repository is absent.
|
||||
func TestRealGoLibrariesHasNoErrors(t *testing.T) {
|
||||
matches, _ := filepath.Glob("../../go-libraries/go-*/*.s")
|
||||
if len(matches) == 0 {
|
||||
t.Skip("go-libraries repository not present")
|
||||
}
|
||||
for _, path := range matches {
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse %s: %v", path, errs)
|
||||
}
|
||||
a := arch.FromFilename(path)
|
||||
diags := File(f, Config{Arch: a})
|
||||
for _, d := range diags {
|
||||
if d.Severity == Error {
|
||||
t.Errorf("%s: %s %s: %s", filepath.Base(path), d.Pos, d.Code, d.Message)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,455 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lint
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
// This file implements register liveness by dataflow over a function's
|
||||
// control-flow graph, and the checks built on it. The def/use model is
|
||||
// deliberately conservative: where an instruction's effect is uncertain it is
|
||||
// treated as both a use and a def of its register operands, which can only
|
||||
// suppress a finding, never invent one.
|
||||
|
||||
// regEffect is the register-level effect of one instruction.
|
||||
type regEffect struct {
|
||||
def []string // registers written (killed)
|
||||
use []string // registers read
|
||||
saveGPR []string // callee-saved-style: register written to the stack
|
||||
restGPR []string // register restored from the stack
|
||||
}
|
||||
|
||||
// analyzeLiveness builds the control-flow graph of a function and computes
|
||||
// live-in/live-out register sets by iterative backward dataflow.
|
||||
type liveness struct {
|
||||
blocks []*block
|
||||
liveIn []map[string]bool
|
||||
}
|
||||
|
||||
type block struct {
|
||||
label string // label that begins this block, if any
|
||||
instrs []*ast.Instr
|
||||
succ []int // successor block indices
|
||||
}
|
||||
|
||||
func analyzeLiveness(t *ast.Text, a arch.Arch) *liveness {
|
||||
l := &liveness{}
|
||||
l.buildCFG(t)
|
||||
l.dataflow(a)
|
||||
return l
|
||||
}
|
||||
|
||||
// buildCFG splits the function body into basic blocks and wires up successors.
|
||||
func (l *liveness) buildCFG(t *ast.Text) {
|
||||
labelToBlock := map[string]int{}
|
||||
var cur *block
|
||||
flush := func() {
|
||||
if cur != nil && len(cur.instrs) > 0 {
|
||||
l.blocks = append(l.blocks, cur)
|
||||
}
|
||||
cur = nil
|
||||
}
|
||||
startBlock := func(lbl string) {
|
||||
flush()
|
||||
cur = &block{label: lbl}
|
||||
}
|
||||
|
||||
startBlock("")
|
||||
for _, s := range t.Body {
|
||||
switch st := s.(type) {
|
||||
case *ast.Label:
|
||||
// A label begins a new block and is a jump target.
|
||||
startBlock(st.Name.Text)
|
||||
labelToBlock[st.Name.Text] = len(l.blocks) // index once flushed
|
||||
case *ast.Instr:
|
||||
if cur == nil {
|
||||
startBlock("")
|
||||
}
|
||||
cur.instrs = append(cur.instrs, st)
|
||||
if terminates(st) || isConditionalBranch(st) {
|
||||
startBlock("")
|
||||
}
|
||||
}
|
||||
}
|
||||
flush()
|
||||
|
||||
// Fix up label->block indices (labels were recorded before the following
|
||||
// block was appended) and build successor edges.
|
||||
for i, b := range l.blocks {
|
||||
if b.label != "" {
|
||||
labelToBlock[b.label] = i
|
||||
}
|
||||
}
|
||||
for i, b := range l.blocks {
|
||||
if len(b.instrs) == 0 {
|
||||
if i+1 < len(l.blocks) {
|
||||
b.succ = append(b.succ, i+1)
|
||||
}
|
||||
continue
|
||||
}
|
||||
last := b.instrs[len(b.instrs)-1]
|
||||
mnem := strings.ToUpper(last.Mnemonic.Text)
|
||||
switch {
|
||||
case mnem == "RET" || mnem == "UNDEF":
|
||||
// No successors.
|
||||
case isConditionalBranch(last):
|
||||
if tgt, ok := branchTarget(last); ok {
|
||||
if j, found := labelToBlock[tgt]; found {
|
||||
b.succ = append(b.succ, j)
|
||||
}
|
||||
}
|
||||
if i+1 < len(l.blocks) {
|
||||
b.succ = append(b.succ, i+1) // fall-through
|
||||
}
|
||||
case isUnconditionalBranchAny(mnem):
|
||||
if tgt, ok := branchTarget(last); ok {
|
||||
if j, found := labelToBlock[tgt]; found {
|
||||
b.succ = append(b.succ, j)
|
||||
}
|
||||
}
|
||||
default:
|
||||
if i+1 < len(l.blocks) {
|
||||
b.succ = append(b.succ, i+1)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// dataflow runs the standard backward liveness iteration to a fixed point.
|
||||
func (l *liveness) dataflow(a arch.Arch) {
|
||||
n := len(l.blocks)
|
||||
l.liveIn = make([]map[string]bool, n)
|
||||
liveOut := make([]map[string]bool, n)
|
||||
use := make([]map[string]bool, n)
|
||||
def := make([]map[string]bool, n)
|
||||
for i, b := range l.blocks {
|
||||
use[i], def[i] = blockUseDef(b, a)
|
||||
l.liveIn[i] = map[string]bool{}
|
||||
liveOut[i] = map[string]bool{}
|
||||
}
|
||||
for changed := true; changed; {
|
||||
changed = false
|
||||
for i := n - 1; i >= 0; i-- {
|
||||
out := map[string]bool{}
|
||||
for _, s := range l.blocks[i].succ {
|
||||
for r := range l.liveIn[s] {
|
||||
out[r] = true
|
||||
}
|
||||
}
|
||||
if !sameSet(out, liveOut[i]) {
|
||||
liveOut[i] = out
|
||||
changed = true
|
||||
}
|
||||
// in = use ∪ (out − def)
|
||||
in := map[string]bool{}
|
||||
for r := range use[i] {
|
||||
in[r] = true
|
||||
}
|
||||
for r := range out {
|
||||
if !def[i][r] {
|
||||
in[r] = true
|
||||
}
|
||||
}
|
||||
if !sameSet(in, l.liveIn[i]) {
|
||||
l.liveIn[i] = in
|
||||
changed = true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// blockUseDef computes the registers used before definition (use) and the
|
||||
// registers defined (def) within a basic block.
|
||||
func blockUseDef(b *block, a arch.Arch) (use, def map[string]bool) {
|
||||
use = map[string]bool{}
|
||||
def = map[string]bool{}
|
||||
for _, in := range b.instrs {
|
||||
eff := instrEffect(in, a)
|
||||
for _, r := range eff.use {
|
||||
if !def[r] {
|
||||
use[r] = true
|
||||
}
|
||||
}
|
||||
for _, r := range eff.def {
|
||||
def[r] = true
|
||||
}
|
||||
}
|
||||
return use, def
|
||||
}
|
||||
|
||||
// terminates reports whether an instruction ends basic-block flow unconditionally.
|
||||
func terminates(in *ast.Instr) bool {
|
||||
m := strings.ToUpper(in.Mnemonic.Text)
|
||||
return m == "RET" || m == "UNDEF" || isUnconditionalBranchAny(m)
|
||||
}
|
||||
|
||||
func isConditionalBranch(in *ast.Instr) bool {
|
||||
m := strings.ToUpper(in.Mnemonic.Text)
|
||||
// Conditional jumps/branches, but not the unconditional ones.
|
||||
if isUnconditionalBranchAny(m) || m == "RET" || m == "UNDEF" || m == "CALL" {
|
||||
return false
|
||||
}
|
||||
return strings.HasPrefix(m, "J") || strings.HasPrefix(m, "B") ||
|
||||
strings.HasPrefix(m, "CBZ") || strings.HasPrefix(m, "CBNZ") ||
|
||||
strings.HasPrefix(m, "TBZ") || strings.HasPrefix(m, "TBNZ") ||
|
||||
strings.HasPrefix(m, "BEQ") || strings.HasPrefix(m, "BNE")
|
||||
}
|
||||
|
||||
// isUnconditionalBranchAny is an arch-agnostic unconditional-branch test.
|
||||
func isUnconditionalBranchAny(m string) bool {
|
||||
switch m {
|
||||
case "JMP", "J", "JR", "B", "BR", "JIRL":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// branchTarget returns the local-label target of a branch, if it is one.
|
||||
func branchTarget(in *ast.Instr) (string, bool) {
|
||||
for _, op := range in.Operands {
|
||||
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
|
||||
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
|
||||
return op.Addr.Sym.Name, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// instrEffect returns the register-level effect of one instruction.
|
||||
func instrEffect(in *ast.Instr, a arch.Arch) regEffect {
|
||||
var eff regEffect
|
||||
mnem := strings.ToUpper(in.Mnemonic.Text)
|
||||
|
||||
// PUSH/POP move a register to/from the stack.
|
||||
if strings.HasPrefix(mnem, "PUSH") {
|
||||
for _, op := range in.Operands {
|
||||
if r := gprName(op, a); r != "" {
|
||||
eff.use = append(eff.use, r)
|
||||
eff.saveGPR = append(eff.saveGPR, r)
|
||||
}
|
||||
}
|
||||
return eff
|
||||
}
|
||||
if strings.HasPrefix(mnem, "POP") {
|
||||
for _, op := range in.Operands {
|
||||
if r := gprName(op, a); r != "" {
|
||||
eff.def = append(eff.def, r)
|
||||
eff.restGPR = append(eff.restGPR, r)
|
||||
}
|
||||
}
|
||||
return eff
|
||||
}
|
||||
|
||||
compare := isCompare(mnem)
|
||||
dstIdx := dstIndex(in, a)
|
||||
|
||||
for i, op := range in.Operands {
|
||||
r := gprName(op, a)
|
||||
if r != "" {
|
||||
if i == dstIdx && !compare {
|
||||
eff.def = append(eff.def, r)
|
||||
// Arithmetic also reads its destination.
|
||||
eff.use = append(eff.use, r)
|
||||
} else {
|
||||
eff.use = append(eff.use, r)
|
||||
}
|
||||
}
|
||||
// Detect saves/restores through the stack frame.
|
||||
if isStackAddr(op) {
|
||||
// The other operand (the register) is being saved or restored.
|
||||
for j, other := range in.Operands {
|
||||
if j == i {
|
||||
continue
|
||||
}
|
||||
if rr := gprName(other, a); rr != "" {
|
||||
if j == dstIdx && !compare {
|
||||
eff.restGPR = append(eff.restGPR, rr) // reg loaded from stack
|
||||
} else {
|
||||
eff.saveGPR = append(eff.saveGPR, rr) // reg stored to stack
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return eff
|
||||
}
|
||||
|
||||
// dstIndex returns the operand index of the destination register: last for the
|
||||
// Plan 9 (amd64) spelling, first for arm64/riscv64/loong64.
|
||||
func dstIndex(in *ast.Instr, a arch.Arch) int {
|
||||
if a == arch.AMD64 {
|
||||
return len(in.Operands) - 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// isCompare reports whether the mnemonic only reads its operands (setting flags).
|
||||
func isCompare(m string) bool {
|
||||
return strings.HasPrefix(m, "CMP") || strings.HasPrefix(m, "TEST") ||
|
||||
strings.HasPrefix(m, "CMN") || strings.HasPrefix(m, "TST") ||
|
||||
m == "FCMP" || m == "FCMPE"
|
||||
}
|
||||
|
||||
// gprName returns the canonical general-purpose register name of an operand, or
|
||||
// "" if the operand is not a bare GPR reference.
|
||||
func gprName(op *ast.Operand, a arch.Arch) string {
|
||||
if op == nil || op.Kind != ast.OpAddr || op.Addr.Sym == nil {
|
||||
return ""
|
||||
}
|
||||
if op.Addr.Base != "" || op.Addr.Sym.Pseudo != "" || op.Addr.Sym.Name == "" {
|
||||
return ""
|
||||
}
|
||||
name := op.Addr.Sym.Name
|
||||
if r, ok := arch.ForArch(a).Register(name); ok && (r.Class == arch.GPR || r.Class == arch.GPRSub) {
|
||||
return canonicalGPR(name)
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// canonicalGPR maps a sized sub-register to its base GPR (amd64 only).
|
||||
func canonicalGPR(name string) string {
|
||||
upper := strings.ToUpper(name)
|
||||
// Named 8/16/32-bit forms of the classic registers.
|
||||
switch upper {
|
||||
case "AL", "AH", "AX":
|
||||
return "AX"
|
||||
case "BL", "BH", "BX":
|
||||
return "BX"
|
||||
case "CL", "CH", "CX":
|
||||
return "CX"
|
||||
case "DL", "DH", "DX":
|
||||
return "DX"
|
||||
case "SIL":
|
||||
return "SI"
|
||||
case "DIL":
|
||||
return "DI"
|
||||
case "BPL":
|
||||
return "BP"
|
||||
case "SPL":
|
||||
return "SP"
|
||||
}
|
||||
// Numbered sub-registers R8B/R8W/R8D → R8.
|
||||
if len(upper) >= 3 && upper[0] == 'R' {
|
||||
switch upper[len(upper)-1] {
|
||||
case 'B', 'W', 'D':
|
||||
return upper[:len(upper)-1]
|
||||
}
|
||||
}
|
||||
return upper
|
||||
}
|
||||
|
||||
// isStackAddr reports whether an operand addresses the stack frame
|
||||
// (base SP, or an FP/SP-relative symbol).
|
||||
func isStackAddr(op *ast.Operand) bool {
|
||||
if op == nil || op.Kind != ast.OpAddr {
|
||||
return false
|
||||
}
|
||||
if op.Addr.Base == "SP" {
|
||||
return true
|
||||
}
|
||||
if op.Addr.Sym != nil && (op.Addr.Sym.Pseudo == "SP" || op.Addr.Sym.Pseudo == "FP") {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func sameSet(a, b map[string]bool) bool {
|
||||
if len(a) != len(b) {
|
||||
return false
|
||||
}
|
||||
for k := range a {
|
||||
if !b[k] {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// calleeSavedGPRs returns the general-purpose registers an assembly function
|
||||
// must preserve for its caller, using the register names the assembler accepts
|
||||
// for each architecture.
|
||||
func calleeSavedGPRs(a arch.Arch) map[string]bool {
|
||||
switch a {
|
||||
case arch.AMD64:
|
||||
return gprSet("BX", "BP", "R12", "R13", "R14", "R15")
|
||||
case arch.ARM64:
|
||||
names := []string{"R29", "R30"} // FP, LR
|
||||
for i := 19; i <= 28; i++ {
|
||||
names = append(names, fmt.Sprintf("R%d", i))
|
||||
}
|
||||
return gprSet(names...)
|
||||
case arch.RISCV:
|
||||
// RA (X1) and the S registers (X8, X9, X18–X27) are callee-saved.
|
||||
names := []string{"X1", "RA", "X8", "X9", "S0", "S1", "FP"}
|
||||
for i := 18; i <= 27; i++ {
|
||||
names = append(names, fmt.Sprintf("X%d", i))
|
||||
}
|
||||
for i := 2; i <= 11; i++ {
|
||||
names = append(names, fmt.Sprintf("S%d", i))
|
||||
}
|
||||
return gprSet(names...)
|
||||
case arch.LOONG64:
|
||||
// RA (R1), FP (R22) and S0–S8 (R23–R31) are callee-saved.
|
||||
names := []string{"R1", "RA", "R22", "FP"}
|
||||
for i := 23; i <= 31; i++ {
|
||||
names = append(names, fmt.Sprintf("R%d", i))
|
||||
}
|
||||
for i := 0; i <= 8; i++ {
|
||||
names = append(names, fmt.Sprintf("S%d", i))
|
||||
}
|
||||
return gprSet(names...)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func gprSet(names ...string) map[string]bool {
|
||||
m := make(map[string]bool, len(names))
|
||||
for _, n := range names {
|
||||
m[n] = true
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// clobberedCalleeSaved returns the callee-saved registers a function writes
|
||||
// without also saving and restoring them — i.e. registers whose caller-owned
|
||||
// value is lost across the call. It walks the blocks of the liveness analysis
|
||||
// (so the control-flow graph is what supplies the instruction set) and
|
||||
// aggregates each instruction's register effects.
|
||||
func clobberedCalleeSaved(l *liveness, a arch.Arch) []string {
|
||||
callee := calleeSavedGPRs(a)
|
||||
if len(callee) == 0 {
|
||||
return nil
|
||||
}
|
||||
def := map[string]bool{}
|
||||
saved := map[string]bool{}
|
||||
restored := map[string]bool{}
|
||||
for _, b := range l.blocks {
|
||||
for _, in := range b.instrs {
|
||||
eff := instrEffect(in, a)
|
||||
for _, r := range eff.def {
|
||||
def[r] = true
|
||||
}
|
||||
for _, r := range eff.saveGPR {
|
||||
saved[r] = true
|
||||
}
|
||||
for _, r := range eff.restGPR {
|
||||
restored[r] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
var out []string
|
||||
for r := range callee {
|
||||
if def[r] && !(saved[r] && restored[r]) {
|
||||
out = append(out, r)
|
||||
}
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lint
|
||||
|
||||
import "testing"
|
||||
|
||||
// TestRegisterClobber detects writes to callee-saved registers that are not
|
||||
// saved and restored.
|
||||
func TestRegisterClobber(t *testing.T) {
|
||||
// BX (callee-saved on amd64) is written but never saved → clobbered.
|
||||
clob := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVQ CX, BX\n"+
|
||||
"\tRET\n")
|
||||
if codes(clob)[CodeRegisterClobber] != 1 {
|
||||
t.Fatalf("unsaved callee-saved write should be flagged: %+v", clob)
|
||||
}
|
||||
|
||||
// Saved and restored → preserved.
|
||||
saved := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $8\n"+
|
||||
"\tPUSHQ BX\n"+
|
||||
"\tMOVQ CX, BX\n"+
|
||||
"\tPOPQ BX\n"+
|
||||
"\tRET\n")
|
||||
if codes(saved)[CodeRegisterClobber] != 0 {
|
||||
t.Fatalf("saved/restored register must not be flagged: %+v", saved)
|
||||
}
|
||||
|
||||
// A caller-saved register (CX) is fine to write.
|
||||
caller := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tMOVQ $1, CX\n"+
|
||||
"\tRET\n")
|
||||
if codes(caller)[CodeRegisterClobber] != 0 {
|
||||
t.Fatalf("caller-saved register must not be flagged: %+v", caller)
|
||||
}
|
||||
}
|
||||
|
||||
// TestFuncdata validates the FUNCDATA/PCDATA structural checks.
|
||||
func TestFuncdata(t *testing.T) {
|
||||
// Well formed: no findings.
|
||||
good := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tFUNCDATA $0, gclocals·abc(SB)\n"+
|
||||
"\tPCDATA $1, $0\n"+
|
||||
"\tRET\n")
|
||||
if codes(good)[CodeFuncdata] != 0 {
|
||||
t.Fatalf("well-formed FUNCDATA/PCDATA must not be flagged: %+v", good)
|
||||
}
|
||||
|
||||
// FUNCDATA with one operand.
|
||||
bad1 := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tFUNCDATA $0\n"+
|
||||
"\tRET\n")
|
||||
if codes(bad1)[CodeFuncdata] == 0 {
|
||||
t.Fatal("FUNCDATA with one operand should be flagged")
|
||||
}
|
||||
|
||||
// PCDATA with a non-immediate value.
|
||||
bad2 := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tPCDATA $0, AX\n"+
|
||||
"\tRET\n")
|
||||
if codes(bad2)[CodeFuncdata] == 0 {
|
||||
t.Fatal("PCDATA with a register value should be flagged")
|
||||
}
|
||||
|
||||
// FUNCDATA index out of range.
|
||||
bad3 := lintSrc(t, "#include \"textflag.h\"\n"+
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n"+
|
||||
"\tFUNCDATA $99, gclocals·abc(SB)\n"+
|
||||
"\tRET\n")
|
||||
if codes(bad3)[CodeFuncdata] == 0 {
|
||||
t.Fatal("out-of-range FUNCDATA index should be flagged")
|
||||
}
|
||||
}
|
||||
+382
@@ -0,0 +1,382 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lsp
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strings"
|
||||
"unicode"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// textflagMacros are the flag names defined by textflag.h; they are highlighted
|
||||
// as macros and offered as completions after a TEXT/GLOBL directive.
|
||||
var textflagMacros = map[string]bool{
|
||||
"NOPROFILE": true, "DUPOK": true, "NOSPLIT": true, "RODATA": true,
|
||||
"NOPTR": true, "WRAPPER": true, "NEEDCTXT": true, "TOPFRAME": true,
|
||||
"LEAF": true, "ABI0": true, "REFLECTDATA": true,
|
||||
}
|
||||
|
||||
// completion builds the completion list for a document.
|
||||
func (s *Server) completion(p completionParams) []CompletionItem {
|
||||
a := arch.ForArch(arch.FromFilename(uriPath(p.TextDocument.URI)))
|
||||
tab := a
|
||||
items := []CompletionItem{
|
||||
{Label: "TEXT", Kind: ciKeyword, Detail: "define a function"},
|
||||
{Label: "DATA", Kind: ciKeyword, Detail: "initialise a data symbol"},
|
||||
{Label: "GLOBL", Kind: ciKeyword, Detail: "declare a global symbol"},
|
||||
}
|
||||
for name := range textflagMacros {
|
||||
items = append(items, CompletionItem{Label: name, Kind: ciKeyword, Detail: "textflag.h flag"})
|
||||
}
|
||||
for name, desc := range map[string]string{
|
||||
"FP": "frame pointer (arguments/results)", "SP": "stack pointer",
|
||||
"SB": "static base (globals)", "PC": "program counter",
|
||||
} {
|
||||
items = append(items, CompletionItem{Label: name, Kind: ciConstant, Detail: desc})
|
||||
}
|
||||
for _, in := range tab.Instructions() {
|
||||
items = append(items, CompletionItem{
|
||||
Label: in.Name, Kind: ciFunction, Detail: in.Summary, Documentation: in.Summary,
|
||||
})
|
||||
}
|
||||
for _, r := range tab.Registers() {
|
||||
kind := ciVariable
|
||||
if r.Class == arch.Vector || r.Class == arch.Mask || r.Class == arch.Float || r.Class == arch.VecARM {
|
||||
kind = ciClass
|
||||
}
|
||||
items = append(items, CompletionItem{Label: r.Name, Kind: kind, Detail: r.Desc})
|
||||
}
|
||||
// Local labels defined in the document.
|
||||
if f, _ := parser.Parse("", s.docs[p.TextDocument.URI]); f != nil {
|
||||
for _, name := range labelNames(f) {
|
||||
items = append(items, CompletionItem{Label: name, Kind: ciModule, Detail: "local label"})
|
||||
}
|
||||
}
|
||||
sort.Slice(items, func(i, j int) bool { return items[i].Label < items[j].Label })
|
||||
return items
|
||||
}
|
||||
|
||||
// hover returns documentation for the symbol under the cursor.
|
||||
func (s *Server) hover(p hoverParams) *Hover {
|
||||
text := s.docs[p.TextDocument.URI]
|
||||
word, rng := wordAt(text, p.Position)
|
||||
if word == "" {
|
||||
return nil
|
||||
}
|
||||
a := arch.ForArch(arch.FromFilename(uriPath(p.TextDocument.URI)))
|
||||
|
||||
var md string
|
||||
if in, ok := a.Lookup(word); ok {
|
||||
md = "**" + in.Name + "** — " + in.Summary
|
||||
} else if r, ok := a.Register(word); ok {
|
||||
md = "**" + r.Name + "** — " + r.Class.String() + " register. " + r.Desc
|
||||
} else if desc, ok := arch.PseudoRegDesc(word); ok {
|
||||
md = "**" + strings.ToUpper(word) + "** — pseudo-register. " + desc
|
||||
} else if textflagMacros[strings.ToUpper(word)] {
|
||||
md = "**" + strings.ToUpper(word) + "** — textflag.h flag"
|
||||
} else {
|
||||
return nil
|
||||
}
|
||||
return &Hover{
|
||||
Contents: markupContent{Kind: "markdown", Value: md},
|
||||
Range: rng,
|
||||
}
|
||||
}
|
||||
|
||||
// documentSymbols returns functions and their labels, plus global symbols.
|
||||
func (s *Server) documentSymbols(p documentSymbolParams) []DocumentSymbol {
|
||||
text := s.docs[p.TextDocument.URI]
|
||||
f, _ := parser.Parse(uriPath(p.TextDocument.URI), text)
|
||||
if f == nil {
|
||||
return nil
|
||||
}
|
||||
var out []DocumentSymbol
|
||||
for _, d := range f.Decls {
|
||||
switch dd := d.(type) {
|
||||
case *ast.Text:
|
||||
sym := DocumentSymbol{
|
||||
Name: dd.Name.Name,
|
||||
Detail: "TEXT " + strings.Join(dd.Flags, " "),
|
||||
Kind: symFunction,
|
||||
Range: textRange(dd),
|
||||
SelectionRange: symRange(dd.Name),
|
||||
}
|
||||
for _, st := range dd.Body {
|
||||
if l, ok := st.(*ast.Label); ok {
|
||||
sym.Children = append(sym.Children, DocumentSymbol{
|
||||
Name: l.Name.Text,
|
||||
Kind: symVariable,
|
||||
Range: tokenRange(l.Name),
|
||||
SelectionRange: tokenRange(l.Name),
|
||||
})
|
||||
}
|
||||
}
|
||||
out = append(out, sym)
|
||||
case *ast.Globl:
|
||||
out = append(out, DocumentSymbol{
|
||||
Name: dd.Name.Name, Detail: "GLOBL", Kind: symConstant,
|
||||
Range: symRange(dd.Name), SelectionRange: symRange(dd.Name),
|
||||
})
|
||||
case *ast.Data:
|
||||
out = append(out, DocumentSymbol{
|
||||
Name: dd.Name.Name, Detail: "DATA", Kind: symConstant,
|
||||
Range: symRange(dd.Name), SelectionRange: symRange(dd.Name),
|
||||
})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// semTok is one classified token before delta encoding.
|
||||
type semTok struct {
|
||||
line, char, length, typ int
|
||||
}
|
||||
|
||||
// semanticTokens encodes syntax highlighting as LSP semantic tokens.
|
||||
func (s *Server) semanticTokens(p semanticTokensParams) SemanticTokens {
|
||||
text := s.docs[p.TextDocument.URI]
|
||||
a := arch.ForArch(arch.FromFilename(uriPath(p.TextDocument.URI)))
|
||||
f, _ := parser.Parse("", text)
|
||||
labels := map[string]bool{}
|
||||
for _, name := range labelNames(f) {
|
||||
labels[name] = true
|
||||
}
|
||||
|
||||
toks := lexer.Tokenize(text)
|
||||
lines := groupLines(toks)
|
||||
|
||||
var encoded []semTok
|
||||
for _, line := range lines {
|
||||
encoded = append(encoded, classifyLine(line, a, labels)...)
|
||||
}
|
||||
|
||||
return SemanticTokens{Data: deltaEncode(encoded)}
|
||||
}
|
||||
|
||||
// classifyLine assigns a semantic token type to each significant token on a line.
|
||||
func classifyLine(line []token.Token, a *arch.Table, labels map[string]bool) []semTok {
|
||||
if len(line) == 0 {
|
||||
return nil
|
||||
}
|
||||
var out []semTok
|
||||
first := firstSignificant(line)
|
||||
if first < 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
isDirective := line[first].Kind == token.Ident &&
|
||||
(line[first].Text == "TEXT" || line[first].Text == "DATA" || line[first].Text == "GLOBL")
|
||||
isLabel := line[first].Kind == token.Ident && first+1 < len(line) &&
|
||||
line[first+1].Kind == token.Colon
|
||||
isInstr := !isDirective && !isLabel && line[first].Kind == token.Ident
|
||||
|
||||
mnemonicDone := false
|
||||
for i, t := range line {
|
||||
typ := -1
|
||||
switch t.Kind {
|
||||
case token.Comment:
|
||||
typ = stComment
|
||||
case token.Number:
|
||||
typ = stNumber
|
||||
case token.String, token.Rune:
|
||||
typ = stString
|
||||
case token.Hash:
|
||||
typ = stMacro
|
||||
case token.Ident:
|
||||
typ = classifyIdent(line, i, first, t.Text, a, labels,
|
||||
isDirective, isLabel, isInstr, &mnemonicDone)
|
||||
case token.Colon, token.Comma, token.LParen, token.RParen,
|
||||
token.Plus, token.Minus, token.Star, token.Slash, token.Dollar,
|
||||
token.LAngle, token.RAngle, token.LShift, token.RShift, token.Arrow, token.At:
|
||||
typ = stOperator
|
||||
}
|
||||
if typ >= 0 {
|
||||
out = append(out, semTok{
|
||||
line: t.Pos.Line - 1,
|
||||
char: t.Pos.Column - 1,
|
||||
length: runeLen(t.Text),
|
||||
typ: typ,
|
||||
})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// classifyIdent decides the semantic type of an identifier token.
|
||||
func classifyIdent(line []token.Token, i, first int, text string, a *arch.Table,
|
||||
labels map[string]bool, isDirective, isLabel, isInstr bool, mnemonicDone *bool) int {
|
||||
|
||||
upper := strings.ToUpper(text)
|
||||
switch {
|
||||
case isDirective && i == first:
|
||||
return stKeyword
|
||||
case isDirective && textflagMacros[upper]:
|
||||
return stMacro
|
||||
case isLabel && i == first:
|
||||
return stNamespace
|
||||
case arch.IsPseudoReg(text):
|
||||
return stProperty
|
||||
case labels[text]:
|
||||
return stNamespace
|
||||
}
|
||||
if r, ok := a.Register(text); ok {
|
||||
switch r.Class {
|
||||
case arch.Vector, arch.Mask, arch.Float, arch.VecARM:
|
||||
return stType
|
||||
default:
|
||||
return stVariable
|
||||
}
|
||||
}
|
||||
if isInstr && i == first && !*mnemonicDone {
|
||||
*mnemonicDone = true
|
||||
return stFunction
|
||||
}
|
||||
// Argument/symbol names and anything else.
|
||||
return stVariable
|
||||
}
|
||||
|
||||
// deltaEncode converts absolute token positions to the LSP relative encoding.
|
||||
func deltaEncode(toks []semTok) []int {
|
||||
data := make([]int, 0, len(toks)*5)
|
||||
prevLine, prevChar := 0, 0
|
||||
for _, t := range toks {
|
||||
dLine := t.line - prevLine
|
||||
dChar := t.char
|
||||
if dLine == 0 {
|
||||
dChar = t.char - prevChar
|
||||
}
|
||||
data = append(data, dLine, dChar, t.length, t.typ, 0)
|
||||
prevLine, prevChar = t.line, t.char
|
||||
}
|
||||
return data
|
||||
}
|
||||
|
||||
// --- shared helpers ---------------------------------------------------------
|
||||
|
||||
// wordAt extracts the identifier surrounding pos and its range.
|
||||
func wordAt(text string, pos Position) (string, Range) {
|
||||
lines := strings.Split(text, "\n")
|
||||
if pos.Line < 0 || pos.Line >= len(lines) {
|
||||
return "", Range{}
|
||||
}
|
||||
runes := []rune(lines[pos.Line])
|
||||
col := pos.Character
|
||||
if col < 0 || col > len(runes) {
|
||||
return "", Range{}
|
||||
}
|
||||
isWord := func(r rune) bool {
|
||||
return r == '_' || r == '\u00B7' || unicode.IsLetter(r) || unicode.IsDigit(r)
|
||||
}
|
||||
start, end := col, col
|
||||
for start > 0 && isWord(runes[start-1]) {
|
||||
start--
|
||||
}
|
||||
for end < len(runes) && isWord(runes[end]) {
|
||||
end++
|
||||
}
|
||||
if start == end {
|
||||
return "", Range{}
|
||||
}
|
||||
rng := Range{
|
||||
Start: Position{Line: pos.Line, Character: start},
|
||||
End: Position{Line: pos.Line, Character: end},
|
||||
}
|
||||
return string(runes[start:end]), rng
|
||||
}
|
||||
|
||||
// labelNames collects every label defined in a file.
|
||||
func labelNames(f *ast.File) []string {
|
||||
if f == nil {
|
||||
return nil
|
||||
}
|
||||
seen := map[string]bool{}
|
||||
var out []string
|
||||
collect := func(body []ast.Stmt) {
|
||||
for _, st := range body {
|
||||
if l, ok := st.(*ast.Label); ok && !seen[l.Name.Text] {
|
||||
seen[l.Name.Text] = true
|
||||
out = append(out, l.Name.Text)
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, d := range f.Decls {
|
||||
if t, ok := d.(*ast.Text); ok {
|
||||
collect(t.Body)
|
||||
}
|
||||
}
|
||||
collect(f.Orphans)
|
||||
return out
|
||||
}
|
||||
|
||||
// groupLines splits a token stream into lines, keeping Newline boundaries but
|
||||
// dropping the Newline and EOF tokens themselves.
|
||||
func groupLines(toks []token.Token) [][]token.Token {
|
||||
var lines [][]token.Token
|
||||
var cur []token.Token
|
||||
for _, t := range toks {
|
||||
if t.Kind == token.EOF {
|
||||
break
|
||||
}
|
||||
if t.Kind == token.Newline {
|
||||
lines = append(lines, cur)
|
||||
cur = nil
|
||||
continue
|
||||
}
|
||||
cur = append(cur, t)
|
||||
}
|
||||
if len(cur) > 0 {
|
||||
lines = append(lines, cur)
|
||||
}
|
||||
return lines
|
||||
}
|
||||
|
||||
func firstSignificant(line []token.Token) int {
|
||||
for i, t := range line {
|
||||
if t.Kind != token.Comment {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
func runeLen(s string) int { return len([]rune(s)) }
|
||||
|
||||
// symRange builds a range covering a symbol from its position and raw text.
|
||||
func symRange(sym *ast.Symbol) Range {
|
||||
start := Position{Line: sym.Pos.Line - 1, Character: sym.Pos.Column - 1}
|
||||
end := start
|
||||
end.Character += runeLen(sym.Name)
|
||||
return Range{Start: start, End: end}
|
||||
}
|
||||
|
||||
// tokenRange builds a range covering one token.
|
||||
func tokenRange(t token.Token) Range {
|
||||
return Range{
|
||||
Start: Position{Line: t.Pos.Line - 1, Character: t.Pos.Column - 1},
|
||||
End: Position{Line: t.End.Line - 1, Character: t.End.Column - 1},
|
||||
}
|
||||
}
|
||||
|
||||
// textRange spans a TEXT function from its keyword to the end of its body.
|
||||
func textRange(t *ast.Text) Range {
|
||||
start := Position{Line: t.Keyword.Pos.Line - 1, Character: t.Keyword.Pos.Column - 1}
|
||||
end := start
|
||||
end.Character += runeLen(t.Keyword.Text)
|
||||
if n := len(t.Body); n > 0 {
|
||||
last := t.Body[n-1]
|
||||
if in, ok := last.(*ast.Instr); ok {
|
||||
end = Position{Line: in.Mnemonic.End.Line - 1, Character: in.Mnemonic.End.Column - 1}
|
||||
} else {
|
||||
lp := last.Pos()
|
||||
end = Position{Line: lp.Line - 1, Character: lp.Column - 1}
|
||||
}
|
||||
}
|
||||
return Range{Start: start, End: end}
|
||||
}
|
||||
+255
@@ -0,0 +1,255 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package lsp implements a Language Server Protocol server for GAsm. It
|
||||
// speaks JSON-RPC 2.0 over any io.Reader/io.Writer pair (normally standard
|
||||
// input/output) and provides completion, hover documentation, document
|
||||
// symbols, diagnostics and semantic-token highlighting — all backed by the
|
||||
// pure-Go lexer, parser, arch and lint packages. It is the vendor-neutral
|
||||
// integration point: any LSP-capable editor can use it with no editor-specific
|
||||
// plugin code.
|
||||
package lsp
|
||||
|
||||
import "encoding/json"
|
||||
|
||||
// --- JSON-RPC 2.0 -----------------------------------------------------------
|
||||
|
||||
// rpcMessage is the common envelope for every JSON-RPC message.
|
||||
type rpcMessage struct {
|
||||
JSONRPC string `json:"jsonrpc"`
|
||||
ID *json.RawMessage `json:"id,omitempty"`
|
||||
Method string `json:"method,omitempty"`
|
||||
Params json.RawMessage `json:"params,omitempty"`
|
||||
Result any `json:"result,omitempty"`
|
||||
Error *rpcError `json:"error,omitempty"`
|
||||
}
|
||||
|
||||
type rpcError struct {
|
||||
Code int `json:"code"`
|
||||
Message string `json:"message"`
|
||||
}
|
||||
|
||||
// JSON-RPC error codes used by LSP.
|
||||
const (
|
||||
errParse = -32700
|
||||
errInvalidRequest = -32600
|
||||
errMethodNotFound = -32601
|
||||
errInvalidParams = -32602
|
||||
errInternal = -32603
|
||||
)
|
||||
|
||||
// --- LSP positions and ranges ----------------------------------------------
|
||||
|
||||
// Position is a zero-based line/character position, as LSP requires.
|
||||
type Position struct {
|
||||
Line int `json:"line"`
|
||||
Character int `json:"character"`
|
||||
}
|
||||
|
||||
// Range is a pair of positions.
|
||||
type Range struct {
|
||||
Start Position `json:"start"`
|
||||
End Position `json:"end"`
|
||||
}
|
||||
|
||||
// Location is a range within a document URI.
|
||||
type Location struct {
|
||||
URI string `json:"uri"`
|
||||
Range Range `json:"range"`
|
||||
}
|
||||
|
||||
// --- diagnostics ------------------------------------------------------------
|
||||
|
||||
// Diagnostic severities (LSP ordering: 1 = error).
|
||||
const (
|
||||
sevError = 1
|
||||
sevWarning = 2
|
||||
sevInformation = 3
|
||||
sevHint = 4
|
||||
)
|
||||
|
||||
// Diagnostic is one published finding.
|
||||
type Diagnostic struct {
|
||||
Range Range `json:"range"`
|
||||
Severity int `json:"severity"`
|
||||
Code string `json:"code,omitempty"`
|
||||
Source string `json:"source,omitempty"`
|
||||
Message string `json:"message"`
|
||||
}
|
||||
|
||||
type publishDiagnosticsParams struct {
|
||||
URI string `json:"uri"`
|
||||
Diagnostics []Diagnostic `json:"diagnostics"`
|
||||
}
|
||||
|
||||
// --- text document synchronisation -----------------------------------------
|
||||
|
||||
type textDocumentItem struct {
|
||||
URI string `json:"uri"`
|
||||
LanguageID string `json:"languageId"`
|
||||
Version int `json:"version"`
|
||||
Text string `json:"text"`
|
||||
}
|
||||
|
||||
type didOpenParams struct {
|
||||
TextDocument textDocumentItem `json:"textDocument"`
|
||||
}
|
||||
|
||||
type versionedTextDocumentIdentifier struct {
|
||||
URI string `json:"uri"`
|
||||
Version int `json:"version"`
|
||||
}
|
||||
|
||||
type textDocumentIdentifier struct {
|
||||
URI string `json:"uri"`
|
||||
}
|
||||
|
||||
type contentChangeEvent struct {
|
||||
Text string `json:"text"`
|
||||
}
|
||||
|
||||
type didChangeParams struct {
|
||||
TextDocument versionedTextDocumentIdentifier `json:"textDocument"`
|
||||
ContentChanges []contentChangeEvent `json:"contentChanges"`
|
||||
}
|
||||
|
||||
type didCloseParams struct {
|
||||
TextDocument textDocumentIdentifier `json:"textDocument"`
|
||||
}
|
||||
|
||||
// --- completion -------------------------------------------------------------
|
||||
|
||||
// Completion item kinds (a useful subset).
|
||||
const (
|
||||
ciFunction = 3
|
||||
ciField = 5
|
||||
ciVariable = 6
|
||||
ciClass = 7
|
||||
ciModule = 9
|
||||
ciKeyword = 14
|
||||
ciConstant = 21
|
||||
ciStruct = 22
|
||||
)
|
||||
|
||||
// CompletionItem is one completion suggestion.
|
||||
type CompletionItem struct {
|
||||
Label string `json:"label"`
|
||||
Kind int `json:"kind,omitempty"`
|
||||
Detail string `json:"detail,omitempty"`
|
||||
Documentation string `json:"documentation,omitempty"`
|
||||
InsertText string `json:"insertText,omitempty"`
|
||||
}
|
||||
|
||||
type completionParams struct {
|
||||
TextDocument textDocumentIdentifier `json:"textDocument"`
|
||||
Position Position `json:"position"`
|
||||
}
|
||||
|
||||
// --- hover ------------------------------------------------------------------
|
||||
|
||||
type hoverParams struct {
|
||||
TextDocument textDocumentIdentifier `json:"textDocument"`
|
||||
Position Position `json:"position"`
|
||||
}
|
||||
|
||||
// Hover is the hover response.
|
||||
type Hover struct {
|
||||
Contents markupContent `json:"contents"`
|
||||
Range Range `json:"range,omitempty"`
|
||||
}
|
||||
|
||||
type markupContent struct {
|
||||
Kind string `json:"kind"`
|
||||
Value string `json:"value"`
|
||||
}
|
||||
|
||||
// --- document symbols -------------------------------------------------------
|
||||
|
||||
// Symbol kinds (a useful subset).
|
||||
const (
|
||||
symFunction = 12
|
||||
symConstant = 14
|
||||
symVariable = 13
|
||||
)
|
||||
|
||||
// DocumentSymbol is a hierarchical symbol.
|
||||
type DocumentSymbol struct {
|
||||
Name string `json:"name"`
|
||||
Detail string `json:"detail,omitempty"`
|
||||
Kind int `json:"kind"`
|
||||
Range Range `json:"range"`
|
||||
SelectionRange Range `json:"selectionRange"`
|
||||
Children []DocumentSymbol `json:"children,omitempty"`
|
||||
}
|
||||
|
||||
type documentSymbolParams struct {
|
||||
TextDocument textDocumentIdentifier `json:"textDocument"`
|
||||
}
|
||||
|
||||
// --- semantic tokens --------------------------------------------------------
|
||||
|
||||
// semanticTokenTypes is the legend of token type names, in index order. The
|
||||
// indices are referenced by the encoder below.
|
||||
var semanticTokenTypes = []string{
|
||||
"comment", // 0
|
||||
"keyword", // 1
|
||||
"function", // 2
|
||||
"variable", // 3
|
||||
"type", // 4
|
||||
"number", // 5
|
||||
"string", // 6
|
||||
"operator", // 7
|
||||
"property", // 8
|
||||
"namespace", // 9
|
||||
"macro", // 10
|
||||
}
|
||||
|
||||
const (
|
||||
stComment = 0
|
||||
stKeyword = 1
|
||||
stFunction = 2
|
||||
stVariable = 3
|
||||
stType = 4
|
||||
stNumber = 5
|
||||
stString = 6
|
||||
stOperator = 7
|
||||
stProperty = 8
|
||||
stNamespace = 9
|
||||
stMacro = 10
|
||||
)
|
||||
|
||||
// SemanticTokensLegend advertises the token classification.
|
||||
type SemanticTokensLegend struct {
|
||||
TokenTypes []string `json:"tokenTypes"`
|
||||
TokenModifiers []string `json:"tokenModifiers"`
|
||||
}
|
||||
|
||||
// SemanticTokens is the encoded token payload.
|
||||
type SemanticTokens struct {
|
||||
Data []int `json:"data"`
|
||||
}
|
||||
|
||||
type semanticTokensParams struct {
|
||||
TextDocument textDocumentIdentifier `json:"textDocument"`
|
||||
}
|
||||
|
||||
// --- initialize -------------------------------------------------------------
|
||||
|
||||
type initializeParams struct {
|
||||
RootURI string `json:"rootUri"`
|
||||
}
|
||||
|
||||
// ServerCapabilities advertises what this server provides.
|
||||
type ServerCapabilities struct {
|
||||
TextDocumentSync int `json:"textDocumentSync"`
|
||||
CompletionProvider map[string]any `json:"completionProvider,omitempty"`
|
||||
HoverProvider bool `json:"hoverProvider,omitempty"`
|
||||
DocumentSymbolProvider bool `json:"documentSymbolProvider,omitempty"`
|
||||
SemanticTokensProvider map[string]any `json:"semanticTokensProvider,omitempty"`
|
||||
DiagnosticProvider map[string]any `json:"diagnosticProvider,omitempty"`
|
||||
}
|
||||
|
||||
type initializeResult struct {
|
||||
Capabilities ServerCapabilities `json:"capabilities"`
|
||||
ServerInfo map[string]string `json:"serverInfo,omitempty"`
|
||||
}
|
||||
+249
@@ -0,0 +1,249 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lsp
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// Server is a GAsm language server bound to a byte stream.
|
||||
type Server struct {
|
||||
in *bufio.Reader
|
||||
out io.Writer
|
||||
mu sync.Mutex // guards writes to out
|
||||
docs map[string]string
|
||||
}
|
||||
|
||||
// New returns a server reading from in and writing to out.
|
||||
func New(in io.Reader, out io.Writer) *Server {
|
||||
return &Server{
|
||||
in: bufio.NewReader(in),
|
||||
out: out,
|
||||
docs: make(map[string]string),
|
||||
}
|
||||
}
|
||||
|
||||
// Run serves requests until the input is exhausted or an exit is requested.
|
||||
func (s *Server) Run() error {
|
||||
for {
|
||||
msg, err := s.read()
|
||||
if err == io.EOF {
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if exit := s.dispatch(msg); exit {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// read parses one Content-Length framed JSON-RPC message.
|
||||
func (s *Server) read() (*rpcMessage, error) {
|
||||
length := -1
|
||||
for {
|
||||
line, err := s.in.ReadString('\n')
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
line = strings.TrimRight(line, "\r\n")
|
||||
if line == "" {
|
||||
break
|
||||
}
|
||||
if k, v, ok := strings.Cut(line, ":"); ok && strings.EqualFold(strings.TrimSpace(k), "Content-Length") {
|
||||
length, _ = strconv.Atoi(strings.TrimSpace(v))
|
||||
}
|
||||
}
|
||||
if length < 0 {
|
||||
return nil, fmt.Errorf("missing Content-Length header")
|
||||
}
|
||||
body := make([]byte, length)
|
||||
if _, err := io.ReadFull(s.in, body); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var msg rpcMessage
|
||||
if err := json.Unmarshal(body, &msg); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &msg, nil
|
||||
}
|
||||
|
||||
// send marshals and writes one framed message.
|
||||
func (s *Server) send(msg *rpcMessage) error {
|
||||
msg.JSONRPC = "2.0"
|
||||
body, err := json.Marshal(msg)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
if _, err := fmt.Fprintf(s.out, "Content-Length: %d\r\n\r\n", len(body)); err != nil {
|
||||
return err
|
||||
}
|
||||
_, err = s.out.Write(body)
|
||||
return err
|
||||
}
|
||||
|
||||
func (s *Server) respond(id *json.RawMessage, result any) {
|
||||
_ = s.send(&rpcMessage{ID: id, Result: result})
|
||||
}
|
||||
|
||||
func (s *Server) respondError(id *json.RawMessage, code int, msg string) {
|
||||
_ = s.send(&rpcMessage{ID: id, Error: &rpcError{Code: code, Message: msg}})
|
||||
}
|
||||
|
||||
func (s *Server) notify(method string, params any) {
|
||||
raw, _ := json.Marshal(params)
|
||||
_ = s.send(&rpcMessage{Method: method, Params: raw})
|
||||
}
|
||||
|
||||
// dispatch routes one message. It returns true when the server should stop.
|
||||
func (s *Server) dispatch(msg *rpcMessage) (exit bool) {
|
||||
switch msg.Method {
|
||||
case "initialize":
|
||||
s.respond(msg.ID, initializeResult{
|
||||
Capabilities: ServerCapabilities{
|
||||
TextDocumentSync: 1, // full sync
|
||||
CompletionProvider: map[string]any{},
|
||||
HoverProvider: true,
|
||||
DocumentSymbolProvider: true,
|
||||
SemanticTokensProvider: map[string]any{
|
||||
"legend": SemanticTokensLegend{
|
||||
TokenTypes: semanticTokenTypes,
|
||||
TokenModifiers: []string{},
|
||||
},
|
||||
"full": true,
|
||||
},
|
||||
},
|
||||
ServerInfo: map[string]string{"name": "gasm", "version": "0.1.0"},
|
||||
})
|
||||
|
||||
case "initialized", "textDocument/didSave":
|
||||
// Notifications with nothing to do.
|
||||
|
||||
case "textDocument/didOpen":
|
||||
var p didOpenParams
|
||||
if json.Unmarshal(msg.Params, &p) == nil {
|
||||
s.docs[p.TextDocument.URI] = p.TextDocument.Text
|
||||
s.publish(p.TextDocument.URI)
|
||||
}
|
||||
|
||||
case "textDocument/didChange":
|
||||
var p didChangeParams
|
||||
if json.Unmarshal(msg.Params, &p) == nil && len(p.ContentChanges) > 0 {
|
||||
// Full sync: the last change carries the whole document.
|
||||
text := p.ContentChanges[len(p.ContentChanges)-1].Text
|
||||
s.docs[p.TextDocument.URI] = text
|
||||
s.publish(p.TextDocument.URI)
|
||||
}
|
||||
|
||||
case "textDocument/didClose":
|
||||
var p didCloseParams
|
||||
if json.Unmarshal(msg.Params, &p) == nil {
|
||||
delete(s.docs, p.TextDocument.URI)
|
||||
// Clear diagnostics for the closed document.
|
||||
s.notify("textDocument/publishDiagnostics", publishDiagnosticsParams{
|
||||
URI: p.TextDocument.URI, Diagnostics: []Diagnostic{},
|
||||
})
|
||||
}
|
||||
|
||||
case "textDocument/completion":
|
||||
var p completionParams
|
||||
json.Unmarshal(msg.Params, &p)
|
||||
s.respond(msg.ID, s.completion(p))
|
||||
|
||||
case "textDocument/hover":
|
||||
var p hoverParams
|
||||
json.Unmarshal(msg.Params, &p)
|
||||
s.respond(msg.ID, s.hover(p))
|
||||
|
||||
case "textDocument/documentSymbol":
|
||||
var p documentSymbolParams
|
||||
json.Unmarshal(msg.Params, &p)
|
||||
s.respond(msg.ID, s.documentSymbols(p))
|
||||
|
||||
case "textDocument/semanticTokens/full":
|
||||
var p semanticTokensParams
|
||||
json.Unmarshal(msg.Params, &p)
|
||||
s.respond(msg.ID, s.semanticTokens(p))
|
||||
|
||||
case "shutdown":
|
||||
s.respond(msg.ID, nil)
|
||||
|
||||
case "exit":
|
||||
return true
|
||||
|
||||
default:
|
||||
if msg.ID != nil {
|
||||
s.respondError(msg.ID, errMethodNotFound, "method not supported: "+msg.Method)
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// publish parses and lints a document and pushes the diagnostics to the client.
|
||||
func (s *Server) publish(uri string) {
|
||||
text := s.docs[uri]
|
||||
f, _ := parser.Parse(uri, text)
|
||||
cfg := lint.Config{Arch: arch.FromFilename(uriPath(uri))}
|
||||
diags := lint.File(f, cfg)
|
||||
|
||||
out := make([]Diagnostic, 0, len(diags))
|
||||
for _, d := range diags {
|
||||
out = append(out, Diagnostic{
|
||||
Range: toRange(d.Pos.Line, d.Pos.Column, d.End),
|
||||
Severity: lintSeverity(d.Severity),
|
||||
Code: d.Code,
|
||||
Source: "gasm",
|
||||
Message: d.Message,
|
||||
})
|
||||
}
|
||||
s.notify("textDocument/publishDiagnostics", publishDiagnosticsParams{URI: uri, Diagnostics: out})
|
||||
}
|
||||
|
||||
// toRange converts one-based line/column plus an optional end position into an
|
||||
// LSP range (zero-based).
|
||||
func toRange(line, col int, end token.Position) Range {
|
||||
start := Position{Line: line - 1, Character: col - 1}
|
||||
finish := start
|
||||
if end.IsValid() {
|
||||
finish = Position{Line: end.Line - 1, Character: end.Column - 1}
|
||||
} else {
|
||||
finish.Character = start.Character + 1
|
||||
}
|
||||
return Range{Start: start, End: finish}
|
||||
}
|
||||
|
||||
func lintSeverity(s lint.Severity) int {
|
||||
switch s {
|
||||
case lint.Error:
|
||||
return sevError
|
||||
case lint.Warning:
|
||||
return sevWarning
|
||||
case lint.Information:
|
||||
return sevInformation
|
||||
default:
|
||||
return sevHint
|
||||
}
|
||||
}
|
||||
|
||||
// uriPath strips a file:// scheme and returns the path component.
|
||||
func uriPath(uri string) string {
|
||||
if rest, ok := strings.CutPrefix(uri, "file://"); ok {
|
||||
return rest
|
||||
}
|
||||
return uri
|
||||
}
|
||||
@@ -0,0 +1,294 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lsp
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
const cleanDoc = "#include \"textflag.h\"\n" +
|
||||
"TEXT ·foo(SB), NOSPLIT, $0\n" +
|
||||
"\tMOVQ AX, CX\n" +
|
||||
"loop:\n" +
|
||||
"\tJMP loop\n" +
|
||||
"\tRET\n"
|
||||
|
||||
const badDoc = "#include \"textflag.h\"\n" +
|
||||
"TEXT ·foo(SB), NOSPLIT, $0\n" +
|
||||
"\tNOSUCHINSTR AX, BX\n" +
|
||||
"\tJMP missing\n"
|
||||
|
||||
// frame renders one Content-Length framed JSON-RPC message.
|
||||
func frame(id any, method string, params any) string {
|
||||
msg := map[string]any{"jsonrpc": "2.0"}
|
||||
if id != nil {
|
||||
msg["id"] = id
|
||||
}
|
||||
if method != "" {
|
||||
msg["method"] = method
|
||||
}
|
||||
if params != nil {
|
||||
msg["params"] = params
|
||||
}
|
||||
body, _ := json.Marshal(msg)
|
||||
return fmt.Sprintf("Content-Length: %d\r\n\r\n%s", len(body), body)
|
||||
}
|
||||
|
||||
// run feeds input to a server and returns every parsed output message.
|
||||
func run(t *testing.T, input string) []rpcMessage {
|
||||
t.Helper()
|
||||
var out bytes.Buffer
|
||||
srv := New(strings.NewReader(input), &out)
|
||||
if err := srv.Run(); err != nil {
|
||||
t.Fatalf("server run: %v", err)
|
||||
}
|
||||
return readFrames(t, &out)
|
||||
}
|
||||
|
||||
// readFrames parses all framed messages from a buffer.
|
||||
func readFrames(t *testing.T, r io.Reader) []rpcMessage {
|
||||
t.Helper()
|
||||
br := bufio.NewReader(r)
|
||||
var msgs []rpcMessage
|
||||
for {
|
||||
length := -1
|
||||
for {
|
||||
line, err := br.ReadString('\n')
|
||||
if err == io.EOF {
|
||||
return msgs
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("read header: %v", err)
|
||||
}
|
||||
line = strings.TrimRight(line, "\r\n")
|
||||
if line == "" {
|
||||
break
|
||||
}
|
||||
if k, v, ok := strings.Cut(line, ":"); ok && strings.EqualFold(strings.TrimSpace(k), "Content-Length") {
|
||||
length, _ = strconv.Atoi(strings.TrimSpace(v))
|
||||
}
|
||||
}
|
||||
if length < 0 {
|
||||
return msgs
|
||||
}
|
||||
body := make([]byte, length)
|
||||
if _, err := io.ReadFull(br, body); err != nil {
|
||||
t.Fatalf("read body: %v", err)
|
||||
}
|
||||
var m rpcMessage
|
||||
if err := json.Unmarshal(body, &m); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
msgs = append(msgs, m)
|
||||
}
|
||||
}
|
||||
|
||||
// session builds a standard scripting of messages around a document.
|
||||
func session(uri, text string) string {
|
||||
var b strings.Builder
|
||||
b.WriteString(frame(1, "initialize", map[string]any{"rootUri": ""}))
|
||||
b.WriteString(frame(nil, "initialized", map[string]any{}))
|
||||
b.WriteString(frame(nil, "textDocument/didOpen", map[string]any{
|
||||
"textDocument": map[string]any{
|
||||
"uri": uri, "languageId": "gasm", "version": 1, "text": text,
|
||||
},
|
||||
}))
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func findByID(msgs []rpcMessage, n int) *rpcMessage {
|
||||
for i := range msgs {
|
||||
if msgs[i].ID != nil {
|
||||
var id int
|
||||
if json.Unmarshal(*msgs[i].ID, &id) == nil && id == n {
|
||||
return &msgs[i]
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func findMethod(msgs []rpcMessage, method string) *rpcMessage {
|
||||
for i := range msgs {
|
||||
if msgs[i].Method == method {
|
||||
return &msgs[i]
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func TestInitialize(t *testing.T) {
|
||||
msgs := run(t, session("file:///f_amd64.s", cleanDoc)+frame(nil, "exit", nil))
|
||||
resp := findByID(msgs, 1)
|
||||
if resp == nil {
|
||||
t.Fatal("no initialize response")
|
||||
}
|
||||
var res initializeResult
|
||||
if err := json.Unmarshal(mustResult(t, resp), &res); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !res.Capabilities.HoverProvider || res.Capabilities.SemanticTokensProvider == nil {
|
||||
t.Fatalf("unexpected capabilities: %+v", res.Capabilities)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDiagnosticsClean(t *testing.T) {
|
||||
msgs := run(t, session("file:///f_amd64.s", cleanDoc)+frame(nil, "exit", nil))
|
||||
pub := findMethod(msgs, "textDocument/publishDiagnostics")
|
||||
if pub == nil {
|
||||
t.Fatal("no publishDiagnostics notification")
|
||||
}
|
||||
var p publishDiagnosticsParams
|
||||
json.Unmarshal(pub.Params, &p)
|
||||
if len(p.Diagnostics) != 0 {
|
||||
t.Fatalf("clean doc should have no diagnostics, got %+v", p.Diagnostics)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDiagnosticsErrors(t *testing.T) {
|
||||
msgs := run(t, session("file:///f_amd64.s", badDoc)+frame(nil, "exit", nil))
|
||||
pub := findMethod(msgs, "textDocument/publishDiagnostics")
|
||||
if pub == nil {
|
||||
t.Fatal("no publishDiagnostics notification")
|
||||
}
|
||||
var p publishDiagnosticsParams
|
||||
json.Unmarshal(pub.Params, &p)
|
||||
codes := map[string]bool{}
|
||||
for _, d := range p.Diagnostics {
|
||||
codes[d.Code] = true
|
||||
}
|
||||
if !codes["unknown-instruction"] || !codes["undefined-label"] {
|
||||
t.Fatalf("expected unknown-instruction and undefined-label, got %+v", p.Diagnostics)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCompletion(t *testing.T) {
|
||||
in := session("file:///f_amd64.s", cleanDoc) +
|
||||
frame(2, "textDocument/completion", map[string]any{
|
||||
"textDocument": map[string]any{"uri": "file:///f_amd64.s"},
|
||||
"position": map[string]any{"line": 2, "character": 1},
|
||||
}) + frame(nil, "exit", nil)
|
||||
msgs := run(t, in)
|
||||
resp := findByID(msgs, 2)
|
||||
if resp == nil {
|
||||
t.Fatal("no completion response")
|
||||
}
|
||||
var items []CompletionItem
|
||||
if err := json.Unmarshal(mustResult(t, resp), &items); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
labels := map[string]bool{}
|
||||
for _, it := range items {
|
||||
labels[it.Label] = true
|
||||
}
|
||||
for _, want := range []string{"MOVQ", "AX", "TEXT", "NOSPLIT", "loop"} {
|
||||
if !labels[want] {
|
||||
t.Errorf("completion missing %q", want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestHover(t *testing.T) {
|
||||
in := session("file:///f_amd64.s", cleanDoc) +
|
||||
frame(3, "textDocument/hover", map[string]any{
|
||||
"textDocument": map[string]any{"uri": "file:///f_amd64.s"},
|
||||
"position": map[string]any{"line": 2, "character": 2}, // on MOVQ
|
||||
}) + frame(nil, "exit", nil)
|
||||
msgs := run(t, in)
|
||||
resp := findByID(msgs, 3)
|
||||
if resp == nil {
|
||||
t.Fatal("no hover response")
|
||||
}
|
||||
var h Hover
|
||||
if err := json.Unmarshal(mustResult(t, resp), &h); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(h.Contents.Value, "MOVQ") {
|
||||
t.Fatalf("hover = %q, want MOVQ docs", h.Contents.Value)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDocumentSymbols(t *testing.T) {
|
||||
in := session("file:///f_amd64.s", cleanDoc) +
|
||||
frame(4, "textDocument/documentSymbol", map[string]any{
|
||||
"textDocument": map[string]any{"uri": "file:///f_amd64.s"},
|
||||
}) + frame(nil, "exit", nil)
|
||||
msgs := run(t, in)
|
||||
resp := findByID(msgs, 4)
|
||||
if resp == nil {
|
||||
t.Fatal("no documentSymbol response")
|
||||
}
|
||||
var syms []DocumentSymbol
|
||||
if err := json.Unmarshal(mustResult(t, resp), &syms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(syms) == 0 || syms[0].Name != "foo" {
|
||||
t.Fatalf("symbols = %+v, want function foo", syms)
|
||||
}
|
||||
found := false
|
||||
for _, c := range syms[0].Children {
|
||||
if c.Name == "loop" {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Errorf("function foo should contain label loop: %+v", syms[0].Children)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSemanticTokens(t *testing.T) {
|
||||
in := session("file:///f_amd64.s", cleanDoc) +
|
||||
frame(5, "textDocument/semanticTokens/full", map[string]any{
|
||||
"textDocument": map[string]any{"uri": "file:///f_amd64.s"},
|
||||
}) + frame(nil, "exit", nil)
|
||||
msgs := run(t, in)
|
||||
resp := findByID(msgs, 5)
|
||||
if resp == nil {
|
||||
t.Fatal("no semanticTokens response")
|
||||
}
|
||||
var st SemanticTokens
|
||||
if err := json.Unmarshal(mustResult(t, resp), &st); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(st.Data) == 0 || len(st.Data)%5 != 0 {
|
||||
t.Fatalf("semantic tokens data invalid: len=%d", len(st.Data))
|
||||
}
|
||||
// There must be at least one "function" (mnemonic) and one "comment"-free
|
||||
// keyword token; sanity-check that a MOVQ-classified function token exists.
|
||||
seenFunction := false
|
||||
for i := 3; i < len(st.Data); i += 5 {
|
||||
if st.Data[i] == stFunction {
|
||||
seenFunction = true
|
||||
}
|
||||
}
|
||||
if !seenFunction {
|
||||
t.Error("expected at least one function (mnemonic) semantic token")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMethodNotFound(t *testing.T) {
|
||||
in := frame(9, "bogus/method", map[string]any{}) + frame(nil, "exit", nil)
|
||||
msgs := run(t, in)
|
||||
resp := findByID(msgs, 9)
|
||||
if resp == nil || resp.Error == nil || resp.Error.Code != errMethodNotFound {
|
||||
t.Fatalf("expected method-not-found error, got %+v", resp)
|
||||
}
|
||||
}
|
||||
|
||||
// mustResult re-marshals a response result into raw JSON for typed decoding.
|
||||
func mustResult(t *testing.T, m *rpcMessage) []byte {
|
||||
t.Helper()
|
||||
b, err := json.Marshal(m.Result)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return b
|
||||
}
|
||||
@@ -0,0 +1,619 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package parser turns a GAsm token stream into an abstract syntax tree. It
|
||||
// is line-oriented, matching how the Plan 9 assembler itself reads a file, and
|
||||
// tolerant: a malformed line is reported as an error but never aborts the
|
||||
// parse of the rest of the file.
|
||||
package parser
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// Error is a single parse diagnostic.
|
||||
type Error struct {
|
||||
Pos token.Position
|
||||
Msg string
|
||||
}
|
||||
|
||||
func (e Error) Error() string {
|
||||
return e.Pos.String() + ": " + e.Msg
|
||||
}
|
||||
|
||||
// Parse scans and parses src, returning the file and any diagnostics. The
|
||||
// returned file is usable even when errors is non-empty.
|
||||
func Parse(path, src string) (*ast.File, []error) {
|
||||
toks := lexer.Tokenize(src)
|
||||
lines := splitLines(toks)
|
||||
p := &state{path: path}
|
||||
p.parse(lines)
|
||||
return p.file, p.errs
|
||||
}
|
||||
|
||||
// state carries the mutable context for one parse.
|
||||
type state struct {
|
||||
path string
|
||||
file *ast.File
|
||||
errs []error
|
||||
|
||||
curText *ast.Text // the TEXT body labels/instructions attach to
|
||||
pending []string // comment lines awaiting a TEXT to become its Doc
|
||||
}
|
||||
|
||||
func (p *state) errorf(pos token.Position, format string, args ...any) {
|
||||
p.errs = append(p.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)})
|
||||
}
|
||||
|
||||
// splitLines groups the token stream into lines, dropping the Newline tokens.
|
||||
func splitLines(toks []token.Token) [][]token.Token {
|
||||
var lines [][]token.Token
|
||||
var cur []token.Token
|
||||
for _, t := range toks {
|
||||
if t.Kind == token.EOF {
|
||||
break
|
||||
}
|
||||
if t.Kind == token.Newline {
|
||||
lines = append(lines, cur)
|
||||
cur = nil
|
||||
continue
|
||||
}
|
||||
cur = append(cur, t)
|
||||
}
|
||||
if len(cur) > 0 {
|
||||
lines = append(lines, cur)
|
||||
}
|
||||
return lines
|
||||
}
|
||||
|
||||
func (p *state) parse(lines [][]token.Token) {
|
||||
p.file = &ast.File{Path: p.path, Macros: map[string]bool{}}
|
||||
for _, line := range lines {
|
||||
line = trimSpace(line)
|
||||
if len(line) == 0 {
|
||||
// Blank line: a comment block ends here only if it was not
|
||||
// directly preceding a declaration; keep pending doc intact
|
||||
// across a single blank line is not desired, so reset.
|
||||
p.pending = nil
|
||||
continue
|
||||
}
|
||||
p.parseLine(line)
|
||||
}
|
||||
}
|
||||
|
||||
// trimSpace is a no-op placeholder kept for symmetry; the lexer already drops
|
||||
// horizontal whitespace, but this documents the intent.
|
||||
func trimSpace(line []token.Token) []token.Token { return line }
|
||||
|
||||
func (p *state) parseLine(line []token.Token) {
|
||||
first := line[0]
|
||||
|
||||
// A lone comment accumulates as documentation for a following TEXT.
|
||||
if len(line) == 1 && first.Kind == token.Comment {
|
||||
p.pending = append(p.pending, commentText(first.Text))
|
||||
return
|
||||
}
|
||||
|
||||
// Preprocessor line.
|
||||
if first.Kind == token.Hash {
|
||||
p.parsePreproc(line)
|
||||
p.pending = nil
|
||||
return
|
||||
}
|
||||
|
||||
// Directives and instructions are identified by a leading identifier.
|
||||
if first.Kind == token.Ident {
|
||||
switch first.Text {
|
||||
case "TEXT":
|
||||
p.parseText(line)
|
||||
return
|
||||
case "GLOBL":
|
||||
p.file.Decls = append(p.file.Decls, p.parseGlobl(line))
|
||||
p.curText = nil
|
||||
p.pending = nil
|
||||
return
|
||||
case "DATA":
|
||||
p.file.Decls = append(p.file.Decls, p.parseData(line))
|
||||
p.curText = nil
|
||||
p.pending = nil
|
||||
return
|
||||
}
|
||||
|
||||
// Label (ident immediately followed by a colon).
|
||||
if len(line) >= 2 && line[1].Kind == token.Colon {
|
||||
lbl := &ast.Label{Name: line[0], Colon: line[1]}
|
||||
p.addStmt(lbl)
|
||||
// A label may share its line with an instruction: "loop: MOVQ …".
|
||||
if rest := dropColon(line); len(rest) > 0 {
|
||||
p.parseInstr(rest)
|
||||
}
|
||||
p.pending = nil
|
||||
return
|
||||
}
|
||||
|
||||
// Otherwise it is an instruction.
|
||||
p.parseInstr(line)
|
||||
p.pending = nil
|
||||
return
|
||||
}
|
||||
|
||||
p.errorf(first.Pos, "unexpected token %s at start of line", first.Kind)
|
||||
p.pending = nil
|
||||
}
|
||||
|
||||
// dropColon removes the leading "ident :" of a label, returning the remainder.
|
||||
func dropColon(line []token.Token) []token.Token {
|
||||
if len(line) >= 2 && line[1].Kind == token.Colon {
|
||||
return line[2:]
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (p *state) addStmt(s ast.Stmt) {
|
||||
if p.curText != nil {
|
||||
p.curText.Body = append(p.curText.Body, s)
|
||||
return
|
||||
}
|
||||
p.file.Orphans = append(p.file.Orphans, s)
|
||||
}
|
||||
|
||||
func (p *state) parsePreproc(line []token.Token) {
|
||||
hash := line[0]
|
||||
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
|
||||
line[2].Kind == token.String {
|
||||
p.file.Decls = append(p.file.Decls, &ast.Include{
|
||||
Hash: hash,
|
||||
Name: line[1],
|
||||
Header: line[2],
|
||||
})
|
||||
return
|
||||
}
|
||||
// Record macro names so the linter can recognise their invocations.
|
||||
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "define" &&
|
||||
line[2].Kind == token.Ident {
|
||||
p.file.Macros[line[2].Text] = true
|
||||
}
|
||||
p.file.Decls = append(p.file.Decls, &ast.Preproc{
|
||||
Hash: hash,
|
||||
Raw: joinRaw(line[1:]),
|
||||
})
|
||||
}
|
||||
|
||||
func (p *state) parseText(line []token.Token) {
|
||||
text := &ast.Text{Keyword: line[0]}
|
||||
if len(p.pending) > 0 {
|
||||
text.Doc = strings.Join(p.pending, "\n")
|
||||
}
|
||||
p.pending = nil
|
||||
|
||||
rest := line[1:]
|
||||
sym, n := parseSymbolPrefix(rest)
|
||||
if sym == nil {
|
||||
p.errorf(line[0].Pos, "TEXT missing a symbol name")
|
||||
}
|
||||
text.Name = sym
|
||||
rest = rest[n:]
|
||||
|
||||
// Consume flags (identifiers, possibly '|' joined) up to the frame '$'.
|
||||
rest = skipComma(rest)
|
||||
for len(rest) > 0 && rest[0].Kind != token.Dollar {
|
||||
if rest[0].Kind == token.Ident {
|
||||
text.Flags = append(text.Flags, rest[0].Text)
|
||||
}
|
||||
// Commas, '|' (Illegal) and anything else between flags is skipped.
|
||||
rest = rest[1:]
|
||||
}
|
||||
|
||||
// Frame: $number ; optional args: -number.
|
||||
if len(rest) > 0 && rest[0].Kind == token.Dollar {
|
||||
text.Frame = parseOperand(rest[:2]) // "$" "number"
|
||||
rest = rest[2:]
|
||||
if len(rest) >= 2 && rest[0].Kind == token.Minus && rest[1].Kind == token.Number {
|
||||
text.Args = &ast.Operand{
|
||||
Kind: ast.OpImmediate,
|
||||
Imm: ast.Immediate{Val: parseInt(rest[1].Text), HasVal: true},
|
||||
Raw: "-" + rest[1].Text,
|
||||
Pos: rest[0].Pos,
|
||||
}
|
||||
rest = rest[2:]
|
||||
}
|
||||
}
|
||||
|
||||
p.file.Decls = append(p.file.Decls, text)
|
||||
p.curText = text
|
||||
}
|
||||
|
||||
func (p *state) parseGlobl(line []token.Token) *ast.Globl {
|
||||
g := &ast.Globl{Keyword: line[0]}
|
||||
rest := skipComma(line[1:])
|
||||
sym, n := parseSymbolPrefix(rest)
|
||||
g.Name = sym
|
||||
rest = skipComma(rest[n:])
|
||||
for len(rest) > 0 && rest[0].Kind != token.Dollar {
|
||||
if rest[0].Kind == token.Ident {
|
||||
g.Flags = append(g.Flags, rest[0].Text)
|
||||
}
|
||||
rest = rest[1:]
|
||||
}
|
||||
if len(rest) > 0 && rest[0].Kind == token.Dollar {
|
||||
g.Size = parseOperand(rest)
|
||||
}
|
||||
return g
|
||||
}
|
||||
|
||||
func (p *state) parseData(line []token.Token) *ast.Data {
|
||||
d := &ast.Data{Keyword: line[0]}
|
||||
rest := line[1:]
|
||||
|
||||
// The name may carry a /width suffix: ·idx+0(SB)/4. Split it off the
|
||||
// symbol group that precedes the first top-level comma.
|
||||
nameGroup, valuePart := splitFirstComma(rest)
|
||||
nameGroup, width := splitTrailingWidth(nameGroup)
|
||||
sym, _ := parseSymbolPrefix(nameGroup)
|
||||
d.Name = sym
|
||||
d.Width = width
|
||||
if len(valuePart) > 0 {
|
||||
d.Value = parseOperand(stripComment(valuePart))
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
func (p *state) parseInstr(line []token.Token) {
|
||||
body, comment := splitTrailingComment(line)
|
||||
if len(body) == 0 {
|
||||
return
|
||||
}
|
||||
instr := &ast.Instr{Mnemonic: body[0], Comment: comment}
|
||||
for _, grp := range splitOperands(body[1:]) {
|
||||
if op := parseOperand(grp); op != nil {
|
||||
instr.Operands = append(instr.Operands, op)
|
||||
}
|
||||
}
|
||||
p.addStmt(instr)
|
||||
}
|
||||
|
||||
// --- symbol parsing ---------------------------------------------------------
|
||||
|
||||
var pseudoRegs = map[string]bool{"FP": true, "SP": true, "SB": true, "PC": true}
|
||||
|
||||
// parseSymbolPrefix parses a leading symbol reference from g and returns it
|
||||
// together with the number of tokens consumed. It returns (nil, 0) when no
|
||||
// symbol is present.
|
||||
func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) {
|
||||
if len(g) == 0 || g[0].Kind != token.Ident {
|
||||
return nil, 0
|
||||
}
|
||||
sym := &ast.Symbol{Pos: g[0].Pos}
|
||||
i := 0
|
||||
setName(g[0].Text, sym)
|
||||
i++
|
||||
|
||||
if i+1 < len(g) && g[i].Kind == token.LAngle && g[i+1].Kind == token.RAngle {
|
||||
sym.Static = true
|
||||
i += 2
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Plus {
|
||||
i++
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
sym.Offset, sym.HasOff = parseInt(g[i].Text), true
|
||||
i++
|
||||
}
|
||||
}
|
||||
if i+2 < len(g) && g[i].Kind == token.LParen && g[i+1].Kind == token.Ident &&
|
||||
pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen {
|
||||
sym.Pseudo = g[i+1].Text
|
||||
i += 3
|
||||
}
|
||||
sym.Raw = joinRaw(g[:i])
|
||||
return sym, i
|
||||
}
|
||||
|
||||
// setName splits a raw identifier on the middle dot into package and name.
|
||||
func setName(raw string, sym *ast.Symbol) {
|
||||
const dot = "\u00B7"
|
||||
switch {
|
||||
case strings.HasPrefix(raw, dot):
|
||||
sym.Pkg = ""
|
||||
sym.Name = strings.TrimPrefix(raw, dot)
|
||||
case strings.Contains(raw, dot):
|
||||
parts := strings.SplitN(raw, dot, 2)
|
||||
sym.Pkg = parts[0]
|
||||
sym.Name = parts[1]
|
||||
default:
|
||||
sym.Name = raw
|
||||
}
|
||||
}
|
||||
|
||||
// --- operand parsing --------------------------------------------------------
|
||||
|
||||
// parseOperand parses one operand group into an Operand.
|
||||
func parseOperand(g []token.Token) *ast.Operand {
|
||||
g = stripComment(g)
|
||||
if len(g) == 0 {
|
||||
return nil
|
||||
}
|
||||
op := &ast.Operand{Raw: joinRaw(g), Pos: g[0].Pos}
|
||||
if g[0].Kind == token.Dollar {
|
||||
op.Kind = ast.OpImmediate
|
||||
op.Imm = parseImmediate(g[1:])
|
||||
return op
|
||||
}
|
||||
op.Kind = ast.OpAddr
|
||||
op.Addr = parseAddress(g)
|
||||
return op
|
||||
}
|
||||
|
||||
// parseImmediate parses the tokens following a '$'.
|
||||
func parseImmediate(g []token.Token) ast.Immediate {
|
||||
var imm ast.Immediate
|
||||
if len(g) == 0 {
|
||||
return imm
|
||||
}
|
||||
// $sym(…) form.
|
||||
if findPseudoParen(g) >= 0 || (g[0].Kind == token.Ident) {
|
||||
if sym, n := parseSymbolPrefix(g); sym != nil && (sym.Pseudo != "" || sym.Static) {
|
||||
imm.Sym = sym
|
||||
_ = n
|
||||
return imm
|
||||
}
|
||||
}
|
||||
i := 0
|
||||
if g[i].Kind == token.Minus {
|
||||
imm.Neg = true
|
||||
i++
|
||||
} else if g[i].Kind == token.Plus {
|
||||
i++
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
text := g[i].Text
|
||||
if v, ok := tryInt(text); ok {
|
||||
imm.Val = v
|
||||
imm.HasVal = true
|
||||
} else {
|
||||
imm.Float = text
|
||||
}
|
||||
i++
|
||||
} else if i < len(g) && (g[i].Kind == token.String || g[i].Kind == token.Rune) {
|
||||
imm.Str = g[i].Text
|
||||
i++
|
||||
}
|
||||
return imm
|
||||
}
|
||||
|
||||
// parseAddress parses a non-immediate operand.
|
||||
func parseAddress(g []token.Token) ast.Address {
|
||||
var addr ast.Address
|
||||
if len(g) == 0 {
|
||||
return addr
|
||||
}
|
||||
// Symbol-with-pseudo form: name[<>][+off](PSEUDO).
|
||||
if idx := findPseudoParen(g); idx >= 0 {
|
||||
sym, _ := parseSymbolPrefix(g[:idx+3])
|
||||
addr.Sym = sym
|
||||
return addr
|
||||
}
|
||||
|
||||
i := 0
|
||||
// Optional leading displacement before a '(' base group.
|
||||
if isSignedNumber(g, i) && i+1 < len(g) && g[i+1].Kind == token.LParen {
|
||||
neg := false
|
||||
if g[i].Kind == token.Minus {
|
||||
neg = true
|
||||
i++
|
||||
} else if g[i].Kind == token.Plus {
|
||||
i++
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
addr.Offset = parseInt(g[i].Text)
|
||||
addr.HasOff = true
|
||||
if neg {
|
||||
addr.Offset = -addr.Offset
|
||||
}
|
||||
i++
|
||||
}
|
||||
}
|
||||
// First parenthesised group: the base register.
|
||||
if i < len(g) && g[i].Kind == token.LParen {
|
||||
i++
|
||||
if i < len(g) && g[i].Kind == token.Ident {
|
||||
addr.Base = g[i].Text
|
||||
i++
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.RParen {
|
||||
i++
|
||||
}
|
||||
}
|
||||
// Optional second group: (index*scale) or (index).
|
||||
if i < len(g) && g[i].Kind == token.LParen {
|
||||
i++
|
||||
if i < len(g) && g[i].Kind == token.Ident {
|
||||
addr.Index = g[i].Text
|
||||
i++
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Star {
|
||||
i++
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
addr.Scale = int(parseInt(g[i].Text))
|
||||
i++
|
||||
}
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.RParen {
|
||||
i++
|
||||
}
|
||||
}
|
||||
// Bare name (register, label or symbol) possibly with an arm64 shift.
|
||||
if addr.Base == "" && addr.Sym == nil && g[0].Kind == token.Ident {
|
||||
sym := &ast.Symbol{Pos: g[0].Pos}
|
||||
setName(g[0].Text, sym)
|
||||
sym.Raw = g[0].Text
|
||||
addr.Sym = sym
|
||||
i = 1
|
||||
}
|
||||
// Any remaining tokens form a verbatim shift/extension suffix (arm64).
|
||||
if i > 0 && i < len(g) {
|
||||
addr.Shift = joinRaw(g[i:])
|
||||
}
|
||||
return addr
|
||||
}
|
||||
|
||||
// findPseudoParen returns the index of the '(' that begins a (PSEUDO) group,
|
||||
// or -1 when none is present.
|
||||
func findPseudoParen(g []token.Token) int {
|
||||
for i := 0; i+2 < len(g); i++ {
|
||||
if g[i].Kind == token.LParen && g[i+1].Kind == token.Ident &&
|
||||
pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// --- token helpers ----------------------------------------------------------
|
||||
|
||||
// splitOperands splits a token slice on top-level commas (commas outside any
|
||||
// parenthesis group).
|
||||
func splitOperands(g []token.Token) [][]token.Token {
|
||||
var out [][]token.Token
|
||||
var cur []token.Token
|
||||
depth := 0
|
||||
for _, t := range g {
|
||||
switch t.Kind {
|
||||
case token.LParen:
|
||||
depth++
|
||||
cur = append(cur, t)
|
||||
case token.RParen:
|
||||
depth--
|
||||
cur = append(cur, t)
|
||||
case token.Comma:
|
||||
if depth == 0 {
|
||||
if len(cur) > 0 {
|
||||
out = append(out, cur)
|
||||
}
|
||||
cur = nil
|
||||
} else {
|
||||
cur = append(cur, t)
|
||||
}
|
||||
case token.Comment:
|
||||
// A comment terminates the operand list.
|
||||
if len(cur) > 0 {
|
||||
out = append(out, cur)
|
||||
}
|
||||
return out
|
||||
default:
|
||||
cur = append(cur, t)
|
||||
}
|
||||
}
|
||||
if len(cur) > 0 {
|
||||
out = append(out, cur)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// splitFirstComma splits g at the first top-level comma.
|
||||
func splitFirstComma(g []token.Token) (before, after []token.Token) {
|
||||
depth := 0
|
||||
for i, t := range g {
|
||||
switch t.Kind {
|
||||
case token.LParen:
|
||||
depth++
|
||||
case token.RParen:
|
||||
depth--
|
||||
case token.Comma:
|
||||
if depth == 0 {
|
||||
return g[:i], g[i+1:]
|
||||
}
|
||||
}
|
||||
}
|
||||
return g, nil
|
||||
}
|
||||
|
||||
// splitTrailingComment separates a trailing comment from the line body.
|
||||
func splitTrailingComment(g []token.Token) (body []token.Token, comment string) {
|
||||
for i, t := range g {
|
||||
if t.Kind == token.Comment {
|
||||
return g[:i], commentText(t.Text)
|
||||
}
|
||||
}
|
||||
return g, ""
|
||||
}
|
||||
|
||||
// stripComment removes a trailing comment token from a group.
|
||||
func stripComment(g []token.Token) []token.Token {
|
||||
for i, t := range g {
|
||||
if t.Kind == token.Comment {
|
||||
return g[:i]
|
||||
}
|
||||
}
|
||||
return g
|
||||
}
|
||||
|
||||
// splitTrailingWidth removes a "/width" suffix from a DATA name group.
|
||||
func splitTrailingWidth(g []token.Token) ([]token.Token, int) {
|
||||
for i := 0; i+1 < len(g); i++ {
|
||||
if g[i].Kind == token.Slash && g[i+1].Kind == token.Number {
|
||||
return g[:i], int(parseInt(g[i+1].Text))
|
||||
}
|
||||
}
|
||||
return g, 0
|
||||
}
|
||||
|
||||
func skipComma(g []token.Token) []token.Token {
|
||||
if len(g) > 0 && g[0].Kind == token.Comma {
|
||||
return g[1:]
|
||||
}
|
||||
return g
|
||||
}
|
||||
|
||||
func isSignedNumber(g []token.Token, i int) bool {
|
||||
if i >= len(g) {
|
||||
return false
|
||||
}
|
||||
if g[i].Kind == token.Number {
|
||||
return true
|
||||
}
|
||||
if (g[i].Kind == token.Minus || g[i].Kind == token.Plus) &&
|
||||
i+1 < len(g) && g[i+1].Kind == token.Number {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func joinRaw(g []token.Token) string {
|
||||
parts := make([]string, len(g))
|
||||
for i, t := range g {
|
||||
parts[i] = t.Text
|
||||
}
|
||||
return strings.Join(parts, " ")
|
||||
}
|
||||
|
||||
// commentText removes a leading // or /* marker from a comment token's text.
|
||||
func commentText(s string) string {
|
||||
if strings.HasPrefix(s, "//") {
|
||||
return strings.TrimSpace(strings.TrimPrefix(s, "//"))
|
||||
}
|
||||
if strings.HasPrefix(s, "/*") {
|
||||
s = strings.TrimPrefix(s, "/*")
|
||||
s = strings.TrimSuffix(s, "*/")
|
||||
return strings.TrimSpace(s)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func parseInt(text string) int64 {
|
||||
v, _ := tryInt(text)
|
||||
return v
|
||||
}
|
||||
|
||||
func tryInt(text string) (int64, bool) {
|
||||
v, err := strconv.ParseInt(text, 0, 64)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
return v, true
|
||||
}
|
||||
@@ -0,0 +1,241 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package parser
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
)
|
||||
|
||||
func mustParse(t *testing.T, path string) *ast.File {
|
||||
t.Helper()
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("read %s: %v", path, err)
|
||||
}
|
||||
file, errs := Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse %s: %v", path, errs)
|
||||
}
|
||||
return file
|
||||
}
|
||||
|
||||
func texts(f *ast.File) []*ast.Text {
|
||||
var out []*ast.Text
|
||||
for _, d := range f.Decls {
|
||||
if t, ok := d.(*ast.Text); ok {
|
||||
out = append(out, t)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func TestParseSample(t *testing.T) {
|
||||
f := mustParse(t, "../testdata/sample_amd64.s")
|
||||
|
||||
// Includes, GLOBL/DATA and two TEXT functions.
|
||||
var includes, globls, datas int
|
||||
for _, d := range f.Decls {
|
||||
switch d.(type) {
|
||||
case *ast.Include:
|
||||
includes++
|
||||
case *ast.Globl:
|
||||
globls++
|
||||
case *ast.Data:
|
||||
datas++
|
||||
}
|
||||
}
|
||||
if includes != 1 {
|
||||
t.Errorf("includes = %d, want 1", includes)
|
||||
}
|
||||
if globls != 2 {
|
||||
t.Errorf("globls = %d, want 2", globls)
|
||||
}
|
||||
if datas != 4 {
|
||||
t.Errorf("datas = %d, want 4", datas)
|
||||
}
|
||||
|
||||
txts := texts(f)
|
||||
if len(txts) != 2 {
|
||||
t.Fatalf("text functions = %d, want 2", len(txts))
|
||||
}
|
||||
|
||||
fn := txts[0]
|
||||
if fn.Name.Name != "analyzeO1RangeAVX2" {
|
||||
t.Errorf("name = %q, want analyzeO1RangeAVX2", fn.Name.Name)
|
||||
}
|
||||
if fn.Name.Pseudo != "SB" {
|
||||
t.Errorf("pseudo = %q, want SB", fn.Name.Pseudo)
|
||||
}
|
||||
if len(fn.Flags) != 1 || fn.Flags[0] != "NOSPLIT" {
|
||||
t.Errorf("flags = %v, want [NOSPLIT]", fn.Flags)
|
||||
}
|
||||
if fn.Frame == nil || !fn.Frame.Imm.HasVal || fn.Frame.Imm.Val != 0 {
|
||||
t.Errorf("frame = %+v, want $0", fn.Frame)
|
||||
}
|
||||
if fn.Args == nil || fn.Args.Imm.Val != 65 {
|
||||
t.Errorf("args = %+v, want 65", fn.Args)
|
||||
}
|
||||
if fn.Doc == "" {
|
||||
t.Error("expected a doc comment on the first TEXT")
|
||||
}
|
||||
|
||||
// The body must contain the two labels vec1 and vec1done.
|
||||
labels := map[string]bool{}
|
||||
for _, s := range fn.Body {
|
||||
if l, ok := s.(*ast.Label); ok {
|
||||
labels[l.Name.Text] = true
|
||||
}
|
||||
}
|
||||
for _, want := range []string{"vec1", "vec1done"} {
|
||||
if !labels[want] {
|
||||
t.Errorf("missing label %q", want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestOperandStructure(t *testing.T) {
|
||||
f := mustParse(t, "../testdata/sample_amd64.s")
|
||||
fn := texts(f)[0]
|
||||
|
||||
// Index instructions by mnemonic for targeted checks.
|
||||
byMnem := map[string]*ast.Instr{}
|
||||
for _, s := range fn.Body {
|
||||
if in, ok := s.(*ast.Instr); ok {
|
||||
byMnem[in.Mnemonic.Text] = in
|
||||
}
|
||||
}
|
||||
|
||||
// MOVQ swin_base+0(FP), SI — the first MOVQ in the body.
|
||||
var mov *ast.Instr
|
||||
for _, s := range fn.Body {
|
||||
if in, ok := s.(*ast.Instr); ok && in.Mnemonic.Text == "MOVQ" {
|
||||
mov = in
|
||||
break
|
||||
}
|
||||
}
|
||||
if mov == nil {
|
||||
t.Fatal("MOVQ not found")
|
||||
}
|
||||
src := mov.Operands[0]
|
||||
if src.Kind != ast.OpAddr || src.Addr.Sym == nil {
|
||||
t.Fatalf("src operand = %+v, want symbol address", src)
|
||||
}
|
||||
if src.Addr.Sym.Name != "swin_base" || src.Addr.Sym.Pseudo != "FP" || src.Addr.Sym.Offset != 0 {
|
||||
t.Errorf("src symbol = %+v, want swin_base+0(FP)", src.Addr.Sym)
|
||||
}
|
||||
if mov.Operands[1].Addr.Sym.Name != "SI" {
|
||||
t.Errorf("dst = %+v, want SI", mov.Operands[1].Addr)
|
||||
}
|
||||
|
||||
// LEAQ (SI)(BX*4), R9
|
||||
leaq := byMnem["LEAQ"]
|
||||
if leaq == nil {
|
||||
t.Fatal("LEAQ not found")
|
||||
}
|
||||
mem := leaq.Operands[0].Addr
|
||||
if mem.Base != "SI" || mem.Index != "BX" || mem.Scale != 4 {
|
||||
t.Errorf("LEAQ addr = %+v, want base SI index BX scale 4", mem)
|
||||
}
|
||||
|
||||
// ANDQ $-8, R10
|
||||
andq := byMnem["ANDQ"]
|
||||
if andq == nil {
|
||||
t.Fatal("ANDQ not found")
|
||||
}
|
||||
imm := andq.Operands[0]
|
||||
if imm.Kind != ast.OpImmediate || !imm.Imm.Neg || imm.Imm.Val != 8 {
|
||||
t.Errorf("ANDQ imm = %+v, want -8", imm.Imm)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAVX512Operands(t *testing.T) {
|
||||
f := mustParse(t, "../testdata/sample_amd64.s")
|
||||
fn := texts(f)[1]
|
||||
byMnem := map[string]*ast.Instr{}
|
||||
for _, s := range fn.Body {
|
||||
if in, ok := s.(*ast.Instr); ok {
|
||||
byMnem[in.Mnemonic.Text] = in
|
||||
}
|
||||
}
|
||||
|
||||
// VALIGND $15, Z9, Z0, Z1 — four operands.
|
||||
val := byMnem["VALIGND"]
|
||||
if val == nil {
|
||||
t.Fatal("VALIGND not found")
|
||||
}
|
||||
if len(val.Operands) != 4 {
|
||||
t.Errorf("VALIGND operands = %d, want 4", len(val.Operands))
|
||||
}
|
||||
if val.Operands[0].Kind != ast.OpImmediate || val.Operands[0].Imm.Val != 15 {
|
||||
t.Errorf("VALIGND first operand = %+v, want $15", val.Operands[0])
|
||||
}
|
||||
|
||||
// VMOVDQU32 Z0, 4(SI)(AX*1)
|
||||
vmov := byMnem["VMOVDQU32"]
|
||||
if vmov == nil {
|
||||
t.Fatal("VMOVDQU32 not found")
|
||||
}
|
||||
dst := vmov.Operands[len(vmov.Operands)-1].Addr
|
||||
if dst.Offset != 4 || dst.Base != "SI" || dst.Index != "AX" || dst.Scale != 1 {
|
||||
t.Errorf("VMOVDQU32 dst = %+v, want 4(SI)(AX*1)", dst)
|
||||
}
|
||||
|
||||
// KTESTW K1, K1 — mask registers parse as bare names.
|
||||
kt := byMnem["KTESTW"]
|
||||
if kt == nil || len(kt.Operands) != 2 {
|
||||
t.Fatalf("KTESTW = %+v, want two operands", kt)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDataWidthAndStatic(t *testing.T) {
|
||||
f := mustParse(t, "../testdata/sample_amd64.s")
|
||||
var datas []*ast.Data
|
||||
for _, d := range f.Decls {
|
||||
if dd, ok := d.(*ast.Data); ok {
|
||||
datas = append(datas, dd)
|
||||
}
|
||||
}
|
||||
if datas[0].Width != 4 {
|
||||
t.Errorf("first DATA width = %d, want 4", datas[0].Width)
|
||||
}
|
||||
if datas[0].Name.Pseudo != "SB" || datas[0].Name.Offset != 0 {
|
||||
t.Errorf("first DATA name = %+v, want +0(SB)", datas[0].Name)
|
||||
}
|
||||
if datas[0].Value.Kind != ast.OpImmediate || datas[0].Value.Imm.Val != 1 {
|
||||
t.Errorf("first DATA value = %+v, want $1", datas[0].Value)
|
||||
}
|
||||
// The mask24<> entries are static.
|
||||
if !datas[2].Name.Static {
|
||||
t.Errorf("mask24 DATA should be static, got %+v", datas[2].Name)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseRealGoLibraries parses every .s file in the sibling go-libraries
|
||||
// repository when it is checked out, asserting a clean, error-free parse. It
|
||||
// is skipped when the repository is not present.
|
||||
func TestParseRealGoLibraries(t *testing.T) {
|
||||
matches, _ := filepath.Glob("../../go-libraries/go-*/*.s")
|
||||
if len(matches) == 0 {
|
||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
||||
}
|
||||
for _, path := range matches {
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("read %s: %v", path, err)
|
||||
}
|
||||
file, errs := Parse(path, string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Errorf("parse %s: %v", path, errs)
|
||||
continue
|
||||
}
|
||||
if len(texts(file)) == 0 {
|
||||
t.Errorf("parse %s: no TEXT functions found", path)
|
||||
}
|
||||
t.Logf("%s: %d decls, %d functions", filepath.Base(path), len(file.Decls), len(texts(file)))
|
||||
}
|
||||
}
|
||||
Vendored
+56
@@ -0,0 +1,56 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// Index vector for the order-2 ramp.
|
||||
GLOBL ·idx16(SB), RODATA, $64
|
||||
DATA ·idx16+0(SB)/4, $1
|
||||
DATA ·idx16+4(SB)/4, $2
|
||||
|
||||
// A file-local (static) constant table.
|
||||
GLOBL mask24<>(SB), RODATA, $16
|
||||
DATA mask24<>+0(SB)/4, $0x80020100
|
||||
DATA mask24<>+4(SB)/4, $0x80050403
|
||||
|
||||
// func analyzeO1RangeAVX2(swin []int32, dstP []uint32, hist *[32]uint16) (partSum uint64, overflow bool)
|
||||
TEXT ·analyzeO1RangeAVX2(SB), NOSPLIT, $0-65
|
||||
MOVQ swin_base+0(FP), SI
|
||||
MOVQ dstP_base+24(FP), DI
|
||||
MOVQ dstP_len+32(FP), BX
|
||||
MOVQ hist+48(FP), R13
|
||||
|
||||
VPCMPEQD Y0, Y0, Y0
|
||||
VPSLLD $31, Y0, Y0
|
||||
|
||||
LEAQ (SI)(BX*4), R9
|
||||
MOVQ BX, R10
|
||||
ANDQ $-8, R10
|
||||
|
||||
vec1:
|
||||
CMPQ SI, R10
|
||||
JGE vec1done
|
||||
VMOVDQU (SI), Y1
|
||||
VMOVDQU 4(SI), Y2
|
||||
VPSUBD Y1, Y2, Y3
|
||||
ADDQ $32, SI
|
||||
JMP vec1
|
||||
vec1done:
|
||||
|
||||
MOVQ AX, partSum+56(FP)
|
||||
MOVB AL, overflow+64(FP)
|
||||
VZEROUPPER
|
||||
RET
|
||||
|
||||
// func decodeFixedO1AVX512(samples []int32, residual []int32)
|
||||
TEXT ·decodeFixedO1AVX512(SB), NOSPLIT, $0-48
|
||||
MOVQ samples_base+0(FP), SI
|
||||
VPBROADCASTD AX, Z15
|
||||
VMOVDQU32 (DI)(AX*1), Z0
|
||||
VALIGND $15, Z9, Z0, Z1
|
||||
VFMADD231PD Z14, Z12, Z10
|
||||
VPCMPEQD Z0, Z3, K1
|
||||
KTESTW K1, K1
|
||||
VPSRAQ X31, Z8, Z8
|
||||
VMOVDQU32 Z0, 4(SI)(AX*1)
|
||||
RET
|
||||
+108
@@ -0,0 +1,108 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package token defines the lexical tokens of Go's Plan 9 assembler (GAsm)
|
||||
// and the source positions attached to them. It has no dependencies and is
|
||||
// shared by the lexer, parser, formatter, linter and language server.
|
||||
package token
|
||||
|
||||
import "strconv"
|
||||
|
||||
// Kind classifies a lexical token.
|
||||
type Kind int
|
||||
|
||||
// The token kinds. The zero value is Illegal so that an uninitialised Token
|
||||
// is obviously invalid.
|
||||
const (
|
||||
Illegal Kind = iota
|
||||
EOF
|
||||
Newline
|
||||
Comment
|
||||
|
||||
// Literals and names.
|
||||
Ident // instruction mnemonic, label, register or symbol name
|
||||
Number // integer or floating-point literal (sign carried separately)
|
||||
String // "..."
|
||||
Rune // '.'
|
||||
|
||||
// Punctuation and operators.
|
||||
LParen // (
|
||||
RParen // )
|
||||
Comma // ,
|
||||
Plus // +
|
||||
Minus // -
|
||||
Star // *
|
||||
Slash // /
|
||||
Colon // :
|
||||
Dollar // $
|
||||
LAngle // <
|
||||
RAngle // >
|
||||
LShift // <<
|
||||
RShift // >>
|
||||
Arrow // ->
|
||||
At // @
|
||||
Hash // #
|
||||
)
|
||||
|
||||
var kindNames = map[Kind]string{
|
||||
Illegal: "ILLEGAL",
|
||||
EOF: "EOF",
|
||||
Newline: "NEWLINE",
|
||||
Comment: "COMMENT",
|
||||
Ident: "IDENT",
|
||||
Number: "NUMBER",
|
||||
String: "STRING",
|
||||
Rune: "RUNE",
|
||||
LParen: "(",
|
||||
RParen: ")",
|
||||
Comma: ",",
|
||||
Plus: "+",
|
||||
Minus: "-",
|
||||
Star: "*",
|
||||
Slash: "/",
|
||||
Colon: ":",
|
||||
Dollar: "$",
|
||||
LAngle: "<",
|
||||
RAngle: ">",
|
||||
LShift: "<<",
|
||||
RShift: ">>",
|
||||
Arrow: "->",
|
||||
At: "@",
|
||||
Hash: "#",
|
||||
}
|
||||
|
||||
// String returns a human-readable name for the kind.
|
||||
func (k Kind) String() string {
|
||||
if s, ok := kindNames[k]; ok {
|
||||
return s
|
||||
}
|
||||
return "Kind(" + strconv.Itoa(int(k)) + ")"
|
||||
}
|
||||
|
||||
// Position is a byte offset plus one-based line and column within a file.
|
||||
type Position struct {
|
||||
Offset int // byte offset, zero-based
|
||||
Line int // line number, one-based
|
||||
Column int // column number, one-based (in runes)
|
||||
}
|
||||
|
||||
// String renders the position as "line:column".
|
||||
func (p Position) String() string {
|
||||
return strconv.Itoa(p.Line) + ":" + strconv.Itoa(p.Column)
|
||||
}
|
||||
|
||||
// IsValid reports whether the position carries a real line number.
|
||||
func (p Position) IsValid() bool { return p.Line > 0 }
|
||||
|
||||
// Token is a single lexical token together with its literal text and span.
|
||||
type Token struct {
|
||||
Kind Kind
|
||||
Text string
|
||||
Pos Position // inclusive start
|
||||
End Position // exclusive end
|
||||
}
|
||||
|
||||
// String renders the token for diagnostics and debugging.
|
||||
func (t Token) String() string {
|
||||
return t.Pos.String() + " " + t.Kind.String() + " " + strconv.Quote(t.Text)
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package token
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestKindString(t *testing.T) {
|
||||
cases := map[Kind]string{
|
||||
EOF: "EOF",
|
||||
Ident: "IDENT",
|
||||
Number: "NUMBER",
|
||||
LParen: "(",
|
||||
LShift: "<<",
|
||||
Arrow: "->",
|
||||
Illegal: "ILLEGAL",
|
||||
}
|
||||
for k, want := range cases {
|
||||
if got := k.String(); got != want {
|
||||
t.Errorf("Kind(%d).String() = %q, want %q", int(k), got, want)
|
||||
}
|
||||
}
|
||||
// An out-of-range kind falls back to the numeric form.
|
||||
if got := Kind(9999).String(); got != "Kind(9999)" {
|
||||
t.Errorf("Kind(9999).String() = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosition(t *testing.T) {
|
||||
p := Position{Offset: 10, Line: 3, Column: 7}
|
||||
if got := p.String(); got != "3:7" {
|
||||
t.Errorf("Position.String() = %q, want 3:7", got)
|
||||
}
|
||||
if !p.IsValid() {
|
||||
t.Error("position with a line should be valid")
|
||||
}
|
||||
var zero Position
|
||||
if zero.IsValid() {
|
||||
t.Error("zero position should be invalid")
|
||||
}
|
||||
}
|
||||
|
||||
func TestTokenString(t *testing.T) {
|
||||
tok := Token{Kind: Ident, Text: "MOVQ", Pos: Position{Line: 1, Column: 2}}
|
||||
got := tok.String()
|
||||
if got == "" || got[0] == ' ' {
|
||||
t.Errorf("Token.String() = %q", got)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user