feat: gasm-devkit 0.1.0 — GAsm lexer, parser, linter, formatter, LSP and amd64 assembler

Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
2026-07-06 09:49:50 +02:00
commit d5a4a6de45
53 changed files with 13166 additions and 0 deletions
+175
View File
@@ -0,0 +1,175 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import (
"go/ast"
"go/parser"
"go/token"
"strconv"
"strings"
)
// abiExpectedArgSize computes the argument-area size (parameters plus results,
// laid out with Go's alignment rules on a 64-bit target) from the `// func …`
// signature in a TEXT function's doc comment. It returns ok=false when there
// is no parseable signature or it uses a type whose size cannot be determined
// (a named type), so the caller can skip the check rather than guess.
//
// The signature is parsed with the standard library's Go parser, so every
// legal signature form (shared names such as `left, right []int32`, nested
// pointers, arrays, structs) is handled correctly.
func abiExpectedArgSize(doc string) (int64, bool) {
sig := signatureLine(doc)
if sig == "" {
return 0, false
}
fset := token.NewFileSet()
f, err := parser.ParseFile(fset, "sig.go", "package p\n"+sig+" {}\n", 0)
if err != nil || len(f.Decls) == 0 {
return 0, false
}
fn, ok := f.Decls[0].(*ast.FuncDecl)
if !ok || fn.Type == nil {
return 0, false
}
return signatureSize(fn.Type.Params, fn.Type.Results)
}
// signatureLine returns the first `func …` line from a doc comment, trimmed.
func signatureLine(doc string) string {
for _, line := range strings.Split(doc, "\n") {
if t := strings.TrimSpace(line); strings.HasPrefix(t, "func ") {
return t
}
}
return ""
}
// signatureSize lays out the parameters and results and returns the total byte
// size of the argument area, matching Go's ABI0 stack layout: parameters are
// laid out first, then the result area begins on a word (8-byte) boundary.
func signatureSize(params, results *ast.FieldList) (int64, bool) {
paramsSize, _, ok := fieldsSizeAlign(params)
if !ok {
return 0, false
}
resultsSize, _, ok := fieldsSizeAlign(results)
if !ok {
return 0, false
}
// With no results the argument area is exactly the parameter size. When
// there are results, the result area begins on a word (8-byte) boundary
// after the parameters (Go's ABI0 stack layout).
if resultsSize == 0 {
return int64(paramsSize), true
}
return int64(alignUp(paramsSize, 8) + resultsSize), true
}
// fieldsSizeAlign lays out a field list sequentially (each field aligned to its
// own alignment) and returns the total size and the maximum field alignment.
func fieldsSizeAlign(list *ast.FieldList) (size, align int, ok bool) {
if list == nil {
return 0, 1, true
}
offset, maxAlign := 0, 1
for _, field := range list.List {
es, ea, fieldOK := typeSizeAlign(field.Type)
if !fieldOK {
return 0, 0, false
}
n := len(field.Names)
if n == 0 {
n = 1
}
for i := 0; i < n; i++ {
offset = alignUp(offset, ea)
offset += es
}
if ea > maxAlign {
maxAlign = ea
}
}
return offset, maxAlign, true
}
// basicSizes maps built-in type names to {size, align} on a 64-bit target.
var basicSizes = map[string][2]int{
"bool": {1, 1}, "byte": {1, 1}, "int8": {1, 1}, "uint8": {1, 1},
"int16": {2, 2}, "uint16": {2, 2},
"int32": {4, 4}, "uint32": {4, 4}, "float32": {4, 4},
"int": {8, 8}, "int64": {8, 8}, "uint": {8, 8}, "uint64": {8, 8},
"uintptr": {8, 8}, "float64": {8, 8},
"complex64": {8, 4}, "complex128": {16, 8},
"string": {16, 8}, "any": {16, 8}, "error": {16, 8},
}
// typeSizeAlign returns the size and alignment in bytes of a type expression,
// or ok=false when the size cannot be determined (an unknown named type).
func typeSizeAlign(e ast.Expr) (size, align int, ok bool) {
switch t := e.(type) {
case *ast.Ident:
if sa, found := basicSizes[t.Name]; found {
return sa[0], sa[1], true
}
return 0, 0, false // named type of unknown size
case *ast.SelectorExpr:
if pkg, isIdent := t.X.(*ast.Ident); isIdent && pkg.Name == "unsafe" && t.Sel.Name == "Pointer" {
return 8, 8, true
}
return 0, 0, false
case *ast.ParenExpr:
return typeSizeAlign(t.X)
case *ast.StarExpr:
return 8, 8, true // pointer
case *ast.MapType, *ast.ChanType, *ast.FuncType:
return 8, 8, true // map / chan / func are pointer-sized
case *ast.InterfaceType:
return 16, 8, true
case *ast.Ellipsis:
return 24, 8, true // variadic parameter is a slice
case *ast.ArrayType:
if t.Len == nil {
return 24, 8, true // slice header
}
n, lenOK := arrayLength(t.Len)
es, ea, elemOK := typeSizeAlign(t.Elt)
if !lenOK || !elemOK {
return 0, 0, false
}
return n * es, ea, true
case *ast.StructType:
return structSizeAlign(t.Fields)
}
return 0, 0, false
}
// structSizeAlign lays out a struct's fields and returns its size (rounded up
// to its alignment) and alignment.
func structSizeAlign(fields *ast.FieldList) (size, align int, ok bool) {
size, align, ok = fieldsSizeAlign(fields)
if !ok {
return 0, 0, false
}
return alignUp(size, align), align, true
}
// arrayLength evaluates a constant array-length expression (a literal, for the
// kernels this toolkit targets).
func arrayLength(e ast.Expr) (int, bool) {
if lit, ok := e.(*ast.BasicLit); ok && lit.Kind == token.INT {
if v, err := strconv.Atoi(lit.Value); err == nil {
return v, true
}
}
return 0, false
}
func alignUp(offset, align int) int {
if align <= 1 {
return offset
}
return (offset + align - 1) &^ (align - 1)
}
+129
View File
@@ -0,0 +1,129 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import "testing"
// TestABIExpectedArgSize checks the Go ABI0 argument-area size computation
// against hand-verified signatures (the same layouts the go-flac kernels use).
func TestABIExpectedArgSize(t *testing.T) {
cases := []struct {
sig string
want int64
}{
{"func f()", 0},
{"func f(a int, b int)", 16},
{"func f(a int32)", 4}, // no results: no word-boundary padding
{"func f(a int32) (r int32)", 12}, // results begin on a word boundary: 4 -> 8, +4
{"func f(a int) int", 16},
{"func f(s []int32, p *[32]uint16) (x uint64, ok bool)", 41},
{"func f(left, right []int32, sums *[4]uint64)", 56},
{"func f(src []byte, dst []int32)", 48},
{"func f(a bool, b int64)", 16}, // bool at 0, int64 aligned to 8
{"func f(x struct{ a int32; b int64 })", 16},
}
for _, c := range cases {
got, ok := abiExpectedArgSize(c.sig)
if !ok {
t.Errorf("%s: could not compute size", c.sig)
continue
}
if got != c.want {
t.Errorf("%s: size = %d, want %d", c.sig, got, c.want)
}
}
}
func TestABIUnknownTypeSkipped(t *testing.T) {
// A bare named type of unknown size must abort the check rather than guess.
// (A *pointer* to a named type is still 8 bytes and is fine.)
if _, ok := abiExpectedArgSize("func f(s Stream)"); ok {
t.Error("bare named type should make the size undecidable")
}
if _, ok := abiExpectedArgSize("func f(s *Stream)"); !ok {
t.Error("pointer to a named type is decidable (8 bytes)")
}
}
// TestABIArgSizeRule checks the lint rule end to end.
func TestABIArgSizeRule(t *testing.T) {
// Matching: the declared arg size agrees with the signature.
clean := lintSrc(t, "#include \"textflag.h\"\n"+
"// func f(a int, b int)\n"+
"TEXT ·f(SB), NOSPLIT, $0-16\n"+
"\tMOVQ a+0(FP), AX\n"+
"\tRET\n")
if codes(clean)[CodeABIArgSize] != 0 {
t.Fatalf("matching arg size should not warn: %+v", clean)
}
// Mismatching: declared 8, signature implies 16.
bad := lintSrc(t, "#include \"textflag.h\"\n"+
"// func f(a int, b int)\n"+
"TEXT ·f(SB), NOSPLIT, $0-8\n"+
"\tMOVQ a+0(FP), AX\n"+
"\tRET\n")
if codes(bad)[CodeABIArgSize] != 1 {
t.Fatalf("mismatching arg size should warn once: %+v", bad)
}
}
// TestABIArgSizeSkipsRegisterABI verifies the check does not fire for functions
// that declare a zero arg area (register ABI) or never touch FP.
func TestABIArgSizeSkipsRegisterABI(t *testing.T) {
diags := lintSrc(t, "#include \"textflag.h\"\n"+
"// func f(a int, b int)\n"+
"TEXT ·f(SB), NOSPLIT, $0-0\n"+
"\tMOVQ AX, BX\n"+
"\tRET\n")
if codes(diags)[CodeABIArgSize] != 0 {
t.Fatalf("register-ABI function must not be checked: %+v", diags)
}
}
// TestUnreachableCode exercises the dead-code detection and its guard rails.
func TestUnreachableCode(t *testing.T) {
// Code after a RET is unreachable.
dead := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tRET\n"+
"\tMOVQ AX, BX\n")
if codes(dead)[CodeUnreachable] != 1 {
t.Fatalf("code after RET should be unreachable: %+v", dead)
}
// A label after the RET makes the following code reachable again.
live := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tRET\n"+
"again:\n"+
"\tJMP again\n")
if codes(live)[CodeUnreachable] != 0 {
t.Fatalf("code after a label is reachable: %+v", live)
}
}
// TestUnreachableGuards verifies the analysis is suppressed where reachability
// cannot be determined statically.
func TestUnreachableGuards(t *testing.T) {
// A PC-relative jump defeats the analysis for the whole function.
pcrel := lintSrcArch(t, "f_amd64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tJCC 2(PC)\n"+
"\tRET\n"+
"\tMOVQ AX, BX\n")
if codes(pcrel)[CodeUnreachable] != 0 {
t.Fatalf("PC-relative functions must be skipped: %+v", pcrel)
}
// A register-indirect branch (riscv JALR) defeats the analysis too.
indirect := lintSrcArch(t, "f_riscv64.s", "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tJALR X1, X5\n"+
"\tRET\n"+
"\tMOV X1, X2\n")
if codes(indirect)[CodeUnreachable] != 0 {
t.Fatalf("indirect-branch functions must be skipped: %+v", indirect)
}
}
+96
View File
@@ -0,0 +1,96 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// checkFuncdata validates the structure of FUNCDATA and PCDATA directives,
// which carry the GC stack-map information. The checks are deliberately
// shallow — they confirm the operands are well formed and that a literal index
// is within the small range the runtime uses — and never try to interpret a
// named index constant such as $PCDATA_StackMapIndex.
func checkFuncdata(t *ast.Text, cfg Config) []Diagnostic {
var out []Diagnostic
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "FUNCDATA":
out = append(out, checkFunCDATA(in, cfg)...)
case "PCDATA":
out = append(out, checkPCDATA(in, cfg)...)
}
}
return out
}
// checkFunCDATA validates `FUNCDATA $index, symbol(SB)`.
func checkFunCDATA(in *ast.Instr, cfg Config) []Diagnostic {
if cfg.Disable[CodeFuncdata] {
return nil
}
var out []Diagnostic
if len(in.Operands) != 2 {
return []Diagnostic{{
Pos: in.Mnemonic.Pos, End: in.Mnemonic.End, Severity: Warning, Code: CodeFuncdata,
Message: fmt.Sprintf("FUNCDATA expects 2 operands (index, symbol), got %d", len(in.Operands)),
}}
}
out = append(out, checkIndex(in.Operands[0], "FUNCDATA")...)
if sym := in.Operands[1].Addr.Sym; in.Operands[1].Kind != ast.OpAddr || sym == nil {
out = append(out, Diagnostic{
Pos: in.Operands[1].Pos, Severity: Warning, Code: CodeFuncdata,
Message: "FUNCDATA second operand must be a symbol reference",
})
}
return out
}
// checkPCDATA validates `PCDATA $index, $value`.
func checkPCDATA(in *ast.Instr, cfg Config) []Diagnostic {
if cfg.Disable[CodeFuncdata] {
return nil
}
if len(in.Operands) != 2 {
return []Diagnostic{{
Pos: in.Mnemonic.Pos, End: in.Mnemonic.End, Severity: Warning, Code: CodeFuncdata,
Message: fmt.Sprintf("PCDATA expects 2 operands (index, value), got %d", len(in.Operands)),
}}
}
var out []Diagnostic
out = append(out, checkIndex(in.Operands[0], "PCDATA")...)
if in.Operands[1].Kind != ast.OpImmediate {
out = append(out, Diagnostic{
Pos: in.Operands[1].Pos, Severity: Warning, Code: CodeFuncdata,
Message: "PCDATA value must be an immediate",
})
}
return out
}
// checkIndex validates an immediate index operand. A literal index must lie in
// the small range the runtime uses; a named constant (e.g. $PCDATA_StackMapIndex)
// cannot be evaluated and is accepted without a range check.
func checkIndex(op *ast.Operand, directive string) []Diagnostic {
if op.Kind != ast.OpImmediate {
return []Diagnostic{{
Pos: op.Pos, Severity: Warning, Code: CodeFuncdata,
Message: directive + " index must be an immediate",
}}
}
if op.Imm.HasVal && (op.Imm.Val < 0 || op.Imm.Val > 10) {
return []Diagnostic{{
Pos: op.Pos, Severity: Warning, Code: CodeFuncdata,
Message: fmt.Sprintf("%s index %d is outside the valid range 0–10", directive, op.Imm.Val),
}}
}
return nil
}
+60
View File
@@ -0,0 +1,60 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import (
"os"
"path/filepath"
"runtime"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGoRuntimeCorpus parses and lints every runtime .s file the local Go
// toolchain ships for all four supported architectures. This is the
// real-world regression net: it exercises the full breadth of each
// architecture's syntax (macros, addressing modes, branch aliases) against
// production assembly. It is skipped when the toolchain source is absent.
//
// The bar is zero parse errors and zero error-severity diagnostics — i.e. no
// false "unknown instruction" / "undefined label" findings on code the real
// assembler accepts. Advisory warnings are reported but not fatal, since they
// are heuristics that may legitimately differ across Go versions.
func TestGoRuntimeCorpus(t *testing.T) {
dir := filepath.Join(runtime.GOROOT(), "src", "runtime")
var files []string
for _, suffix := range []string{"_amd64.s", "_arm64.s", "_riscv64.s", "_loong64.s"} {
matches, _ := filepath.Glob(filepath.Join(dir, "*"+suffix))
files = append(files, matches...)
}
if len(files) == 0 {
t.Skip("Go toolchain source (src/runtime/*.s) not present")
}
warnings := 0
for _, path := range files {
src, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read %s: %v", path, err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Errorf("parse %s: %v", filepath.Base(path), errs)
continue
}
diags := File(f, Config{Arch: arch.FromFilename(path)})
for _, d := range diags {
if d.Severity == Error {
t.Errorf("%s:%d: error %s: %s", filepath.Base(path), d.Pos.Line, d.Code, d.Message)
} else {
warnings++
}
}
}
if warnings > 0 {
t.Logf("%d advisory warnings across %d files (non-fatal)", warnings, len(files))
}
}
+510
View File
@@ -0,0 +1,510 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package lint runs static checks over a parsed GAsm file. The rules are
// deliberately conservative: where a check cannot be certain (for example an
// instruction whose operand count varies), it stays silent rather than emit a
// false positive. Every diagnostic carries a stable rule code so callers can
// disable individual rules.
package lint
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Severity ranks a diagnostic.
type Severity int
// Diagnostic severities, mirroring the language-server protocol ordering.
const (
Error Severity = iota
Warning
Information
Hint
)
// String returns a lower-case label for the severity.
func (s Severity) String() string {
switch s {
case Error:
return "error"
case Warning:
return "warning"
case Information:
return "information"
default:
return "hint"
}
}
// Diagnostic is one lint finding.
type Diagnostic struct {
Pos token.Position
End token.Position
Severity Severity
Code string
Message string
}
// Config controls a lint run.
type Config struct {
// Arch is the target architecture. When it is arch.Unknown the
// architecture-specific rules (unknown instruction, operand count) are
// skipped because no instruction table can be selected.
Arch arch.Arch
// Disable lists rule codes to suppress.
Disable map[string]bool
}
// Rule codes.
const (
CodeUnknownInstr = "unknown-instruction"
CodeOperandCount = "operand-count"
CodeUndefinedLabel = "undefined-label"
CodeDuplicateLabel = "duplicate-label"
CodeMissingRet = "missing-ret"
CodeMissingTextflag = "missing-textflag-include"
CodeUnreachable = "unreachable-code"
CodeABIArgSize = "abi-argsize"
CodeNosplitFrame = "nosplit-frame"
CodeRegisterClobber = "register-clobber"
CodeFuncdata = "funcdata-pcdata"
)
// pseudoOps are assembler pseudo-operations that are valid instruction-position
// tokens but are not machine instructions and so absent from the arch tables.
var pseudoOps = map[string]bool{
"BYTE": true, "WORD": true, "LONG": true, "QUAD": true, "FLOAT": true,
"PCALIGN": true, "FUNCDATA": true, "PCDATA": true, "GO_ARGS": true,
}
// File lints a parsed file and returns the diagnostics in source order.
func File(f *ast.File, cfg Config) []Diagnostic {
var out []Diagnostic
tab := arch.ForArch(cfg.Arch)
archKnown := cfg.Arch != arch.Unknown
hasTextflag := false
usesFlags := false
var firstFlagPos token.Position
// Macros (in-file #define, or any #include other than textflag.h, which
// only defines flag constants) make label resolution unreliable.
macrosInPlay := len(f.Macros) > 0
// Preprocessor conditionals (#ifdef …) make control-flow analysis
// unreliable, since mutually exclusive branches look sequential.
hasConditionals := false
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Include:
if !strings.Contains(dd.Header.Text, "textflag.h") {
macrosInPlay = true
}
case *ast.Preproc:
if isConditionalDirective(dd.Raw) {
hasConditionals = true
}
}
}
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Include:
if strings.Contains(dd.Header.Text, "textflag.h") {
hasTextflag = true
}
case *ast.Text:
out = append(out, lintText(dd, tab, archKnown, cfg, f.Macros, !macrosInPlay, !hasConditionals)...)
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
usesFlags = true
firstFlagPos = dd.Pos()
}
case *ast.Globl:
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
usesFlags = true
firstFlagPos = dd.Pos()
}
}
}
if !cfg.Disable[CodeMissingTextflag] && usesFlags && !hasTextflag {
out = append(out, Diagnostic{
Pos: firstFlagPos,
Severity: Warning,
Code: CodeMissingTextflag,
Message: "TEXT/GLOBL flags are used but textflag.h is not #included",
})
}
sortDiagnostics(out)
return out
}
// lintText lints one TEXT function body. doLabelChecks is false for files
// that use macros (an in-file #define or a non-textflag #include): without a
// preprocessor we cannot resolve labels that macros define or reference, so the
// label and RET heuristics are suppressed there to avoid false positives.
func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros map[string]bool, doLabelChecks bool, doUnreachable bool) []Diagnostic {
var out []Diagnostic
defined := map[string]token.Position{}
referenced := map[string]token.Position{}
hasRet := false
lastTerminal := false
hasMacro := false
instrCount := 0
dead := false // inside a region unreachable from above
reportedDead := false // the current dead region has already been reported
hasPCRel := referencesPC(t) // PC-relative jumps defeat reachability analysis
hasIndirect := hasIndirectBranch(t) // register-indirect branches do too
// Unreachable-code analysis is only sound in functions whose control flow is
// fully label-resolvable: no PC-relative jumps, no register-indirect
// branches, and (file-level) no preprocessor conditionals.
analyzable := doUnreachable && !hasPCRel && !hasIndirect
for _, s := range t.Body {
switch st := s.(type) {
case *ast.Label:
name := st.Name.Text
if prev, dup := defined[name]; dup {
if !cfg.Disable[CodeDuplicateLabel] {
out = append(out, Diagnostic{
Pos: st.Name.Pos,
End: st.Name.End,
Severity: Error,
Code: CodeDuplicateLabel,
Message: fmt.Sprintf("label %q already defined at %s", name, prev),
})
}
} else {
defined[name] = st.Name.Pos
}
// A label is a jump target: code after it is reachable again.
dead = false
reportedDead = false
case *ast.Instr:
instrCount++
mnem := st.Mnemonic.Text
upper := strings.ToUpper(mnem)
// Unreachable code: a real instruction following a RET/UNDEF and
// before any label, in a function whose control flow is fully
// resolvable. Only RET/UNDEF are treated as terminators here — an
// unconditional jump may be one entry of a hand-arranged branch
// table (e.g. the generated callback tables), so it is not assumed
// to make the following code dead. Pseudo-ops and macro invocations
// are never flagged.
if analyzable && dead && !reportedDead && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) &&
!cfg.Disable[CodeUnreachable] {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Warning,
Code: CodeUnreachable,
Message: "unreachable code after terminating instruction",
})
reportedDead = true
}
// RET never falls through. (UNDEF is a trap/marker rather than a
// control-flow terminator: code placed after it is occasionally
// deliberate metadata, so it is not treated as making the following
// code dead.)
if upper == "RET" {
dead = true
}
// A function need not RET if it ends in an unconditional jump (tail
// call / loop) or in UNDEF (a deliberate trap that never returns).
lastTerminal = isUnconditionalJump(cfg.Arch, upper) || upper == "UNDEF"
if isMacroInvocation(mnem, macros) {
hasMacro = true
}
if upper == "RET" {
hasRet = true
}
if archKnown && !cfg.Disable[CodeUnknownInstr] && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) {
if _, ok := tab.Lookup(mnem); !ok {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Error,
Code: CodeUnknownInstr,
Message: fmt.Sprintf("unknown %s instruction %q", cfg.Arch, mnem),
})
}
}
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) {
if in, ok := tab.Lookup(mnem); ok && in.MinOps >= 0 {
n := len(st.Operands)
if n < in.MinOps || n > in.MaxOps {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Warning,
Code: CodeOperandCount,
Message: fmt.Sprintf("%s expects %s, got %d operand(s)",
mnem, countRange(in.MinOps, in.MaxOps), n),
})
}
}
}
if isJump(cfg.Arch, upper) {
for _, op := range st.Operands {
if name, pos, ok := localLabelRef(op); ok && !tab.IsRegister(name) && !arch.IsPseudoReg(name) {
referenced[name] = pos
}
}
}
}
}
// Undefined labels.
if doLabelChecks && !cfg.Disable[CodeUndefinedLabel] {
for name, pos := range referenced {
if _, ok := defined[name]; !ok {
out = append(out, Diagnostic{
Pos: pos,
Severity: Error,
Code: CodeUndefinedLabel,
Message: fmt.Sprintf("jump to undefined label %q", name),
})
}
}
}
// Missing RET heuristic. Functions that invoke a macro are skipped: the
// macro body (opaque to us) may supply the RET.
if doLabelChecks && !cfg.Disable[CodeMissingRet] && instrCount > 0 && !hasRet && !lastTerminal && !hasMacro {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeMissingRet,
Message: fmt.Sprintf("function %q has no RET", t.Name.Name),
})
}
// ABI conformance: the argument area declared in the TEXT directive should
// match the size computed from the // func signature in the doc comment.
// Only applies to stack-argument (ABI0) functions, which reference their
// arguments through FP; register-ABI functions declare a zero arg area. Also
// skipped when there is no parseable signature or it uses an unknown type.
if !cfg.Disable[CodeABIArgSize] {
got := int64(0)
if t.Args != nil && t.Args.Imm.HasVal {
got = t.Args.Imm.Val
}
// Only meaningful for stack-argument (ABI0) functions: a non-zero
// declared arg area that is actually addressed through FP.
if got > 0 && usesFPArgs(t) {
if want, ok := abiExpectedArgSize(t.Doc); ok {
if want != got {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeABIArgSize,
Message: fmt.Sprintf("TEXT declares arg size %d but the // func signature implies %d", got, want),
})
}
}
}
}
// Register liveness: a callee-saved register that is written but never
// saved and restored is clobbered across the call. The check runs over the
// control-flow graph and is skipped for macro-using files, where an opaque
// macro may perform the save/restore.
if doLabelChecks && archKnown && !cfg.Disable[CodeRegisterClobber] {
live := analyzeLiveness(t, cfg.Arch)
if clobbered := clobberedCalleeSaved(live, cfg.Arch); len(clobbered) > 0 {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeRegisterClobber,
Message: fmt.Sprintf("callee-saved register(s) %s written but never saved/restored", strings.Join(clobbered, ", ")),
})
}
}
// FUNCDATA / PCDATA structural validation.
out = append(out, checkFuncdata(t, cfg)...)
return out
}
// usesFPArgs reports whether a function references its arguments through the FP
// pseudo-register — i.e. it uses the stack-based ABI0 layout, where the
// declared argument size must match the signature.
func usesFPArgs(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
for _, op := range in.Operands {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
return true
}
}
}
return false
}
// referencesPC reports whether a function uses a PC-relative operand (e.g.
// `JMP 2(PC)`). Such jumps target a computed offset rather than a label, so
// reachability cannot be determined statically and the unreachable-code check
// is suppressed for the whole function.
func referencesPC(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
for _, op := range in.Operands {
if strings.Contains(strings.ReplaceAll(op.Raw, " ", ""), "(PC)") {
return true
}
}
}
return false
}
// hasIndirectBranch reports whether a function transfers control through a
// register (JALR/JR/JIRL/BR/BLR). Such targets are computed at runtime, so
// reachability cannot be determined statically and the unreachable-code check is
// suppressed for the whole function.
func hasIndirectBranch(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "JALR", "JR", "JIRL", "BR", "BLR":
return true
}
}
return false
}
// isMacroInvocation reports whether a mnemonic is a macro invocation rather
// than a machine instruction. No Plan 9 mnemonic contains an underscore, so an
// underscore is a reliable macro marker (the runtime headers define macros such
// as get_tls and NO_LOCAL_POINTERS). Names introduced by an in-file #define
// are recognised too (CALLFN, DISPATCH, …). Full macro expansion is out of
// scope; this only keeps the linter quiet on invocations it cannot expand.
func isMacroInvocation(mnem string, macros map[string]bool) bool {
return strings.Contains(mnem, "_") || macros[mnem]
}
// isConditionalDirective reports whether a preprocessor directive (the text
// after '#') is a conditional-compilation directive whose branches the parser
// cannot resolve.
func isConditionalDirective(raw string) bool {
fields := strings.Fields(raw)
if len(fields) == 0 {
return false
}
switch fields[0] {
case "if", "ifdef", "ifndef", "else", "elif", "endif":
return true
}
return false
}
// localLabelRef returns the name and position of a bare local-label reference
// operand (no pseudo-register, no memory base), if op is one.
func localLabelRef(op *ast.Operand) (string, token.Position, bool) {
if op == nil || op.Kind != ast.OpAddr || op.Addr.Sym == nil {
return "", token.Position{}, false
}
sym := op.Addr.Sym
if sym.Pseudo != "" || op.Addr.Base != "" || sym.Name == "" {
return "", token.Position{}, false
}
return sym.Name, op.Pos, true
}
// riscvBranches and loong64Branches are the conditional-branch mnemonics; they
// are listed explicitly rather than matched by a "B" prefix so that bit-manip
// instructions (BCLR, BSET, …) are never mistaken for branches.
var riscvBranches = map[string]bool{
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
"BEQZ": true, "BNEZ": true, "BLEZ": true, "BGEZ": true, "BLTZ": true, "BGTZ": true,
}
var loong64Branches = map[string]bool{
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
"BLEZ": true, "BLTZ": true, "BGEZ": true, "BGTZ": true,
}
// isJump reports whether the mnemonic is any branch.
func isJump(a arch.Arch, upper string) bool {
switch a {
case arch.ARM64:
return upper == "CALL" || upper == "BR" || upper == "BLR" || upper == "JMP" ||
strings.HasPrefix(upper, "B") ||
strings.HasPrefix(upper, "CBZ") || strings.HasPrefix(upper, "CBNZ") ||
strings.HasPrefix(upper, "TBZ") || strings.HasPrefix(upper, "TBNZ")
case arch.RISCV:
return upper == "CALL" || riscvBranches[upper] ||
upper == "JMP" || upper == "J" || upper == "JAL" || upper == "JALR" ||
upper == "JR" || upper == "BR"
case arch.LOONG64:
return upper == "CALL" || loong64Branches[upper] ||
upper == "JIRL" || upper == "JMP" || upper == "BR"
default: // amd64
return upper == "CALL" || strings.HasPrefix(upper, "J")
}
}
// isUnconditionalJump reports whether the mnemonic is an unconditional branch
// (used to suppress the missing-RET heuristic for tail calls and loops).
func isUnconditionalJump(a arch.Arch, upper string) bool {
switch a {
case arch.ARM64:
return upper == "B" || upper == "BR" || upper == "JMP"
case arch.RISCV:
return upper == "JMP" || upper == "J" || upper == "JAL" ||
upper == "JALR" || upper == "JR" || upper == "BR"
case arch.LOONG64:
return upper == "JMP" || upper == "JIRL" || upper == "BR"
default:
return upper == "JMP"
}
}
func countRange(min, max int) string {
if min == max {
return fmt.Sprintf("%d operand(s)", min)
}
return fmt.Sprintf("%d–%d operands", min, max)
}
// sortDiagnostics orders diagnostics by line, then column, then code.
func sortDiagnostics(d []Diagnostic) {
for i := 1; i < len(d); i++ {
for j := i; j > 0 && lessDiag(d[j], d[j-1]); j-- {
d[j], d[j-1] = d[j-1], d[j]
}
}
}
func lessDiag(a, b Diagnostic) bool {
if a.Pos.Line != b.Pos.Line {
return a.Pos.Line < b.Pos.Line
}
if a.Pos.Column != b.Pos.Column {
return a.Pos.Column < b.Pos.Column
}
return a.Code < b.Code
}
+242
View File
@@ -0,0 +1,242 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import (
"os"
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
func lintSrc(t *testing.T, src string) []Diagnostic {
t.Helper()
f, errs := parser.Parse("test_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
return File(f, Config{Arch: arch.AMD64})
}
// lintSrcArch lints src under the architecture inferred from filename.
func lintSrcArch(t *testing.T, filename, src string) []Diagnostic {
t.Helper()
f, errs := parser.Parse(filename, src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
return File(f, Config{Arch: arch.FromFilename(filename)})
}
func codes(diags []Diagnostic) map[string]int {
m := map[string]int{}
for _, d := range diags {
m[d.Code]++
}
return m
}
func TestFixtureIsClean(t *testing.T) {
src, err := os.ReadFile("../testdata/sample_amd64.s")
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse("sample_amd64.s", string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
// The fixture mirrors the go-flac kernels, which use callee-saved registers
// (BX, R13) without saving them; the register-clobber audit flags that by
// design. This test targets the other rules, so the audit is disabled here
// (it is covered by TestRegisterClobber).
diags := File(f, Config{Arch: arch.AMD64, Disable: map[string]bool{CodeRegisterClobber: true}})
if len(diags) != 0 {
t.Fatalf("expected no diagnostics on the fixture, got %+v", diags)
}
}
func TestUnknownInstruction(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
FOOBAR AX, BX
RET
`)
if codes(diags)[CodeUnknownInstr] != 1 {
t.Fatalf("want one unknown-instruction, got %+v", diags)
}
}
func TestUndefinedLabel(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
JMP nowhere
RET
`)
if codes(diags)[CodeUndefinedLabel] != 1 {
t.Fatalf("want one undefined-label, got %+v", diags)
}
}
func TestDuplicateLabel(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
loop:
ADDQ $1, AX
loop:
SUBQ $1, AX
JMP loop
RET
`)
if codes(diags)[CodeDuplicateLabel] != 1 {
t.Fatalf("want one duplicate-label, got %+v", diags)
}
}
func TestMissingRet(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
ADDQ $1, AX
`)
if codes(diags)[CodeMissingRet] != 1 {
t.Fatalf("want one missing-ret, got %+v", diags)
}
}
func TestOperandCount(t *testing.T) {
// RET takes zero operands; JMP takes exactly one.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
RET AX
JMP
RET
`)
c := codes(diags)
if c[CodeOperandCount] != 2 {
t.Fatalf("want two operand-count findings, got %+v", diags)
}
}
func TestMissingTextflag(t *testing.T) {
diags := lintSrc(t, `
TEXT ·f(SB), NOSPLIT, $0
RET
`)
if codes(diags)[CodeMissingTextflag] != 1 {
t.Fatalf("want one missing-textflag-include, got %+v", diags)
}
}
func TestDisableRule(t *testing.T) {
f, _ := parser.Parse("t_amd64.s", `
TEXT ·f(SB), NOSPLIT, $0
RET
`)
diags := File(f, Config{Arch: arch.AMD64, Disable: map[string]bool{CodeMissingTextflag: true}})
if len(diags) != 0 {
t.Fatalf("disabling the rule should silence it, got %+v", diags)
}
}
func TestMacroInvocationSkipped(t *testing.T) {
// DISPATCH is defined in-file; get_tls carries an underscore. Neither is a
// machine instruction, so both must be ignored by the unknown-instruction
// rule rather than flagged.
diags := lintSrc(t, `
#include "textflag.h"
#define DISPATCH CALL ·x(SB)
TEXT ·f(SB), NOSPLIT, $0
DISPATCH
get_tls CX
RET
`)
if codes(diags)[CodeUnknownInstr] != 0 {
t.Fatalf("macro invocations must not be flagged: %+v", diags)
}
}
func TestUndefIsTerminal(t *testing.T) {
// A function whose body is UNDEF traps and never returns; it needs no RET.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
UNDEF
`)
if codes(diags)[CodeMissingRet] != 0 {
t.Fatalf("UNDEF should count as terminal: %+v", diags)
}
}
func TestArm64BranchAlias(t *testing.T) {
diags := lintSrcArch(t, "f_arm64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
B done
done:
RET
`)
if len(diags) != 0 {
t.Fatalf("arm64 B to a defined label should be clean: %+v", diags)
}
}
func TestArm64AddressingSuffix(t *testing.T) {
// .W (pre-index) and .P (post-index) suffixes must resolve to the base
// instruction.
diags := lintSrcArch(t, "f_arm64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
LDP.W (R0), (R1, R2)
VST1.P (R3), (R4)
RET
`)
if codes(diags)[CodeUnknownInstr] != 0 {
t.Fatalf("suffixed load/store should be recognised: %+v", diags)
}
}
func TestMacrosInPlaySuppressesLabelRules(t *testing.T) {
// Including a non-textflag header means macros may define labels and supply
// the RET, so undefined-label and missing-ret are suppressed.
diags := lintSrcArch(t, "f_arm64.s", `
#include "go_asm.h"
TEXT ·f(SB), NOSPLIT, $0
JMP RARG0
`)
if codes(diags)[CodeUndefinedLabel] != 0 || codes(diags)[CodeMissingRet] != 0 {
t.Fatalf("label rules should be suppressed in macro files: %+v", diags)
}
}
// TestRealGoLibrariesHasNoErrors asserts that the production go-flac kernels
// lint free of errors. Skipped when the sibling repository is absent.
func TestRealGoLibrariesHasNoErrors(t *testing.T) {
matches, _ := filepath.Glob("../../go-libraries/go-*/*.s")
if len(matches) == 0 {
t.Skip("go-libraries repository not present")
}
for _, path := range matches {
src, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse %s: %v", path, errs)
}
a := arch.FromFilename(path)
diags := File(f, Config{Arch: a})
for _, d := range diags {
if d.Severity == Error {
t.Errorf("%s: %s %s: %s", filepath.Base(path), d.Pos, d.Code, d.Message)
}
}
}
}
+455
View File
@@ -0,0 +1,455 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import (
"fmt"
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// This file implements register liveness by dataflow over a function's
// control-flow graph, and the checks built on it. The def/use model is
// deliberately conservative: where an instruction's effect is uncertain it is
// treated as both a use and a def of its register operands, which can only
// suppress a finding, never invent one.
// regEffect is the register-level effect of one instruction.
type regEffect struct {
def []string // registers written (killed)
use []string // registers read
saveGPR []string // callee-saved-style: register written to the stack
restGPR []string // register restored from the stack
}
// analyzeLiveness builds the control-flow graph of a function and computes
// live-in/live-out register sets by iterative backward dataflow.
type liveness struct {
blocks []*block
liveIn []map[string]bool
}
type block struct {
label string // label that begins this block, if any
instrs []*ast.Instr
succ []int // successor block indices
}
func analyzeLiveness(t *ast.Text, a arch.Arch) *liveness {
l := &liveness{}
l.buildCFG(t)
l.dataflow(a)
return l
}
// buildCFG splits the function body into basic blocks and wires up successors.
func (l *liveness) buildCFG(t *ast.Text) {
labelToBlock := map[string]int{}
var cur *block
flush := func() {
if cur != nil && len(cur.instrs) > 0 {
l.blocks = append(l.blocks, cur)
}
cur = nil
}
startBlock := func(lbl string) {
flush()
cur = &block{label: lbl}
}
startBlock("")
for _, s := range t.Body {
switch st := s.(type) {
case *ast.Label:
// A label begins a new block and is a jump target.
startBlock(st.Name.Text)
labelToBlock[st.Name.Text] = len(l.blocks) // index once flushed
case *ast.Instr:
if cur == nil {
startBlock("")
}
cur.instrs = append(cur.instrs, st)
if terminates(st) || isConditionalBranch(st) {
startBlock("")
}
}
}
flush()
// Fix up label->block indices (labels were recorded before the following
// block was appended) and build successor edges.
for i, b := range l.blocks {
if b.label != "" {
labelToBlock[b.label] = i
}
}
for i, b := range l.blocks {
if len(b.instrs) == 0 {
if i+1 < len(l.blocks) {
b.succ = append(b.succ, i+1)
}
continue
}
last := b.instrs[len(b.instrs)-1]
mnem := strings.ToUpper(last.Mnemonic.Text)
switch {
case mnem == "RET" || mnem == "UNDEF":
// No successors.
case isConditionalBranch(last):
if tgt, ok := branchTarget(last); ok {
if j, found := labelToBlock[tgt]; found {
b.succ = append(b.succ, j)
}
}
if i+1 < len(l.blocks) {
b.succ = append(b.succ, i+1) // fall-through
}
case isUnconditionalBranchAny(mnem):
if tgt, ok := branchTarget(last); ok {
if j, found := labelToBlock[tgt]; found {
b.succ = append(b.succ, j)
}
}
default:
if i+1 < len(l.blocks) {
b.succ = append(b.succ, i+1)
}
}
}
}
// dataflow runs the standard backward liveness iteration to a fixed point.
func (l *liveness) dataflow(a arch.Arch) {
n := len(l.blocks)
l.liveIn = make([]map[string]bool, n)
liveOut := make([]map[string]bool, n)
use := make([]map[string]bool, n)
def := make([]map[string]bool, n)
for i, b := range l.blocks {
use[i], def[i] = blockUseDef(b, a)
l.liveIn[i] = map[string]bool{}
liveOut[i] = map[string]bool{}
}
for changed := true; changed; {
changed = false
for i := n - 1; i >= 0; i-- {
out := map[string]bool{}
for _, s := range l.blocks[i].succ {
for r := range l.liveIn[s] {
out[r] = true
}
}
if !sameSet(out, liveOut[i]) {
liveOut[i] = out
changed = true
}
// in = use ∪ (out − def)
in := map[string]bool{}
for r := range use[i] {
in[r] = true
}
for r := range out {
if !def[i][r] {
in[r] = true
}
}
if !sameSet(in, l.liveIn[i]) {
l.liveIn[i] = in
changed = true
}
}
}
}
// blockUseDef computes the registers used before definition (use) and the
// registers defined (def) within a basic block.
func blockUseDef(b *block, a arch.Arch) (use, def map[string]bool) {
use = map[string]bool{}
def = map[string]bool{}
for _, in := range b.instrs {
eff := instrEffect(in, a)
for _, r := range eff.use {
if !def[r] {
use[r] = true
}
}
for _, r := range eff.def {
def[r] = true
}
}
return use, def
}
// terminates reports whether an instruction ends basic-block flow unconditionally.
func terminates(in *ast.Instr) bool {
m := strings.ToUpper(in.Mnemonic.Text)
return m == "RET" || m == "UNDEF" || isUnconditionalBranchAny(m)
}
func isConditionalBranch(in *ast.Instr) bool {
m := strings.ToUpper(in.Mnemonic.Text)
// Conditional jumps/branches, but not the unconditional ones.
if isUnconditionalBranchAny(m) || m == "RET" || m == "UNDEF" || m == "CALL" {
return false
}
return strings.HasPrefix(m, "J") || strings.HasPrefix(m, "B") ||
strings.HasPrefix(m, "CBZ") || strings.HasPrefix(m, "CBNZ") ||
strings.HasPrefix(m, "TBZ") || strings.HasPrefix(m, "TBNZ") ||
strings.HasPrefix(m, "BEQ") || strings.HasPrefix(m, "BNE")
}
// isUnconditionalBranchAny is an arch-agnostic unconditional-branch test.
func isUnconditionalBranchAny(m string) bool {
switch m {
case "JMP", "J", "JR", "B", "BR", "JIRL":
return true
}
return false
}
// branchTarget returns the local-label target of a branch, if it is one.
func branchTarget(in *ast.Instr) (string, bool) {
for _, op := range in.Operands {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
}
return "", false
}
// instrEffect returns the register-level effect of one instruction.
func instrEffect(in *ast.Instr, a arch.Arch) regEffect {
var eff regEffect
mnem := strings.ToUpper(in.Mnemonic.Text)
// PUSH/POP move a register to/from the stack.
if strings.HasPrefix(mnem, "PUSH") {
for _, op := range in.Operands {
if r := gprName(op, a); r != "" {
eff.use = append(eff.use, r)
eff.saveGPR = append(eff.saveGPR, r)
}
}
return eff
}
if strings.HasPrefix(mnem, "POP") {
for _, op := range in.Operands {
if r := gprName(op, a); r != "" {
eff.def = append(eff.def, r)
eff.restGPR = append(eff.restGPR, r)
}
}
return eff
}
compare := isCompare(mnem)
dstIdx := dstIndex(in, a)
for i, op := range in.Operands {
r := gprName(op, a)
if r != "" {
if i == dstIdx && !compare {
eff.def = append(eff.def, r)
// Arithmetic also reads its destination.
eff.use = append(eff.use, r)
} else {
eff.use = append(eff.use, r)
}
}
// Detect saves/restores through the stack frame.
if isStackAddr(op) {
// The other operand (the register) is being saved or restored.
for j, other := range in.Operands {
if j == i {
continue
}
if rr := gprName(other, a); rr != "" {
if j == dstIdx && !compare {
eff.restGPR = append(eff.restGPR, rr) // reg loaded from stack
} else {
eff.saveGPR = append(eff.saveGPR, rr) // reg stored to stack
}
}
}
}
}
return eff
}
// dstIndex returns the operand index of the destination register: last for the
// Plan 9 (amd64) spelling, first for arm64/riscv64/loong64.
func dstIndex(in *ast.Instr, a arch.Arch) int {
if a == arch.AMD64 {
return len(in.Operands) - 1
}
return 0
}
// isCompare reports whether the mnemonic only reads its operands (setting flags).
func isCompare(m string) bool {
return strings.HasPrefix(m, "CMP") || strings.HasPrefix(m, "TEST") ||
strings.HasPrefix(m, "CMN") || strings.HasPrefix(m, "TST") ||
m == "FCMP" || m == "FCMPE"
}
// gprName returns the canonical general-purpose register name of an operand, or
// "" if the operand is not a bare GPR reference.
func gprName(op *ast.Operand, a arch.Arch) string {
if op == nil || op.Kind != ast.OpAddr || op.Addr.Sym == nil {
return ""
}
if op.Addr.Base != "" || op.Addr.Sym.Pseudo != "" || op.Addr.Sym.Name == "" {
return ""
}
name := op.Addr.Sym.Name
if r, ok := arch.ForArch(a).Register(name); ok && (r.Class == arch.GPR || r.Class == arch.GPRSub) {
return canonicalGPR(name)
}
return ""
}
// canonicalGPR maps a sized sub-register to its base GPR (amd64 only).
func canonicalGPR(name string) string {
upper := strings.ToUpper(name)
// Named 8/16/32-bit forms of the classic registers.
switch upper {
case "AL", "AH", "AX":
return "AX"
case "BL", "BH", "BX":
return "BX"
case "CL", "CH", "CX":
return "CX"
case "DL", "DH", "DX":
return "DX"
case "SIL":
return "SI"
case "DIL":
return "DI"
case "BPL":
return "BP"
case "SPL":
return "SP"
}
// Numbered sub-registers R8B/R8W/R8D → R8.
if len(upper) >= 3 && upper[0] == 'R' {
switch upper[len(upper)-1] {
case 'B', 'W', 'D':
return upper[:len(upper)-1]
}
}
return upper
}
// isStackAddr reports whether an operand addresses the stack frame
// (base SP, or an FP/SP-relative symbol).
func isStackAddr(op *ast.Operand) bool {
if op == nil || op.Kind != ast.OpAddr {
return false
}
if op.Addr.Base == "SP" {
return true
}
if op.Addr.Sym != nil && (op.Addr.Sym.Pseudo == "SP" || op.Addr.Sym.Pseudo == "FP") {
return true
}
return false
}
func sameSet(a, b map[string]bool) bool {
if len(a) != len(b) {
return false
}
for k := range a {
if !b[k] {
return false
}
}
return true
}
// calleeSavedGPRs returns the general-purpose registers an assembly function
// must preserve for its caller, using the register names the assembler accepts
// for each architecture.
func calleeSavedGPRs(a arch.Arch) map[string]bool {
switch a {
case arch.AMD64:
return gprSet("BX", "BP", "R12", "R13", "R14", "R15")
case arch.ARM64:
names := []string{"R29", "R30"} // FP, LR
for i := 19; i <= 28; i++ {
names = append(names, fmt.Sprintf("R%d", i))
}
return gprSet(names...)
case arch.RISCV:
// RA (X1) and the S registers (X8, X9, X18–X27) are callee-saved.
names := []string{"X1", "RA", "X8", "X9", "S0", "S1", "FP"}
for i := 18; i <= 27; i++ {
names = append(names, fmt.Sprintf("X%d", i))
}
for i := 2; i <= 11; i++ {
names = append(names, fmt.Sprintf("S%d", i))
}
return gprSet(names...)
case arch.LOONG64:
// RA (R1), FP (R22) and S0–S8 (R23–R31) are callee-saved.
names := []string{"R1", "RA", "R22", "FP"}
for i := 23; i <= 31; i++ {
names = append(names, fmt.Sprintf("R%d", i))
}
for i := 0; i <= 8; i++ {
names = append(names, fmt.Sprintf("S%d", i))
}
return gprSet(names...)
}
return nil
}
func gprSet(names ...string) map[string]bool {
m := make(map[string]bool, len(names))
for _, n := range names {
m[n] = true
}
return m
}
// clobberedCalleeSaved returns the callee-saved registers a function writes
// without also saving and restoring them — i.e. registers whose caller-owned
// value is lost across the call. It walks the blocks of the liveness analysis
// (so the control-flow graph is what supplies the instruction set) and
// aggregates each instruction's register effects.
func clobberedCalleeSaved(l *liveness, a arch.Arch) []string {
callee := calleeSavedGPRs(a)
if len(callee) == 0 {
return nil
}
def := map[string]bool{}
saved := map[string]bool{}
restored := map[string]bool{}
for _, b := range l.blocks {
for _, in := range b.instrs {
eff := instrEffect(in, a)
for _, r := range eff.def {
def[r] = true
}
for _, r := range eff.saveGPR {
saved[r] = true
}
for _, r := range eff.restGPR {
restored[r] = true
}
}
}
var out []string
for r := range callee {
if def[r] && !(saved[r] && restored[r]) {
out = append(out, r)
}
}
sort.Strings(out)
return out
}
+79
View File
@@ -0,0 +1,79 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lint
import "testing"
// TestRegisterClobber detects writes to callee-saved registers that are not
// saved and restored.
func TestRegisterClobber(t *testing.T) {
// BX (callee-saved on amd64) is written but never saved → clobbered.
clob := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVQ CX, BX\n"+
"\tRET\n")
if codes(clob)[CodeRegisterClobber] != 1 {
t.Fatalf("unsaved callee-saved write should be flagged: %+v", clob)
}
// Saved and restored → preserved.
saved := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $8\n"+
"\tPUSHQ BX\n"+
"\tMOVQ CX, BX\n"+
"\tPOPQ BX\n"+
"\tRET\n")
if codes(saved)[CodeRegisterClobber] != 0 {
t.Fatalf("saved/restored register must not be flagged: %+v", saved)
}
// A caller-saved register (CX) is fine to write.
caller := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tMOVQ $1, CX\n"+
"\tRET\n")
if codes(caller)[CodeRegisterClobber] != 0 {
t.Fatalf("caller-saved register must not be flagged: %+v", caller)
}
}
// TestFuncdata validates the FUNCDATA/PCDATA structural checks.
func TestFuncdata(t *testing.T) {
// Well formed: no findings.
good := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tFUNCDATA $0, gclocals·abc(SB)\n"+
"\tPCDATA $1, $0\n"+
"\tRET\n")
if codes(good)[CodeFuncdata] != 0 {
t.Fatalf("well-formed FUNCDATA/PCDATA must not be flagged: %+v", good)
}
// FUNCDATA with one operand.
bad1 := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tFUNCDATA $0\n"+
"\tRET\n")
if codes(bad1)[CodeFuncdata] == 0 {
t.Fatal("FUNCDATA with one operand should be flagged")
}
// PCDATA with a non-immediate value.
bad2 := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tPCDATA $0, AX\n"+
"\tRET\n")
if codes(bad2)[CodeFuncdata] == 0 {
t.Fatal("PCDATA with a register value should be flagged")
}
// FUNCDATA index out of range.
bad3 := lintSrc(t, "#include \"textflag.h\"\n"+
"TEXT ·f(SB), NOSPLIT, $0\n"+
"\tFUNCDATA $99, gclocals·abc(SB)\n"+
"\tRET\n")
if codes(bad3)[CodeFuncdata] == 0 {
t.Fatal("out-of-range FUNCDATA index should be flagged")
}
}