Files
gasm-sdk/lint/lint.go
T
2026-07-14 21:03:26 +02:00

565 lines
18 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package lint runs static checks over a parsed GAsm file. The rules are
// deliberately conservative: where a check cannot be certain (for example an
// instruction whose operand count varies), it stays silent rather than emit a
// false positive. Every diagnostic carries a stable rule code so callers can
// disable individual rules.
package lint
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Severity ranks a diagnostic.
type Severity int
// Diagnostic severities, mirroring the language-server protocol ordering.
const (
Error Severity = iota
Warning
Information
Hint
)
// String returns a lower-case label for the severity.
func (s Severity) String() string {
switch s {
case Error:
return "error"
case Warning:
return "warning"
case Information:
return "information"
default:
return "hint"
}
}
// Diagnostic is one lint finding.
type Diagnostic struct {
Pos token.Position
End token.Position
Severity Severity
Code string
Message string
}
// Config controls a lint run.
type Config struct {
// Arch is the target architecture. When it is arch.Unknown the
// architecture-specific rules (unknown instruction, operand count) are
// skipped because no instruction table can be selected.
Arch arch.Arch
// Disable lists rule codes to suppress.
Disable map[string]bool
}
// Rule codes.
const (
CodeUnknownInstr = "unknown-instruction"
CodeOperandCount = "operand-count"
CodeUndefinedLabel = "undefined-label"
CodeDuplicateLabel = "duplicate-label"
CodeMissingRet = "missing-ret"
CodeMissingTextflag = "missing-textflag-include"
CodeUnreachable = "unreachable-code"
CodeABIArgSize = "abi-argsize"
CodeNosplitFrame = "nosplit-frame"
CodeRegisterClobber = "register-clobber"
CodeFuncdata = "funcdata-pcdata"
)
// pseudoOps are assembler pseudo-operations that are valid instruction-position
// tokens but are not machine instructions and so absent from the arch tables.
var pseudoOps = map[string]bool{
"BYTE": true, "WORD": true, "LONG": true, "QUAD": true, "FLOAT": true,
"PCALIGN": true, "FUNCDATA": true, "PCDATA": true, "GO_ARGS": true,
}
// File lints a parsed file and returns the diagnostics in source order.
func File(f *ast.File, cfg Config) []Diagnostic {
var out []Diagnostic
tab := arch.ForArch(cfg.Arch)
archKnown := cfg.Arch != arch.Unknown
hasTextflag := false
usesFlags := false
var firstFlagPos token.Position
// Macros (in-file #define, or any #include other than textflag.h, which
// only defines flag constants) make label resolution unreliable.
macrosInPlay := len(f.Macros) > 0
// Preprocessor conditionals (#ifdef …) make control-flow analysis
// unreliable, since mutually exclusive branches look sequential.
hasConditionals := false
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Include:
if !strings.Contains(dd.Header.Text, "textflag.h") {
macrosInPlay = true
}
case *ast.Preproc:
if isConditionalDirective(dd.Raw) {
hasConditionals = true
}
}
}
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Include:
if strings.Contains(dd.Header.Text, "textflag.h") {
hasTextflag = true
}
case *ast.Text:
out = append(out, lintText(dd, tab, archKnown, cfg, f.Macros, !macrosInPlay, !hasConditionals)...)
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
usesFlags = true
firstFlagPos = dd.Pos()
}
case *ast.Globl:
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
usesFlags = true
firstFlagPos = dd.Pos()
}
}
}
if !cfg.Disable[CodeMissingTextflag] && usesFlags && !hasTextflag {
out = append(out, Diagnostic{
Pos: firstFlagPos,
Severity: Warning,
Code: CodeMissingTextflag,
Message: "TEXT/GLOBL flags are used but textflag.h is not #included",
})
}
sortDiagnostics(out)
return out
}
// lintText lints one TEXT function body. doLabelChecks is false for files
// that use macros (an in-file #define or a non-textflag #include): without a
// preprocessor we cannot resolve labels that macros define or reference, so the
// label and RET heuristics are suppressed there to avoid false positives.
func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros map[string]bool, doLabelChecks bool, doUnreachable bool) []Diagnostic {
var out []Diagnostic
defined := map[string]token.Position{}
referenced := map[string]token.Position{}
hasRet := false
lastTerminal := false
hasMacro := false
instrCount := 0
dead := false // inside a region unreachable from above
reportedDead := false // the current dead region has already been reported
hasPCRel := referencesPC(t) // PC-relative jumps defeat reachability analysis
hasIndirect := hasIndirectBranch(t) // register-indirect branches do too
// Unreachable-code analysis is only sound in functions whose control flow is
// fully label-resolvable: no PC-relative jumps, no register-indirect
// branches, and (file-level) no preprocessor conditionals.
analyzable := doUnreachable && !hasPCRel && !hasIndirect
for _, s := range t.Body {
switch st := s.(type) {
case *ast.Label:
name := st.Name.Text
if prev, dup := defined[name]; dup {
if !cfg.Disable[CodeDuplicateLabel] {
out = append(out, Diagnostic{
Pos: st.Name.Pos,
End: st.Name.End,
Severity: Error,
Code: CodeDuplicateLabel,
Message: fmt.Sprintf("label %q already defined at %s", name, prev),
})
}
} else {
defined[name] = st.Name.Pos
}
// A label is a jump target: code after it is reachable again.
dead = false
reportedDead = false
case *ast.Instr:
instrCount++
mnem := st.Mnemonic.Text
upper := strings.ToUpper(mnem)
// Unreachable code: a real instruction following a RET/UNDEF and
// before any label, in a function whose control flow is fully
// resolvable. Only RET/UNDEF are treated as terminators here — an
// unconditional jump may be one entry of a hand-arranged branch
// table (e.g. the generated callback tables), so it is not assumed
// to make the following code dead. Pseudo-ops and macro invocations
// are never flagged.
if analyzable && dead && !reportedDead && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) &&
!cfg.Disable[CodeUnreachable] {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Warning,
Code: CodeUnreachable,
Message: "unreachable code after terminating instruction",
})
reportedDead = true
}
// RET never falls through. (UNDEF is a trap/marker rather than a
// control-flow terminator: code placed after it is occasionally
// deliberate metadata, so it is not treated as making the following
// code dead.)
if upper == "RET" {
dead = true
}
// A function need not RET if it ends in an unconditional jump (tail
// call / loop) or in UNDEF (a deliberate trap that never returns).
lastTerminal = isUnconditionalJump(cfg.Arch, upper) || upper == "UNDEF"
if isMacroInvocation(mnem, macros) {
hasMacro = true
}
if upper == "RET" {
hasRet = true
}
if archKnown && !cfg.Disable[CodeUnknownInstr] && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) {
if _, ok := tab.Lookup(mnem); !ok {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Error,
Code: CodeUnknownInstr,
Message: fmt.Sprintf("unknown %s instruction %q", cfg.Arch, mnem),
})
}
}
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) && !maskedEvex(mnem, st.Operands) {
if in, ok := tab.Lookup(mnem); ok && in.MinOps >= 0 {
n := len(st.Operands)
if n < in.MinOps || n > in.MaxOps {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Warning,
Code: CodeOperandCount,
Message: fmt.Sprintf("%s expects %s, got %d operand(s)",
mnem, countRange(in.MinOps, in.MaxOps), n),
})
}
}
}
if isJump(cfg.Arch, upper) {
for _, op := range st.Operands {
if name, pos, ok := localLabelRef(op); ok && !tab.IsRegister(name) && !arch.IsPseudoReg(name) {
referenced[name] = pos
}
}
}
}
}
// Undefined labels.
if doLabelChecks && !cfg.Disable[CodeUndefinedLabel] {
for name, pos := range referenced {
if _, ok := defined[name]; !ok {
out = append(out, Diagnostic{
Pos: pos,
Severity: Error,
Code: CodeUndefinedLabel,
Message: fmt.Sprintf("jump to undefined label %q", name),
})
}
}
}
// Missing RET heuristic. Functions that invoke a macro are skipped: the
// macro body (opaque to us) may supply the RET.
if doLabelChecks && !cfg.Disable[CodeMissingRet] && instrCount > 0 && !hasRet && !lastTerminal && !hasMacro {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeMissingRet,
Message: fmt.Sprintf("function %q has no RET", t.Name.Name),
})
}
// ABI conformance: the argument area declared in the TEXT directive should
// match the size computed from the // func signature in the doc comment.
// Only applies to stack-argument (ABI0) functions, which reference their
// arguments through FP; register-ABI functions declare a zero arg area. Also
// skipped when there is no parseable signature or it uses an unknown type.
if !cfg.Disable[CodeABIArgSize] {
got := int64(0)
if t.Args != nil && t.Args.Imm.HasVal {
got = t.Args.Imm.Val
}
// Only meaningful for stack-argument (ABI0) functions: a non-zero
// declared arg area that is actually addressed through FP.
if got > 0 && usesFPArgs(t) {
if want, ok := abiExpectedArgSize(t.Doc); ok {
if want != got {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeABIArgSize,
Message: fmt.Sprintf("TEXT declares arg size %d but the // func signature implies %d", got, want),
})
}
}
}
}
// Register liveness: a register the Go ABI fixes across calls that is
// written but never saved and restored is clobbered. The check runs over
// the control-flow graph and is skipped for macro-using files, where an
// opaque macro may perform the save/restore.
if doLabelChecks && archKnown && !cfg.Disable[CodeRegisterClobber] {
live := analyzeLiveness(t, cfg.Arch)
always, rt := clobberedGoFixed(live, cfg.Arch, reachesRuntime(t))
if len(always) > 0 {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeRegisterClobber,
Message: fmt.Sprintf("register(s) %s written but never saved/restored: fixed by the Go ABI (frame/goroutine pointer)", strings.Join(always, ", ")),
})
}
if len(rt) > 0 {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeRegisterClobber,
Message: fmt.Sprintf("goroutine-pointer register(s) %s written but never saved/restored in a function that can reach the Go runtime", strings.Join(rt, ", ")),
})
}
}
// FUNCDATA / PCDATA structural validation.
out = append(out, checkFuncdata(t, cfg)...)
return out
}
// reachesRuntime reports whether a function can reach the Go runtime: it is
// not NOSPLIT (so the stack-split and traceback machinery runs) or it makes a
// CALL. Goroutine-pointer registers must survive such functions; a NOSPLIT
// leaf may clobber them, since the ABI0 transition restores them (the
// runtime's own assembly relies on this, e.g. R14 on amd64).
func reachesRuntime(t *ast.Text) bool {
nosplit := false
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
nosplit = true
}
}
for _, s := range t.Body {
if in, ok := s.(*ast.Instr); ok {
switch strings.ToUpper(in.Mnemonic.Text) {
case "CALL", "BL", "JAL": // amd64, arm64/loong64, riscv64 calls
return true
}
}
}
return !nosplit
}
// usesFPArgs reports whether a function references its arguments through the FP
// pseudo-register — i.e. it uses the stack-based ABI0 layout, where the
// declared argument size must match the signature.
func usesFPArgs(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
for _, op := range in.Operands {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
return true
}
}
}
return false
}
// referencesPC reports whether a function uses a PC-relative operand (e.g.
// `JMP 2(PC)`). Such jumps target a computed offset rather than a label, so
// reachability cannot be determined statically and the unreachable-code check
// is suppressed for the whole function.
func referencesPC(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
for _, op := range in.Operands {
if strings.Contains(strings.ReplaceAll(op.Raw, " ", ""), "(PC)") {
return true
}
}
}
return false
}
// hasIndirectBranch reports whether a function transfers control through a
// register (JALR/JR/JIRL/BR/BLR). Such targets are computed at runtime, so
// reachability cannot be determined statically and the unreachable-code check is
// suppressed for the whole function.
func hasIndirectBranch(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "JALR", "JR", "JIRL", "BR", "BLR":
return true
}
}
return false
}
// isMacroInvocation reports whether a mnemonic is a macro invocation rather
// than a machine instruction. No Plan 9 mnemonic contains an underscore, so an
// underscore is a reliable macro marker (the runtime headers define macros such
// as get_tls and NO_LOCAL_POINTERS). Names introduced by an in-file #define
// are recognised too (CALLFN, DISPATCH, …). Full macro expansion is out of
// scope; this only keeps the linter quiet on invocations it cannot expand.
func isMacroInvocation(mnem string, macros map[string]bool) bool {
return strings.Contains(mnem, "_") || macros[mnem]
}
// maskedEvex reports whether the instruction is a masked EVEX form: the
// mnemonic carries a .Z suffix, or the operand list contains an opmask
// register (K1–K7). Either way the operand count differs from the unmasked
// form, so count checks are skipped.
func maskedEvex(mnem string, ops []*ast.Operand) bool {
if strings.Contains(mnem, ".") {
return true
}
for _, op := range ops {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Base == "" &&
op.Addr.Index == "" && op.Addr.Sym.Pseudo == "" && isMaskReg(op.Addr.Sym.Name) {
return true
}
}
return false
}
// isMaskReg reports whether name is an opmask register K0–K7.
func isMaskReg(name string) bool {
return len(name) == 2 && name[0] == 'K' && name[1] >= '0' && name[1] <= '7'
}
// isConditionalDirective reports whether a preprocessor directive (the text
// after '#') is a conditional-compilation directive whose branches the parser
// cannot resolve.
func isConditionalDirective(raw string) bool {
fields := strings.Fields(raw)
if len(fields) == 0 {
return false
}
switch fields[0] {
case "if", "ifdef", "ifndef", "else", "elif", "endif":
return true
}
return false
}
// localLabelRef returns the name and position of a bare local-label reference
// operand (no pseudo-register, no memory base), if op is one.
func localLabelRef(op *ast.Operand) (string, token.Position, bool) {
if op == nil || op.Kind != ast.OpAddr || op.Addr.Sym == nil {
return "", token.Position{}, false
}
sym := op.Addr.Sym
if sym.Pseudo != "" || op.Addr.Base != "" || sym.Name == "" {
return "", token.Position{}, false
}
return sym.Name, op.Pos, true
}
// riscvBranches and loong64Branches are the conditional-branch mnemonics; they
// are listed explicitly rather than matched by a "B" prefix so that bit-manip
// instructions (BCLR, BSET, …) are never mistaken for branches.
var riscvBranches = map[string]bool{
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
"BEQZ": true, "BNEZ": true, "BLEZ": true, "BGEZ": true, "BLTZ": true, "BGTZ": true,
}
var loong64Branches = map[string]bool{
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
"BLEZ": true, "BLTZ": true, "BGEZ": true, "BGTZ": true,
}
// isJump reports whether the mnemonic is any branch.
func isJump(a arch.Arch, upper string) bool {
switch a {
case arch.ARM64:
return upper == "CALL" || upper == "BR" || upper == "BLR" || upper == "JMP" ||
strings.HasPrefix(upper, "B") ||
strings.HasPrefix(upper, "CBZ") || strings.HasPrefix(upper, "CBNZ") ||
strings.HasPrefix(upper, "TBZ") || strings.HasPrefix(upper, "TBNZ")
case arch.RISCV:
return upper == "CALL" || riscvBranches[upper] ||
upper == "JMP" || upper == "J" || upper == "JAL" || upper == "JALR" ||
upper == "JR" || upper == "BR"
case arch.LOONG64:
return upper == "CALL" || loong64Branches[upper] ||
upper == "JIRL" || upper == "JMP" || upper == "BR"
default: // amd64
return upper == "CALL" || strings.HasPrefix(upper, "J")
}
}
// isUnconditionalJump reports whether the mnemonic is an unconditional branch
// (used to suppress the missing-RET heuristic for tail calls and loops).
func isUnconditionalJump(a arch.Arch, upper string) bool {
switch a {
case arch.ARM64:
return upper == "B" || upper == "BR" || upper == "JMP"
case arch.RISCV:
return upper == "JMP" || upper == "J" || upper == "JAL" ||
upper == "JALR" || upper == "JR" || upper == "BR"
case arch.LOONG64:
return upper == "JMP" || upper == "JIRL" || upper == "BR"
default:
return upper == "JMP"
}
}
func countRange(min, max int) string {
if min == max {
return fmt.Sprintf("%d operand(s)", min)
}
return fmt.Sprintf("%d–%d operands", min, max)
}
// sortDiagnostics orders diagnostics by line, then column, then code.
func sortDiagnostics(d []Diagnostic) {
for i := 1; i < len(d); i++ {
for j := i; j > 0 && lessDiag(d[j], d[j-1]); j-- {
d[j], d[j-1] = d[j-1], d[j]
}
}
}
func lessDiag(a, b Diagnostic) bool {
if a.Pos.Line != b.Pos.Line {
return a.Pos.Line < b.Pos.Line
}
if a.Pos.Column != b.Pos.Column {
return a.Pos.Column < b.Pos.Column
}
return a.Code < b.Code
}