Files
gasm-sdk/lint/lint.go
T

565 lines
18 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package lint runs static checks over a parsed GAsm file. The rules are
// deliberately conservative: where a check cannot be certain (for example an
// instruction whose operand count varies), it stays silent rather than emit a
// false positive. Every diagnostic carries a stable rule code so callers can
// disable individual rules.
package lint
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Severity ranks a diagnostic.
type Severity int
// Diagnostic severities, mirroring the language-server protocol ordering.
const (
Error Severity = iota
Warning
Information
Hint
)
// String returns a lower-case label for the severity.
func (s Severity) String() string {
switch s {
case Error:
return "error"
case Warning:
return "warning"
case Information:
return "information"
default:
return "hint"
}
}
// Diagnostic is one lint finding.
type Diagnostic struct {
Pos token.Position
End token.Position
Severity Severity
Code string
Message string
}
// Config controls a lint run.
type Config struct {
// Arch is the target architecture. When it is arch.Unknown the
// architecture-specific rules (unknown instruction, operand count) are
// skipped because no instruction table can be selected.
Arch arch.Arch
// Disable lists rule codes to suppress.
Disable map[string]bool
}
// Rule codes.
const (
CodeUnknownInstr = "unknown-instruction"
CodeOperandCount = "operand-count"
CodeUndefinedLabel = "undefined-label"
CodeDuplicateLabel = "duplicate-label"
CodeMissingRet = "missing-ret"
CodeMissingTextflag = "missing-textflag-include"
CodeUnreachable = "unreachable-code"
CodeABIArgSize = "abi-argsize"
CodeNosplitFrame = "nosplit-frame"
CodeRegisterClobber = "register-clobber"
CodeFuncdata = "funcdata-pcdata"
)
// pseudoOps are assembler pseudo-operations that are valid instruction-position
// tokens but are not machine instructions and so absent from the arch tables.
var pseudoOps = map[string]bool{
"BYTE": true, "WORD": true, "LONG": true, "QUAD": true, "FLOAT": true,
"PCALIGN": true, "FUNCDATA": true, "PCDATA": true, "GO_ARGS": true,
}
// File lints a parsed file and returns the diagnostics in source order.
func File(f *ast.File, cfg Config) []Diagnostic {
var out []Diagnostic
tab := arch.ForArch(cfg.Arch)
archKnown := cfg.Arch != arch.Unknown
hasTextflag := false
usesFlags := false
var firstFlagPos token.Position
// Macros (in-file #define, or any #include other than textflag.h, which
// only defines flag constants) make label resolution unreliable.
macrosInPlay := len(f.Macros) > 0
// Preprocessor conditionals (#ifdef …) make control-flow analysis
// unreliable, since mutually exclusive branches look sequential.
hasConditionals := false
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Include:
if !strings.Contains(dd.Header.Text, "textflag.h") {
macrosInPlay = true
}
case *ast.Preproc:
if isConditionalDirective(dd.Raw) {
hasConditionals = true
}
}
}
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Include:
if strings.Contains(dd.Header.Text, "textflag.h") {
hasTextflag = true
}
case *ast.Text:
out = append(out, lintText(dd, tab, archKnown, cfg, f.Macros, !macrosInPlay, !hasConditionals)...)
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
usesFlags = true
firstFlagPos = dd.Pos()
}
case *ast.Globl:
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
usesFlags = true
firstFlagPos = dd.Pos()
}
}
}
if !cfg.Disable[CodeMissingTextflag] && usesFlags && !hasTextflag {
out = append(out, Diagnostic{
Pos: firstFlagPos,
Severity: Warning,
Code: CodeMissingTextflag,
Message: "TEXT/GLOBL flags are used but textflag.h is not #included",
})
}
sortDiagnostics(out)
return out
}
// lintText lints one TEXT function body. doLabelChecks is false for files
// that use macros (an in-file #define or a non-textflag #include): without a
// preprocessor we cannot resolve labels that macros define or reference, so the
// label and RET heuristics are suppressed there to avoid false positives.
func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros map[string]bool, doLabelChecks bool, doUnreachable bool) []Diagnostic {
var out []Diagnostic
defined := map[string]token.Position{}
referenced := map[string]token.Position{}
hasRet := false
lastTerminal := false
hasMacro := false
instrCount := 0
dead := false // inside a region unreachable from above
reportedDead := false // the current dead region has already been reported
hasPCRel := referencesPC(t) // PC-relative jumps defeat reachability analysis
hasIndirect := hasIndirectBranch(t) // register-indirect branches do too
// Unreachable-code analysis is only sound in functions whose control flow is
// fully label-resolvable: no PC-relative jumps, no register-indirect
// branches, and (file-level) no preprocessor conditionals.
analyzable := doUnreachable && !hasPCRel && !hasIndirect
for _, s := range t.Body {
switch st := s.(type) {
case *ast.Label:
name := st.Name.Text
if prev, dup := defined[name]; dup {
if !cfg.Disable[CodeDuplicateLabel] {
out = append(out, Diagnostic{
Pos: st.Name.Pos,
End: st.Name.End,
Severity: Error,
Code: CodeDuplicateLabel,
Message: fmt.Sprintf("label %q already defined at %s", name, prev),
})
}
} else {
defined[name] = st.Name.Pos
}
// A label is a jump target: code after it is reachable again.
dead = false
reportedDead = false
case *ast.Instr:
instrCount++
mnem := st.Mnemonic.Text
upper := strings.ToUpper(mnem)
// Unreachable code: a real instruction following a RET/UNDEF and
// before any label, in a function whose control flow is fully
// resolvable. Only RET/UNDEF are treated as terminators here — an
// unconditional jump may be one entry of a hand-arranged branch
// table (e.g. the generated callback tables), so it is not assumed
// to make the following code dead. Pseudo-ops and macro invocations
// are never flagged.
if analyzable && dead && !reportedDead && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) &&
!cfg.Disable[CodeUnreachable] {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Warning,
Code: CodeUnreachable,
Message: "unreachable code after terminating instruction",
})
reportedDead = true
}
// RET never falls through. (UNDEF is a trap/marker rather than a
// control-flow terminator: code placed after it is occasionally
// deliberate metadata, so it is not treated as making the following
// code dead.)
if upper == "RET" {
dead = true
}
// A function need not RET if it ends in an unconditional jump (tail
// call / loop) or in UNDEF (a deliberate trap that never returns).
lastTerminal = isUnconditionalJump(cfg.Arch, upper) || upper == "UNDEF"
if isMacroInvocation(mnem, macros) {
hasMacro = true
}
if upper == "RET" {
hasRet = true
}
if archKnown && !cfg.Disable[CodeUnknownInstr] && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) {
if _, ok := tab.Lookup(mnem); !ok {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Error,
Code: CodeUnknownInstr,
Message: fmt.Sprintf("unknown %s instruction %q", cfg.Arch, mnem),
})
}
}
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) && !maskedEvex(mnem, st.Operands) {
if in, ok := tab.Lookup(mnem); ok && in.MinOps >= 0 {
n := len(st.Operands)
if n < in.MinOps || n > in.MaxOps {
out = append(out, Diagnostic{
Pos: st.Mnemonic.Pos,
End: st.Mnemonic.End,
Severity: Warning,
Code: CodeOperandCount,
Message: fmt.Sprintf("%s expects %s, got %d operand(s)",
mnem, countRange(in.MinOps, in.MaxOps), n),
})
}
}
}
if isJump(cfg.Arch, upper) {
for _, op := range st.Operands {
if name, pos, ok := localLabelRef(op); ok && !tab.IsRegister(name) && !arch.IsPseudoReg(name) {
referenced[name] = pos
}
}
}
}
}
// Undefined labels.
if doLabelChecks && !cfg.Disable[CodeUndefinedLabel] {
for name, pos := range referenced {
if _, ok := defined[name]; !ok {
out = append(out, Diagnostic{
Pos: pos,
Severity: Error,
Code: CodeUndefinedLabel,
Message: fmt.Sprintf("jump to undefined label %q", name),
})
}
}
}
// Missing RET heuristic. Functions that invoke a macro are skipped: the
// macro body (opaque to us) may supply the RET.
if doLabelChecks && !cfg.Disable[CodeMissingRet] && instrCount > 0 && !hasRet && !lastTerminal && !hasMacro {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeMissingRet,
Message: fmt.Sprintf("function %q has no RET", t.Name.Name),
})
}
// ABI conformance: the argument area declared in the TEXT directive should
// match the size computed from the // func signature in the doc comment.
// Only applies to stack-argument (ABI0) functions, which reference their
// arguments through FP; register-ABI functions declare a zero arg area. Also
// skipped when there is no parseable signature or it uses an unknown type.
if !cfg.Disable[CodeABIArgSize] {
got := int64(0)
if t.Args != nil && t.Args.Imm.HasVal {
got = t.Args.Imm.Val
}
// Only meaningful for stack-argument (ABI0) functions: a non-zero
// declared arg area that is actually addressed through FP.
if got > 0 && usesFPArgs(t) {
if want, ok := abiExpectedArgSize(t.Doc); ok {
if want != got {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeABIArgSize,
Message: fmt.Sprintf("TEXT declares arg size %d but the // func signature implies %d", got, want),
})
}
}
}
}
// Register liveness: a register the Go ABI fixes across calls that is
// written but never saved and restored is clobbered. The check runs over
// the control-flow graph and is skipped for macro-using files, where an
// opaque macro may perform the save/restore.
if doLabelChecks && archKnown && !cfg.Disable[CodeRegisterClobber] {
live := analyzeLiveness(t, cfg.Arch)
always, rt := clobberedGoFixed(live, cfg.Arch, reachesRuntime(t))
if len(always) > 0 {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeRegisterClobber,
Message: fmt.Sprintf("register(s) %s written but never saved/restored: fixed by the Go ABI (frame/goroutine pointer)", strings.Join(always, ", ")),
})
}
if len(rt) > 0 {
out = append(out, Diagnostic{
Pos: t.Keyword.Pos,
Severity: Warning,
Code: CodeRegisterClobber,
Message: fmt.Sprintf("goroutine-pointer register(s) %s written but never saved/restored in a function that can reach the Go runtime", strings.Join(rt, ", ")),
})
}
}
// FUNCDATA / PCDATA structural validation.
out = append(out, checkFuncdata(t, cfg)...)
return out
}
// reachesRuntime reports whether a function can reach the Go runtime: it is
// not NOSPLIT (so the stack-split and traceback machinery runs) or it makes a
// CALL. Goroutine-pointer registers must survive such functions; a NOSPLIT
// leaf may clobber them, since the ABI0 transition restores them (the
// runtime's own assembly relies on this, e.g. R14 on amd64).
func reachesRuntime(t *ast.Text) bool {
nosplit := false
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
nosplit = true
}
}
for _, s := range t.Body {
if in, ok := s.(*ast.Instr); ok {
switch strings.ToUpper(in.Mnemonic.Text) {
case "CALL", "BL", "JAL": // amd64, arm64/loong64, riscv64 calls
return true
}
}
}
return !nosplit
}
// usesFPArgs reports whether a function references its arguments through the FP
// pseudo-register — i.e. it uses the stack-based ABI0 layout, where the
// declared argument size must match the signature.
func usesFPArgs(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
for _, op := range in.Operands {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
return true
}
}
}
return false
}
// referencesPC reports whether a function uses a PC-relative operand (e.g.
// `JMP 2(PC)`). Such jumps target a computed offset rather than a label, so
// reachability cannot be determined statically and the unreachable-code check
// is suppressed for the whole function.
func referencesPC(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
for _, op := range in.Operands {
if strings.Contains(strings.ReplaceAll(op.Raw, " ", ""), "(PC)") {
return true
}
}
}
return false
}
// hasIndirectBranch reports whether a function transfers control through a
// register (JALR/JR/JIRL/BR/BLR). Such targets are computed at runtime, so
// reachability cannot be determined statically and the unreachable-code check is
// suppressed for the whole function.
func hasIndirectBranch(t *ast.Text) bool {
for _, s := range t.Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "JALR", "JR", "JIRL", "BR", "BLR":
return true
}
}
return false
}
// isMacroInvocation reports whether a mnemonic is a macro invocation rather
// than a machine instruction. No Plan 9 mnemonic contains an underscore, so an
// underscore is a reliable macro marker (the runtime headers define macros such
// as get_tls and NO_LOCAL_POINTERS). Names introduced by an in-file #define
// are recognised too (CALLFN, DISPATCH, …). Full macro expansion is out of
// scope; this only keeps the linter quiet on invocations it cannot expand.
func isMacroInvocation(mnem string, macros map[string]bool) bool {
return strings.Contains(mnem, "_") || macros[mnem]
}
// maskedEvex reports whether the instruction is a masked EVEX form: the
// mnemonic carries a .Z suffix, or the operand list contains an opmask
// register (K1–K7). Either way the operand count differs from the unmasked
// form, so count checks are skipped.
func maskedEvex(mnem string, ops []*ast.Operand) bool {
if strings.Contains(mnem, ".") {
return true
}
for _, op := range ops {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Base == "" &&
op.Addr.Index == "" && op.Addr.Sym.Pseudo == "" && isMaskReg(op.Addr.Sym.Name) {
return true
}
}
return false
}
// isMaskReg reports whether name is an opmask register K0–K7.
func isMaskReg(name string) bool {
return len(name) == 2 && name[0] == 'K' && name[1] >= '0' && name[1] <= '7'
}
// isConditionalDirective reports whether a preprocessor directive (the text
// after '#') is a conditional-compilation directive whose branches the parser
// cannot resolve.
func isConditionalDirective(raw string) bool {
fields := strings.Fields(raw)
if len(fields) == 0 {
return false
}
switch fields[0] {
case "if", "ifdef", "ifndef", "else", "elif", "endif":
return true
}
return false
}
// localLabelRef returns the name and position of a bare local-label reference
// operand (no pseudo-register, no memory base), if op is one.
func localLabelRef(op *ast.Operand) (string, token.Position, bool) {
if op == nil || op.Kind != ast.OpAddr || op.Addr.Sym == nil {
return "", token.Position{}, false
}
sym := op.Addr.Sym
if sym.Pseudo != "" || op.Addr.Base != "" || sym.Name == "" {
return "", token.Position{}, false
}
return sym.Name, op.Pos, true
}
// riscvBranches and loong64Branches are the conditional-branch mnemonics; they
// are listed explicitly rather than matched by a "B" prefix so that bit-manip
// instructions (BCLR, BSET, …) are never mistaken for branches.
var riscvBranches = map[string]bool{
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
"BEQZ": true, "BNEZ": true, "BLEZ": true, "BGEZ": true, "BLTZ": true, "BGTZ": true,
}
var loong64Branches = map[string]bool{
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
"BLEZ": true, "BLTZ": true, "BGEZ": true, "BGTZ": true,
}
// isJump reports whether the mnemonic is any branch.
func isJump(a arch.Arch, upper string) bool {
switch a {
case arch.ARM64:
return upper == "CALL" || upper == "BR" || upper == "BLR" || upper == "JMP" ||
strings.HasPrefix(upper, "B") ||
strings.HasPrefix(upper, "CBZ") || strings.HasPrefix(upper, "CBNZ") ||
strings.HasPrefix(upper, "TBZ") || strings.HasPrefix(upper, "TBNZ")
case arch.RISCV:
return upper == "CALL" || riscvBranches[upper] ||
upper == "JMP" || upper == "J" || upper == "JAL" || upper == "JALR" ||
upper == "JR" || upper == "BR"
case arch.LOONG64:
return upper == "CALL" || loong64Branches[upper] ||
upper == "JIRL" || upper == "JMP" || upper == "BR"
default: // amd64
return upper == "CALL" || strings.HasPrefix(upper, "J")
}
}
// isUnconditionalJump reports whether the mnemonic is an unconditional branch
// (used to suppress the missing-RET heuristic for tail calls and loops).
func isUnconditionalJump(a arch.Arch, upper string) bool {
switch a {
case arch.ARM64:
return upper == "B" || upper == "BR" || upper == "JMP"
case arch.RISCV:
return upper == "JMP" || upper == "J" || upper == "JAL" ||
upper == "JALR" || upper == "JR" || upper == "BR"
case arch.LOONG64:
return upper == "JMP" || upper == "JIRL" || upper == "BR"
default:
return upper == "JMP"
}
}
func countRange(min, max int) string {
if min == max {
return fmt.Sprintf("%d operand(s)", min)
}
return fmt.Sprintf("%d–%d operands", min, max)
}
// sortDiagnostics orders diagnostics by line, then column, then code.
func sortDiagnostics(d []Diagnostic) {
for i := 1; i < len(d); i++ {
for j := i; j > 0 && lessDiag(d[j], d[j-1]); j-- {
d[j], d[j-1] = d[j-1], d[j]
}
}
}
func lessDiag(a, b Diagnostic) bool {
if a.Pos.Line != b.Pos.Line {
return a.Pos.Line < b.Pos.Line
}
if a.Pos.Column != b.Pos.Column {
return a.Pos.Column < b.Pos.Column
}
return a.Code < b.Code
}