2026-07-06 09:49:50 +02:00
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package lint runs static checks over a parsed GAsm file. The rules are
// deliberately conservative: where a check cannot be certain (for example an
// instruction whose operand count varies), it stays silent rather than emit a
// false positive. Every diagnostic carries a stable rule code so callers can
// disable individual rules.
package lint
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Severity ranks a diagnostic.
type Severity int
// Diagnostic severities, mirroring the language-server protocol ordering.
const (
Error Severity = iota
Warning
Information
Hint
)
// String returns a lower-case label for the severity.
func ( s Severity ) String () string {
switch s {
case Error :
return "error"
case Warning :
return "warning"
case Information :
return "information"
default :
return "hint"
}
}
// Diagnostic is one lint finding.
type Diagnostic struct {
Pos token . Position
End token . Position
Severity Severity
Code string
Message string
}
// Config controls a lint run.
type Config struct {
// Arch is the target architecture. When it is arch.Unknown the
// architecture-specific rules (unknown instruction, operand count) are
// skipped because no instruction table can be selected.
Arch arch . Arch
// Disable lists rule codes to suppress.
Disable map [ string ] bool
}
// Rule codes.
const (
CodeUnknownInstr = "unknown-instruction"
CodeOperandCount = "operand-count"
CodeUndefinedLabel = "undefined-label"
CodeDuplicateLabel = "duplicate-label"
CodeMissingRet = "missing-ret"
CodeMissingTextflag = "missing-textflag-include"
CodeUnreachable = "unreachable-code"
CodeABIArgSize = "abi-argsize"
CodeNosplitFrame = "nosplit-frame"
CodeRegisterClobber = "register-clobber"
CodeFuncdata = "funcdata-pcdata"
)
// pseudoOps are assembler pseudo-operations that are valid instruction-position
// tokens but are not machine instructions and so absent from the arch tables.
var pseudoOps = map [ string ] bool {
"BYTE" : true , "WORD" : true , "LONG" : true , "QUAD" : true , "FLOAT" : true ,
"PCALIGN" : true , "FUNCDATA" : true , "PCDATA" : true , "GO_ARGS" : true ,
}
// File lints a parsed file and returns the diagnostics in source order.
func File ( f * ast . File , cfg Config ) [] Diagnostic {
var out [] Diagnostic
tab := arch . ForArch ( cfg . Arch )
archKnown := cfg . Arch != arch . Unknown
hasTextflag := false
usesFlags := false
var firstFlagPos token . Position
// Macros (in-file #define, or any #include other than textflag.h, which
// only defines flag constants) make label resolution unreliable.
macrosInPlay := len ( f . Macros ) > 0
// Preprocessor conditionals (#ifdef …) make control-flow analysis
// unreliable, since mutually exclusive branches look sequential.
hasConditionals := false
for _ , d := range f . Decls {
switch dd := d .( type ) {
case * ast . Include :
if ! strings . Contains ( dd . Header . Text , "textflag.h" ) {
macrosInPlay = true
}
case * ast . Preproc :
if isConditionalDirective ( dd . Raw ) {
hasConditionals = true
}
}
}
for _ , d := range f . Decls {
switch dd := d .( type ) {
case * ast . Include :
if strings . Contains ( dd . Header . Text , "textflag.h" ) {
hasTextflag = true
}
case * ast . Text :
out = append ( out , lintText ( dd , tab , archKnown , cfg , f . Macros , ! macrosInPlay , ! hasConditionals ) ... )
if len ( dd . Flags ) > 0 && ! firstFlagPos . IsValid () {
usesFlags = true
firstFlagPos = dd . Pos ()
}
case * ast . Globl :
if len ( dd . Flags ) > 0 && ! firstFlagPos . IsValid () {
usesFlags = true
firstFlagPos = dd . Pos ()
}
}
}
if ! cfg . Disable [ CodeMissingTextflag ] && usesFlags && ! hasTextflag {
out = append ( out , Diagnostic {
Pos : firstFlagPos ,
Severity : Warning ,
Code : CodeMissingTextflag ,
Message : "TEXT/GLOBL flags are used but textflag.h is not #included" ,
})
}
sortDiagnostics ( out )
return out
}
// lintText lints one TEXT function body. doLabelChecks is false for files
// that use macros (an in-file #define or a non-textflag #include): without a
// preprocessor we cannot resolve labels that macros define or reference, so the
// label and RET heuristics are suppressed there to avoid false positives.
func lintText ( t * ast . Text , tab * arch . Table , archKnown bool , cfg Config , macros map [ string ] bool , doLabelChecks bool , doUnreachable bool ) [] Diagnostic {
var out [] Diagnostic
defined := map [ string ] token . Position {}
referenced := map [ string ] token . Position {}
hasRet := false
lastTerminal := false
hasMacro := false
instrCount := 0
dead := false // inside a region unreachable from above
reportedDead := false // the current dead region has already been reported
hasPCRel := referencesPC ( t ) // PC-relative jumps defeat reachability analysis
hasIndirect := hasIndirectBranch ( t ) // register-indirect branches do too
// Unreachable-code analysis is only sound in functions whose control flow is
// fully label-resolvable: no PC-relative jumps, no register-indirect
// branches, and (file-level) no preprocessor conditionals.
analyzable := doUnreachable && ! hasPCRel && ! hasIndirect
for _ , s := range t . Body {
switch st := s .( type ) {
case * ast . Label :
name := st . Name . Text
if prev , dup := defined [ name ]; dup {
if ! cfg . Disable [ CodeDuplicateLabel ] {
out = append ( out , Diagnostic {
Pos : st . Name . Pos ,
End : st . Name . End ,
Severity : Error ,
Code : CodeDuplicateLabel ,
Message : fmt . Sprintf ( "label %q already defined at %s" , name , prev ),
})
}
} else {
defined [ name ] = st . Name . Pos
}
// A label is a jump target: code after it is reachable again.
dead = false
reportedDead = false
case * ast . Instr :
instrCount ++
mnem := st . Mnemonic . Text
upper := strings . ToUpper ( mnem )
// Unreachable code: a real instruction following a RET/UNDEF and
// before any label, in a function whose control flow is fully
// resolvable. Only RET/UNDEF are treated as terminators here — an
// unconditional jump may be one entry of a hand-arranged branch
// table (e.g. the generated callback tables), so it is not assumed
// to make the following code dead. Pseudo-ops and macro invocations
// are never flagged.
if analyzable && dead && ! reportedDead && ! pseudoOps [ upper ] && ! isMacroInvocation ( mnem , macros ) &&
! cfg . Disable [ CodeUnreachable ] {
out = append ( out , Diagnostic {
Pos : st . Mnemonic . Pos ,
End : st . Mnemonic . End ,
Severity : Warning ,
Code : CodeUnreachable ,
Message : "unreachable code after terminating instruction" ,
})
reportedDead = true
}
// RET never falls through. (UNDEF is a trap/marker rather than a
// control-flow terminator: code placed after it is occasionally
// deliberate metadata, so it is not treated as making the following
// code dead.)
if upper == "RET" {
dead = true
}
// A function need not RET if it ends in an unconditional jump (tail
// call / loop) or in UNDEF (a deliberate trap that never returns).
lastTerminal = isUnconditionalJump ( cfg . Arch , upper ) || upper == "UNDEF"
if isMacroInvocation ( mnem , macros ) {
hasMacro = true
}
if upper == "RET" {
hasRet = true
}
if archKnown && ! cfg . Disable [ CodeUnknownInstr ] && ! pseudoOps [ upper ] && ! isMacroInvocation ( mnem , macros ) {
if _ , ok := tab . Lookup ( mnem ); ! ok {
out = append ( out , Diagnostic {
Pos : st . Mnemonic . Pos ,
End : st . Mnemonic . End ,
Severity : Error ,
Code : CodeUnknownInstr ,
Message : fmt . Sprintf ( "unknown %s instruction %q" , cfg . Arch , mnem ),
})
}
}
2026-07-14 21:03:26 +02:00
if archKnown && ! cfg . Disable [ CodeOperandCount ] && ! isMacroInvocation ( mnem , macros ) && ! maskedEvex ( mnem , st . Operands ) {
2026-07-06 09:49:50 +02:00
if in , ok := tab . Lookup ( mnem ); ok && in . MinOps >= 0 {
n := len ( st . Operands )
if n < in . MinOps || n > in . MaxOps {
out = append ( out , Diagnostic {
Pos : st . Mnemonic . Pos ,
End : st . Mnemonic . End ,
Severity : Warning ,
Code : CodeOperandCount ,
Message : fmt . Sprintf ( "%s expects %s, got %d operand(s)" ,
mnem , countRange ( in . MinOps , in . MaxOps ), n ),
})
}
}
}
if isJump ( cfg . Arch , upper ) {
for _ , op := range st . Operands {
if name , pos , ok := localLabelRef ( op ); ok && ! tab . IsRegister ( name ) && ! arch . IsPseudoReg ( name ) {
referenced [ name ] = pos
}
}
}
}
}
// Undefined labels.
if doLabelChecks && ! cfg . Disable [ CodeUndefinedLabel ] {
for name , pos := range referenced {
if _ , ok := defined [ name ]; ! ok {
out = append ( out , Diagnostic {
Pos : pos ,
Severity : Error ,
Code : CodeUndefinedLabel ,
Message : fmt . Sprintf ( "jump to undefined label %q" , name ),
})
}
}
}
// Missing RET heuristic. Functions that invoke a macro are skipped: the
// macro body (opaque to us) may supply the RET.
if doLabelChecks && ! cfg . Disable [ CodeMissingRet ] && instrCount > 0 && ! hasRet && ! lastTerminal && ! hasMacro {
out = append ( out , Diagnostic {
Pos : t . Keyword . Pos ,
Severity : Warning ,
Code : CodeMissingRet ,
Message : fmt . Sprintf ( "function %q has no RET" , t . Name . Name ),
})
}
// ABI conformance: the argument area declared in the TEXT directive should
// match the size computed from the // func signature in the doc comment.
// Only applies to stack-argument (ABI0) functions, which reference their
// arguments through FP; register-ABI functions declare a zero arg area. Also
// skipped when there is no parseable signature or it uses an unknown type.
if ! cfg . Disable [ CodeABIArgSize ] {
got := int64 ( 0 )
if t . Args != nil && t . Args . Imm . HasVal {
got = t . Args . Imm . Val
}
// Only meaningful for stack-argument (ABI0) functions: a non-zero
// declared arg area that is actually addressed through FP.
if got > 0 && usesFPArgs ( t ) {
if want , ok := abiExpectedArgSize ( t . Doc ); ok {
if want != got {
out = append ( out , Diagnostic {
Pos : t . Keyword . Pos ,
Severity : Warning ,
Code : CodeABIArgSize ,
Message : fmt . Sprintf ( "TEXT declares arg size %d but the // func signature implies %d" , got , want ),
})
}
}
}
}
2026-07-11 17:36:52 +02:00
// Register liveness: a register the Go ABI fixes across calls that is
// written but never saved and restored is clobbered. The check runs over
// the control-flow graph and is skipped for macro-using files, where an
// opaque macro may perform the save/restore.
2026-07-06 09:49:50 +02:00
if doLabelChecks && archKnown && ! cfg . Disable [ CodeRegisterClobber ] {
live := analyzeLiveness ( t , cfg . Arch )
2026-07-11 17:36:52 +02:00
always , rt := clobberedGoFixed ( live , cfg . Arch , reachesRuntime ( t ))
if len ( always ) > 0 {
2026-07-06 09:49:50 +02:00
out = append ( out , Diagnostic {
Pos : t . Keyword . Pos ,
Severity : Warning ,
Code : CodeRegisterClobber ,
2026-07-11 17:36:52 +02:00
Message : fmt . Sprintf ( "register(s) %s written but never saved/restored: fixed by the Go ABI (frame/goroutine pointer)" , strings . Join ( always , ", " )),
})
}
if len ( rt ) > 0 {
out = append ( out , Diagnostic {
Pos : t . Keyword . Pos ,
Severity : Warning ,
Code : CodeRegisterClobber ,
Message : fmt . Sprintf ( "goroutine-pointer register(s) %s written but never saved/restored in a function that can reach the Go runtime" , strings . Join ( rt , ", " )),
2026-07-06 09:49:50 +02:00
})
}
}
// FUNCDATA / PCDATA structural validation.
out = append ( out , checkFuncdata ( t , cfg ) ... )
return out
}
2026-07-11 17:36:52 +02:00
// reachesRuntime reports whether a function can reach the Go runtime: it is
// not NOSPLIT (so the stack-split and traceback machinery runs) or it makes a
// CALL. Goroutine-pointer registers must survive such functions; a NOSPLIT
// leaf may clobber them, since the ABI0 transition restores them (the
// runtime's own assembly relies on this, e.g. R14 on amd64).
func reachesRuntime ( t * ast . Text ) bool {
nosplit := false
for _ , f := range t . Flags {
if strings . EqualFold ( f , "NOSPLIT" ) {
nosplit = true
}
}
for _ , s := range t . Body {
if in , ok := s .( * ast . Instr ); ok {
switch strings . ToUpper ( in . Mnemonic . Text ) {
case "CALL" , "BL" , "JAL" : // amd64, arm64/loong64, riscv64 calls
return true
}
}
}
return ! nosplit
}
2026-07-06 09:49:50 +02:00
// usesFPArgs reports whether a function references its arguments through the FP
// pseudo-register — i.e. it uses the stack-based ABI0 layout, where the
// declared argument size must match the signature.
func usesFPArgs ( t * ast . Text ) bool {
for _ , s := range t . Body {
in , ok := s .( * ast . Instr )
if ! ok {
continue
}
for _ , op := range in . Operands {
if op . Addr . Sym != nil && op . Addr . Sym . Pseudo == "FP" {
return true
}
}
}
return false
}
// referencesPC reports whether a function uses a PC-relative operand (e.g.
// `JMP 2(PC)`). Such jumps target a computed offset rather than a label, so
// reachability cannot be determined statically and the unreachable-code check
// is suppressed for the whole function.
func referencesPC ( t * ast . Text ) bool {
for _ , s := range t . Body {
in , ok := s .( * ast . Instr )
if ! ok {
continue
}
for _ , op := range in . Operands {
if strings . Contains ( strings . ReplaceAll ( op . Raw , " " , "" ), "(PC)" ) {
return true
}
}
}
return false
}
// hasIndirectBranch reports whether a function transfers control through a
// register (JALR/JR/JIRL/BR/BLR). Such targets are computed at runtime, so
// reachability cannot be determined statically and the unreachable-code check is
// suppressed for the whole function.
func hasIndirectBranch ( t * ast . Text ) bool {
for _ , s := range t . Body {
in , ok := s .( * ast . Instr )
if ! ok {
continue
}
switch strings . ToUpper ( in . Mnemonic . Text ) {
case "JALR" , "JR" , "JIRL" , "BR" , "BLR" :
return true
}
}
return false
}
// isMacroInvocation reports whether a mnemonic is a macro invocation rather
// than a machine instruction. No Plan 9 mnemonic contains an underscore, so an
// underscore is a reliable macro marker (the runtime headers define macros such
// as get_tls and NO_LOCAL_POINTERS). Names introduced by an in-file #define
// are recognised too (CALLFN, DISPATCH, …). Full macro expansion is out of
// scope; this only keeps the linter quiet on invocations it cannot expand.
func isMacroInvocation ( mnem string , macros map [ string ] bool ) bool {
return strings . Contains ( mnem , "_" ) || macros [ mnem ]
}
2026-07-14 21:03:26 +02:00
// maskedEvex reports whether the instruction is a masked EVEX form: the
// mnemonic carries a .Z suffix, or the operand list contains an opmask
// register (K1– K7). Either way the operand count differs from the unmasked
// form, so count checks are skipped.
func maskedEvex ( mnem string , ops [] * ast . Operand ) bool {
if strings . Contains ( mnem , "." ) {
return true
}
for _ , op := range ops {
if op . Kind == ast . OpAddr && op . Addr . Sym != nil && op . Addr . Base == "" &&
op . Addr . Index == "" && op . Addr . Sym . Pseudo == "" && isMaskReg ( op . Addr . Sym . Name ) {
return true
}
}
return false
}
// isMaskReg reports whether name is an opmask register K0– K7.
func isMaskReg ( name string ) bool {
return len ( name ) == 2 && name [ 0 ] == 'K' && name [ 1 ] >= '0' && name [ 1 ] <= '7'
}
2026-07-06 09:49:50 +02:00
// isConditionalDirective reports whether a preprocessor directive (the text
// after '#') is a conditional-compilation directive whose branches the parser
// cannot resolve.
func isConditionalDirective ( raw string ) bool {
fields := strings . Fields ( raw )
if len ( fields ) == 0 {
return false
}
switch fields [ 0 ] {
case "if" , "ifdef" , "ifndef" , "else" , "elif" , "endif" :
return true
}
return false
}
// localLabelRef returns the name and position of a bare local-label reference
// operand (no pseudo-register, no memory base), if op is one.
func localLabelRef ( op * ast . Operand ) ( string , token . Position , bool ) {
if op == nil || op . Kind != ast . OpAddr || op . Addr . Sym == nil {
return "" , token . Position {}, false
}
sym := op . Addr . Sym
if sym . Pseudo != "" || op . Addr . Base != "" || sym . Name == "" {
return "" , token . Position {}, false
}
return sym . Name , op . Pos , true
}
// riscvBranches and loong64Branches are the conditional-branch mnemonics; they
// are listed explicitly rather than matched by a "B" prefix so that bit-manip
// instructions (BCLR, BSET, …) are never mistaken for branches.
var riscvBranches = map [ string ] bool {
"BEQ" : true , "BNE" : true , "BLT" : true , "BGE" : true , "BLTU" : true , "BGEU" : true ,
"BEQZ" : true , "BNEZ" : true , "BLEZ" : true , "BGEZ" : true , "BLTZ" : true , "BGTZ" : true ,
}
var loong64Branches = map [ string ] bool {
"BEQ" : true , "BNE" : true , "BLT" : true , "BGE" : true , "BLTU" : true , "BGEU" : true ,
"BLEZ" : true , "BLTZ" : true , "BGEZ" : true , "BGTZ" : true ,
}
// isJump reports whether the mnemonic is any branch.
func isJump ( a arch . Arch , upper string ) bool {
switch a {
case arch . ARM64 :
return upper == "CALL" || upper == "BR" || upper == "BLR" || upper == "JMP" ||
strings . HasPrefix ( upper , "B" ) ||
strings . HasPrefix ( upper , "CBZ" ) || strings . HasPrefix ( upper , "CBNZ" ) ||
strings . HasPrefix ( upper , "TBZ" ) || strings . HasPrefix ( upper , "TBNZ" )
case arch . RISCV :
return upper == "CALL" || riscvBranches [ upper ] ||
upper == "JMP" || upper == "J" || upper == "JAL" || upper == "JALR" ||
upper == "JR" || upper == "BR"
case arch . LOONG64 :
return upper == "CALL" || loong64Branches [ upper ] ||
upper == "JIRL" || upper == "JMP" || upper == "BR"
default : // amd64
return upper == "CALL" || strings . HasPrefix ( upper , "J" )
}
}
// isUnconditionalJump reports whether the mnemonic is an unconditional branch
// (used to suppress the missing-RET heuristic for tail calls and loops).
func isUnconditionalJump ( a arch . Arch , upper string ) bool {
switch a {
case arch . ARM64 :
return upper == "B" || upper == "BR" || upper == "JMP"
case arch . RISCV :
return upper == "JMP" || upper == "J" || upper == "JAL" ||
upper == "JALR" || upper == "JR" || upper == "BR"
case arch . LOONG64 :
return upper == "JMP" || upper == "JIRL" || upper == "BR"
default :
return upper == "JMP"
}
}
func countRange ( min , max int ) string {
if min == max {
return fmt . Sprintf ( "%d operand(s)" , min )
}
return fmt . Sprintf ( "%d– %d operands" , min , max )
}
// sortDiagnostics orders diagnostics by line, then column, then code.
func sortDiagnostics ( d [] Diagnostic ) {
for i := 1 ; i < len ( d ); i ++ {
for j := i ; j > 0 && lessDiag ( d [ j ], d [ j - 1 ]); j -- {
d [ j ], d [ j - 1 ] = d [ j - 1 ], d [ j ]
}
}
}
func lessDiag ( a , b Diagnostic ) bool {
if a . Pos . Line != b . Pos . Line {
return a . Pos . Line < b . Pos . Line
}
if a . Pos . Column != b . Pos . Column {
return a . Pos . Column < b . Pos . Column
}
return a . Code < b . Code
}