984 lines
33 KiB
Go
984 lines
33 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
// Package lint runs static checks over a parsed GAsm file. The rules are
|
|
// deliberately conservative: where a check cannot be certain (for example an
|
|
// instruction whose operand count varies), it stays silent rather than emit a
|
|
// false positive. Every diagnostic carries a stable rule code so callers can
|
|
// disable individual rules.
|
|
package lint
|
|
|
|
import (
|
|
"fmt"
|
|
"strconv"
|
|
"strings"
|
|
"unicode/utf8"
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
|
"sourcedock.dev/petrbalvin/gasm-sdk/token"
|
|
)
|
|
|
|
// Severity ranks a diagnostic.
|
|
type Severity int
|
|
|
|
// Diagnostic severities, mirroring the language-server protocol ordering.
|
|
const (
|
|
Error Severity = iota
|
|
Warning
|
|
Information
|
|
Hint
|
|
)
|
|
|
|
// String returns a lower-case label for the severity.
|
|
func (s Severity) String() string {
|
|
switch s {
|
|
case Error:
|
|
return "error"
|
|
case Warning:
|
|
return "warning"
|
|
case Information:
|
|
return "information"
|
|
default:
|
|
return "hint"
|
|
}
|
|
}
|
|
|
|
// Diagnostic is one lint finding.
|
|
type Diagnostic struct {
|
|
Pos token.Position
|
|
End token.Position
|
|
Severity Severity
|
|
Code string
|
|
Message string
|
|
}
|
|
|
|
// Config controls a lint run.
|
|
type Config struct {
|
|
// Arch is the target architecture. When it is arch.Unknown the
|
|
// architecture-specific rules (unknown instruction, operand count) are
|
|
// skipped because no instruction table can be selected.
|
|
Arch arch.Arch
|
|
// Disable lists rule codes to suppress.
|
|
Disable map[string]bool
|
|
}
|
|
|
|
// Rule codes.
|
|
const (
|
|
CodeUnknownInstr = "unknown-instruction"
|
|
CodeOperandCount = "operand-count"
|
|
CodeUndefinedLabel = "undefined-label"
|
|
CodeDuplicateLabel = "duplicate-label"
|
|
CodeMissingRet = "missing-ret"
|
|
CodeMissingTextflag = "missing-textflag-include"
|
|
CodeUnreachable = "unreachable-code"
|
|
CodeABIArgSize = "abi-argsize"
|
|
CodeRegisterClobber = "register-clobber"
|
|
CodeFuncdata = "funcdata-pcdata"
|
|
CodeUnusedLabel = "unused-label"
|
|
CodeInvalidFlag = "invalid-textflag"
|
|
CodeStackImbalance = "stack-imbalance"
|
|
CodeRegisterWidthMismatch = "register-width-mismatch"
|
|
CodeABI0RegisterArgs = "abi0-register-args"
|
|
CodeNonportableRegister = "nonportable-register-name"
|
|
CodeUnencodable = "unencodable-instruction"
|
|
CodeReservedRegister = "reserved-register-write"
|
|
)
|
|
|
|
// knownTextFlags are the flags recognised by runtime/textflag.h, plus the
|
|
// older REFLECTED spelling of REFLECTMETHOD.
|
|
var knownTextFlags = map[string]bool{
|
|
"NOSPLIT": true, "DUPOK": true, "RODATA": true, "NOPROF": true,
|
|
"NOPTR": true, "WRAPPER": true, "NEEDCTXT": true, "TLSBSS": true,
|
|
"NOFRAME": true, "REFLECTED": true, "REFLECTMETHOD": true,
|
|
"TOPFRAME": true, "ABIWRAPPER": true,
|
|
}
|
|
|
|
// pseudoOps are assembler pseudo-operations that are valid instruction-position
|
|
// tokens but are not machine instructions and so absent from the arch tables.
|
|
var pseudoOps = map[string]bool{
|
|
"BYTE": true, "WORD": true, "LONG": true, "QUAD": true, "FLOAT": true,
|
|
"PCALIGN": true, "FUNCDATA": true, "PCDATA": true, "GO_ARGS": true,
|
|
}
|
|
|
|
// File lints a parsed file and returns the diagnostics in source order.
|
|
func File(f *ast.File, cfg Config) []Diagnostic {
|
|
var out []Diagnostic
|
|
tab := arch.ForArch(cfg.Arch)
|
|
archKnown := cfg.Arch != arch.Unknown
|
|
|
|
hasTextflag := false
|
|
usesFlags := false
|
|
var firstFlagPos token.Position
|
|
|
|
// Macros (in-file #define, or any #include other than textflag.h, which
|
|
// only defines flag constants) make label resolution unreliable.
|
|
macrosInPlay := len(f.Macros) > 0
|
|
// Preprocessor conditionals (#ifdef …) make control-flow analysis
|
|
// unreliable, since mutually exclusive branches look sequential.
|
|
hasConditionals := false
|
|
for _, d := range f.Decls {
|
|
switch dd := d.(type) {
|
|
case *ast.Include:
|
|
if !strings.Contains(dd.Header.Text, "textflag.h") {
|
|
macrosInPlay = true
|
|
}
|
|
case *ast.Preproc:
|
|
if isConditionalDirective(dd.Raw) {
|
|
hasConditionals = true
|
|
}
|
|
}
|
|
}
|
|
|
|
for _, d := range f.Decls {
|
|
switch dd := d.(type) {
|
|
case *ast.Include:
|
|
if strings.Contains(dd.Header.Text, "textflag.h") {
|
|
hasTextflag = true
|
|
}
|
|
case *ast.Text:
|
|
out = append(out, lintText(dd, tab, archKnown, cfg, f.Macros, !macrosInPlay, !hasConditionals)...)
|
|
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
|
|
usesFlags = true
|
|
firstFlagPos = dd.Pos()
|
|
}
|
|
case *ast.Globl:
|
|
if len(dd.Flags) > 0 && !firstFlagPos.IsValid() {
|
|
usesFlags = true
|
|
firstFlagPos = dd.Pos()
|
|
}
|
|
}
|
|
}
|
|
|
|
if !cfg.Disable[CodeMissingTextflag] && usesFlags && !hasTextflag {
|
|
out = append(out, Diagnostic{
|
|
Pos: firstFlagPos,
|
|
Severity: Warning,
|
|
Code: CodeMissingTextflag,
|
|
Message: "TEXT/GLOBL flags are used but textflag.h is not #included",
|
|
})
|
|
}
|
|
|
|
// Invalid TEXT/GLOBL flags.
|
|
if !cfg.Disable[CodeInvalidFlag] {
|
|
for _, d := range f.Decls {
|
|
var flags []string
|
|
var pos token.Position
|
|
switch dd := d.(type) {
|
|
case *ast.Text:
|
|
flags = dd.Flags
|
|
pos = dd.Keyword.Pos
|
|
case *ast.Globl:
|
|
flags = dd.Flags
|
|
if dd.Name != nil {
|
|
pos = dd.Name.Pos
|
|
}
|
|
}
|
|
for _, fl := range flags {
|
|
// Numeric flags are legacy textflag.h constants (1, 2, 8,
|
|
// 9, 10, …); their meaning is decided at assembly time.
|
|
if _, err := strconv.Atoi(fl); err == nil {
|
|
continue
|
|
}
|
|
if !knownTextFlags[fl] {
|
|
out = append(out, Diagnostic{
|
|
Pos: pos,
|
|
Severity: Warning,
|
|
Code: CodeInvalidFlag,
|
|
Message: fmt.Sprintf("unknown TEXT/GLOBL flag %q", fl),
|
|
})
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
sortDiagnostics(out)
|
|
return out
|
|
}
|
|
|
|
// lintText lints one TEXT function body. doLabelChecks is false for files
|
|
// that use macros (an in-file #define or a non-textflag #include): without a
|
|
// preprocessor we cannot resolve labels that macros define or reference, so the
|
|
// label and RET heuristics are suppressed there to avoid false positives.
|
|
func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros map[string]bool, doLabelChecks bool, doUnreachable bool) []Diagnostic {
|
|
var out []Diagnostic
|
|
|
|
defined := map[string]token.Position{}
|
|
referenced := map[string]token.Position{}
|
|
hasRet := false
|
|
lastTerminal := false
|
|
hasMacro := false
|
|
instrCount := 0
|
|
dead := false // inside a region unreachable from above
|
|
reportedDead := false // the current dead region has already been reported
|
|
hasPCRel := referencesPC(t) // PC-relative jumps defeat reachability analysis
|
|
hasIndirect := hasIndirectBranch(t, tab) // register-indirect branches do too
|
|
// Unreachable-code analysis is only sound in functions whose control flow is
|
|
// fully label-resolvable: no PC-relative jumps, no register-indirect
|
|
// branches, and (file-level) no preprocessor conditionals.
|
|
analyzable := doUnreachable && !hasPCRel && !hasIndirect
|
|
|
|
for _, s := range t.Body {
|
|
switch st := s.(type) {
|
|
case *ast.Label:
|
|
name := st.Name.Text
|
|
if prev, dup := defined[name]; dup {
|
|
if !cfg.Disable[CodeDuplicateLabel] {
|
|
out = append(out, Diagnostic{
|
|
Pos: st.Name.Pos,
|
|
End: st.Name.End,
|
|
Severity: Error,
|
|
Code: CodeDuplicateLabel,
|
|
Message: fmt.Sprintf("label %q already defined at %s", name, prev),
|
|
})
|
|
}
|
|
} else {
|
|
defined[name] = st.Name.Pos
|
|
}
|
|
// A label is a jump target: code after it is reachable again.
|
|
dead = false
|
|
reportedDead = false
|
|
|
|
case *ast.Instr:
|
|
instrCount++
|
|
mnem := st.Mnemonic.Text
|
|
upper := strings.ToUpper(mnem)
|
|
|
|
// Unreachable code: a real instruction following a RET/UNDEF and
|
|
// before any label, in a function whose control flow is fully
|
|
// resolvable. Only RET/UNDEF are treated as terminators here, an
|
|
// unconditional jump may be one entry of a hand-arranged branch
|
|
// table (e.g. the generated callback tables), so it is not assumed
|
|
// to make the following code dead. Pseudo-ops and macro invocations
|
|
// are never flagged.
|
|
if analyzable && dead && !reportedDead && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) &&
|
|
!cfg.Disable[CodeUnreachable] {
|
|
out = append(out, Diagnostic{
|
|
Pos: st.Mnemonic.Pos,
|
|
End: st.Mnemonic.End,
|
|
Severity: Warning,
|
|
Code: CodeUnreachable,
|
|
Message: "unreachable code after terminating instruction",
|
|
})
|
|
reportedDead = true
|
|
}
|
|
|
|
// RET never falls through. (UNDEF is a trap/marker rather than a
|
|
// control-flow terminator: code placed after it is occasionally
|
|
// deliberate metadata, so it is not treated as making the following
|
|
// code dead.)
|
|
if upper == "RET" {
|
|
dead = true
|
|
}
|
|
// A function need not RET if it ends in an unconditional jump (tail
|
|
// call / loop) or in UNDEF (a deliberate trap that never returns).
|
|
lastTerminal = isUnconditionalJump(cfg.Arch, upper) || upper == "UNDEF"
|
|
if isMacroInvocation(mnem, macros) {
|
|
hasMacro = true
|
|
}
|
|
|
|
if upper == "RET" {
|
|
hasRet = true
|
|
}
|
|
|
|
if archKnown && !cfg.Disable[CodeUnknownInstr] && !pseudoOps[upper] && !isMacroInvocation(mnem, macros) {
|
|
if _, ok := tab.Lookup(mnem); !ok {
|
|
out = append(out, Diagnostic{
|
|
Pos: st.Mnemonic.Pos,
|
|
End: st.Mnemonic.End,
|
|
Severity: Error,
|
|
Code: CodeUnknownInstr,
|
|
Message: fmt.Sprintf("unknown %s instruction %q", cfg.Arch, mnem),
|
|
})
|
|
} else if cfg.Arch == arch.AMD64 && !cfg.Disable[CodeUnencodable] && !asm.Encodable(upper) {
|
|
// Known to the architecture table but missing from the
|
|
// encoder: the file parses everywhere and then fails at
|
|
// assembly time. Flag it at lint so the gap is visible
|
|
// in the editor, and so the audit can close it.
|
|
out = append(out, Diagnostic{
|
|
Pos: st.Mnemonic.Pos,
|
|
End: st.Mnemonic.End,
|
|
Severity: Warning,
|
|
Code: CodeUnencodable,
|
|
Message: fmt.Sprintf("instruction %q is known but the encoder cannot assemble it yet", mnem),
|
|
})
|
|
}
|
|
}
|
|
|
|
if archKnown && !cfg.Disable[CodeOperandCount] && !isMacroInvocation(mnem, macros) && !maskedEvex(mnem, st.Operands) {
|
|
if in, ok := tab.Lookup(mnem); ok && in.MinOps >= 0 {
|
|
n := len(st.Operands)
|
|
if n < in.MinOps || n > in.MaxOps {
|
|
out = append(out, Diagnostic{
|
|
Pos: st.Mnemonic.Pos,
|
|
End: st.Mnemonic.End,
|
|
Severity: Warning,
|
|
Code: CodeOperandCount,
|
|
Message: fmt.Sprintf("%s expects %s, got %d operand(s)",
|
|
mnem, countRange(in.MinOps, in.MaxOps), n),
|
|
})
|
|
}
|
|
}
|
|
}
|
|
|
|
// Register-width mismatch: amd64 instructions with Q suffix
|
|
// should use 64-bit registers, L/W/B suffix should use
|
|
// 32/16/8-bit registers.
|
|
if cfg.Arch == arch.AMD64 && !cfg.Disable[CodeRegisterWidthMismatch] && !isMacroInvocation(mnem, macros) {
|
|
if msg := checkRegisterWidth(upper, st.Operands); msg != "" {
|
|
out = append(out, Diagnostic{
|
|
Pos: st.Mnemonic.Pos,
|
|
End: st.Mnemonic.End,
|
|
Severity: Warning,
|
|
Code: CodeRegisterWidthMismatch,
|
|
Message: msg,
|
|
})
|
|
}
|
|
}
|
|
|
|
if isJump(cfg.Arch, upper) {
|
|
if name, pos, ok := branchTargetRef(cfg.Arch, upper, st.Operands, tab); ok {
|
|
referenced[name] = pos
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Undefined labels.
|
|
if doLabelChecks && !cfg.Disable[CodeUndefinedLabel] {
|
|
for name, pos := range referenced {
|
|
if _, ok := defined[name]; !ok {
|
|
out = append(out, Diagnostic{
|
|
Pos: pos,
|
|
Severity: Error,
|
|
Code: CodeUndefinedLabel,
|
|
Message: fmt.Sprintf("jump to undefined label %q", name),
|
|
})
|
|
}
|
|
}
|
|
}
|
|
|
|
// Unused labels: defined but never referenced. Suppressed when macros
|
|
// or indirect branches are present (the reference may be invisible).
|
|
if doLabelChecks && !cfg.Disable[CodeUnusedLabel] && !hasIndirect {
|
|
for name, pos := range defined {
|
|
if _, ok := referenced[name]; !ok {
|
|
out = append(out, Diagnostic{
|
|
Pos: pos,
|
|
// Columns are rune-based, so the end offset is too.
|
|
End: token.Position{Line: pos.Line, Column: pos.Column + utf8.RuneCountInString(name)},
|
|
Severity: Hint,
|
|
Code: CodeUnusedLabel,
|
|
Message: fmt.Sprintf("label %q is defined but never referenced", name),
|
|
})
|
|
}
|
|
}
|
|
}
|
|
|
|
// Missing RET heuristic. Functions that invoke a macro are skipped: the
|
|
// macro body (opaque to us) may supply the RET. A TEXT whose symbol is
|
|
// missing (already reported by the parser) is skipped too.
|
|
if doLabelChecks && !cfg.Disable[CodeMissingRet] && t.Name != nil &&
|
|
instrCount > 0 && !hasRet && !lastTerminal && !hasMacro {
|
|
out = append(out, Diagnostic{
|
|
Pos: t.Keyword.Pos,
|
|
Severity: Warning,
|
|
Code: CodeMissingRet,
|
|
Message: fmt.Sprintf("function %q has no RET", t.Name.Name),
|
|
})
|
|
}
|
|
|
|
// Stack imbalance: track SP changes and flag if the net delta at RET
|
|
// does not match the declared frame size. Only checked for functions
|
|
// with a declared frame, no macros, and no indirect branches.
|
|
if doLabelChecks && !cfg.Disable[CodeStackImbalance] && !hasMacro && !hasIndirect {
|
|
frameSize := int64(0)
|
|
if t.Frame != nil && t.Frame.Imm.HasVal {
|
|
frameSize = t.Frame.Imm.Val
|
|
}
|
|
if frameSize > 0 {
|
|
delta := stackDelta(t, cfg.Arch)
|
|
if delta != 0 && delta != -frameSize {
|
|
out = append(out, Diagnostic{
|
|
Pos: t.Keyword.Pos,
|
|
Severity: Warning,
|
|
Code: CodeStackImbalance,
|
|
Message: fmt.Sprintf("function %q has net SP delta %d (frame size %d)", t.Name.Name, delta, frameSize),
|
|
})
|
|
}
|
|
}
|
|
}
|
|
|
|
// ABI conformance: the argument area declared in the TEXT directive should
|
|
// match the size computed from the // func signature in the doc comment.
|
|
// Only applies to stack-argument (ABI0) functions, which reference their
|
|
// arguments through FP; register-ABI functions declare a zero arg area. Also
|
|
// skipped when there is no parseable signature or it uses an unknown type.
|
|
if !cfg.Disable[CodeABIArgSize] {
|
|
got := int64(0)
|
|
if t.Args != nil && t.Args.Imm.HasVal {
|
|
got = t.Args.Imm.Val
|
|
}
|
|
// Only meaningful for stack-argument (ABI0) functions: a non-zero
|
|
// declared arg area that is actually addressed through FP.
|
|
if got > 0 && usesFPArgs(t) {
|
|
if want, ok := abiExpectedArgSize(t.Doc); ok {
|
|
if want != got {
|
|
out = append(out, Diagnostic{
|
|
Pos: t.Keyword.Pos,
|
|
Severity: Warning,
|
|
Code: CodeABIArgSize,
|
|
Message: fmt.Sprintf("TEXT declares arg size %d but the // func signature implies %d", got, want),
|
|
})
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Register liveness: a register the Go ABI fixes across calls that is
|
|
// written but never saved and restored is clobbered. The check runs over
|
|
// the control-flow graph and is skipped for macro-using files, where an
|
|
// opaque macro may perform the save/restore.
|
|
if doLabelChecks && archKnown && !cfg.Disable[CodeRegisterClobber] {
|
|
live := analyzeLiveness(t, cfg.Arch)
|
|
always, rt := clobberedGoFixed(live, cfg.Arch, reachesRuntime(t))
|
|
if len(always) > 0 {
|
|
out = append(out, Diagnostic{
|
|
Pos: t.Keyword.Pos,
|
|
Severity: Warning,
|
|
Code: CodeRegisterClobber,
|
|
Message: fmt.Sprintf("register(s) %s written but never saved/restored: fixed by the Go ABI (frame/goroutine pointer)", strings.Join(always, ", ")),
|
|
})
|
|
}
|
|
if len(rt) > 0 {
|
|
out = append(out, Diagnostic{
|
|
Pos: t.Keyword.Pos,
|
|
Severity: Warning,
|
|
Code: CodeRegisterClobber,
|
|
Message: fmt.Sprintf("goroutine-pointer register(s) %s written but never saved/restored in a function that can reach the Go runtime", strings.Join(rt, ", ")),
|
|
})
|
|
}
|
|
}
|
|
|
|
// ABI0 argument-read check: every parameter the // func signature
|
|
// declares must be read from the frame (name+offset(FP)). A kernel that
|
|
// declares parameters but never touches their frame slots is almost
|
|
// certainly reading its arguments from registers (AX, BX, …), which is
|
|
// the classic ABI0 port bug: Go pushes the arguments on the stack and
|
|
// the register content is whatever the caller left behind.
|
|
if cfg.Arch == arch.AMD64 && !cfg.Disable[CodeABI0RegisterArgs] && !hasMacro && instrCount > 0 {
|
|
out = append(out, checkABI0Args(t)...)
|
|
}
|
|
|
|
// Non-portable register spellings (RAX/EAX under go tool asm).
|
|
if cfg.Arch == arch.AMD64 && !cfg.Disable[CodeNonportableRegister] {
|
|
out = append(out, scanNonportableRegisters(t)...)
|
|
}
|
|
|
|
// Writes to the platform-reserved register (arm64 R18), which the ABI
|
|
// checks cannot observe at runtime.
|
|
if cfg.Arch == arch.ARM64 && !hasMacro && !cfg.Disable[CodeReservedRegister] {
|
|
out = append(out, scanReservedRegisterWrites(t, cfg.Arch)...)
|
|
}
|
|
|
|
// FUNCDATA / PCDATA structural validation.
|
|
out = append(out, checkFuncdata(t, cfg)...)
|
|
|
|
return out
|
|
}
|
|
|
|
// reachesRuntime reports whether a function can reach the Go runtime: it is
|
|
// not NOSPLIT (so the stack-split and traceback machinery runs) or it makes a
|
|
// CALL. Goroutine-pointer registers must survive such functions; a NOSPLIT
|
|
// leaf may clobber them, since the ABI0 transition restores them (the
|
|
// runtime's own assembly relies on this, e.g. R14 on amd64).
|
|
func reachesRuntime(t *ast.Text) bool {
|
|
nosplit := false
|
|
for _, f := range t.Flags {
|
|
if strings.EqualFold(f, "NOSPLIT") {
|
|
nosplit = true
|
|
}
|
|
}
|
|
for _, s := range t.Body {
|
|
if in, ok := s.(*ast.Instr); ok {
|
|
switch strings.ToUpper(in.Mnemonic.Text) {
|
|
case "CALL", "BL", "JAL": // amd64, arm64/loong64, riscv64 calls
|
|
return true
|
|
}
|
|
}
|
|
}
|
|
return !nosplit
|
|
}
|
|
|
|
// usesFPArgs reports whether a function references its arguments through the FP
|
|
// pseudo-register, i.e. it uses the stack-based ABI0 layout, where the
|
|
// declared argument size must match the signature.
|
|
func usesFPArgs(t *ast.Text) bool {
|
|
for _, s := range t.Body {
|
|
in, ok := s.(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
for _, op := range in.Operands {
|
|
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
|
|
return true
|
|
}
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// referencesPC reports whether a function uses a PC-relative operand (e.g.
|
|
// `JMP 2(PC)`). Such jumps target a computed offset rather than a label, so
|
|
// reachability cannot be determined statically and the unreachable-code check
|
|
// is suppressed for the whole function.
|
|
func referencesPC(t *ast.Text) bool {
|
|
for _, s := range t.Body {
|
|
in, ok := s.(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
for _, op := range in.Operands {
|
|
if strings.Contains(strings.ReplaceAll(op.Raw, " ", ""), "(PC)") {
|
|
return true
|
|
}
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// hasIndirectBranch reports whether a function transfers control through a
|
|
// register or a computed memory address: the RISC branch-register mnemonics
|
|
// (JALR/JR/JIRL/BR/BLR), or a JMP/CALL whose target is a register or memory
|
|
// operand rather than a label or symbol. Such targets are computed at
|
|
// runtime, so reachability cannot be determined statically and the
|
|
// unreachable-code check is suppressed for the whole function.
|
|
func hasIndirectBranch(t *ast.Text, tab *arch.Table) bool {
|
|
for _, s := range t.Body {
|
|
in, ok := s.(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
switch strings.ToUpper(in.Mnemonic.Text) {
|
|
case "JALR", "JR", "JIRL", "BR", "BLR":
|
|
return true
|
|
case "JMP", "CALL":
|
|
if indirectJumpTarget(in, tab) {
|
|
return true
|
|
}
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// indirectJumpTarget reports whether the JMP/CALL operand addresses a
|
|
// register or a memory location rather than a label or a static symbol. The
|
|
// parser delivers a bare register and a bare label in the same shape, so
|
|
// register membership decides.
|
|
func indirectJumpTarget(in *ast.Instr, tab *arch.Table) bool {
|
|
if len(in.Operands) != 1 || in.Operands[0].Kind != ast.OpAddr {
|
|
return false
|
|
}
|
|
a := in.Operands[0].Addr
|
|
if a.Base != "" || a.Index != "" {
|
|
return true
|
|
}
|
|
return a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" && tab.IsRegister(a.Sym.Name)
|
|
}
|
|
|
|
// isMacroInvocation reports whether a mnemonic is a macro invocation rather
|
|
// than a machine instruction. No Plan 9 mnemonic contains an underscore, so an
|
|
// underscore is a reliable macro marker (the runtime headers define macros such
|
|
// as get_tls and NO_LOCAL_POINTERS). Names introduced by an in-file #define
|
|
// are recognised too (CALLFN, DISPATCH, …). Full macro expansion is out of
|
|
// scope; this only keeps the linter quiet on invocations it cannot expand.
|
|
func isMacroInvocation(mnem string, macros map[string]bool) bool {
|
|
return strings.Contains(mnem, "_") || macros[mnem]
|
|
}
|
|
|
|
// maskedEvex reports whether the instruction is a masked EVEX form: the
|
|
// mnemonic carries a .Z suffix, or the operand list contains an opmask
|
|
// register (K1-K7). Either way the operand count differs from the unmasked
|
|
// form, so count checks are skipped.
|
|
func maskedEvex(mnem string, ops []*ast.Operand) bool {
|
|
if strings.Contains(mnem, ".") {
|
|
return true
|
|
}
|
|
for _, op := range ops {
|
|
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Base == "" &&
|
|
op.Addr.Index == "" && op.Addr.Sym.Pseudo == "" && isMaskReg(op.Addr.Sym.Name) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// isMaskReg reports whether name is an opmask register K0-K7.
|
|
func isMaskReg(name string) bool {
|
|
return len(name) == 2 && name[0] == 'K' && name[1] >= '0' && name[1] <= '7'
|
|
}
|
|
|
|
// isConditionalDirective reports whether a preprocessor directive (the text
|
|
// after '#') is a conditional-compilation directive whose branches the parser
|
|
// cannot resolve.
|
|
func isConditionalDirective(raw string) bool {
|
|
fields := strings.Fields(raw)
|
|
if len(fields) == 0 {
|
|
return false
|
|
}
|
|
switch fields[0] {
|
|
case "if", "ifdef", "ifndef", "else", "elif", "endif":
|
|
return true
|
|
}
|
|
return false
|
|
}
|
|
|
|
// localLabelRef returns the name and position of a bare local-label reference
|
|
// operand (no pseudo-register, no memory base), if op is one.
|
|
func localLabelRef(op *ast.Operand) (string, token.Position, bool) {
|
|
if op == nil || op.Kind != ast.OpAddr || op.Addr.Sym == nil {
|
|
return "", token.Position{}, false
|
|
}
|
|
sym := op.Addr.Sym
|
|
if sym.Pseudo != "" || op.Addr.Base != "" || sym.Name == "" {
|
|
return "", token.Position{}, false
|
|
}
|
|
return sym.Name, op.Pos, true
|
|
}
|
|
|
|
// branchTargetRef returns the local label a branch transfers control to: the
|
|
// bare symbol in the destination position, the last operand, since that is
|
|
// where the Plan 9 branch target sits. A register-named target is a
|
|
// register-indirect branch (JMP AX, arm64 BR R5, riscv64 JALR X6, loong64
|
|
// JIRL R1) and yields no reference, unless the encoder reads the target
|
|
// positionally (positionalBranchTarget): there a label may legitimately
|
|
// collide with a register alias, riscv64 ZERO being the ABI name of X0, and
|
|
// a label named zero is ordinary code.
|
|
func branchTargetRef(a arch.Arch, upper string, ops []*ast.Operand, tab *arch.Table) (string, token.Position, bool) {
|
|
if len(ops) == 0 {
|
|
return "", token.Position{}, false
|
|
}
|
|
name, pos, ok := localLabelRef(ops[len(ops)-1])
|
|
if !ok {
|
|
return "", token.Position{}, false
|
|
}
|
|
if !positionalBranchTarget(a, upper) && (tab.IsRegister(name) || arch.IsPseudoReg(name)) {
|
|
return "", token.Position{}, false
|
|
}
|
|
return name, pos, true
|
|
}
|
|
|
|
// positionalBranchTarget reports whether the encoder reads a bare-symbol
|
|
// operand of the branch as its label target from a fixed position, without
|
|
// consulting the register file. The riscv64 branch, JMP and JAL encoders do
|
|
// (labelFromOperand in asm/riscv_assemble.go), as do the loong64 branch,
|
|
// BFPT/BFPF and jump encoders (l64Label in asm/loong64_assemble.go). amd64
|
|
// never does, because a bare register operand to JMP/CALL/Jcc is a
|
|
// register-indirect branch; nor do the register-indirect forms of the RISC
|
|
// families (arm64 BR/BLR, riscv64 JALR/JR, loong64 JIRL).
|
|
func positionalBranchTarget(a arch.Arch, upper string) bool {
|
|
switch a {
|
|
case arch.RISCV:
|
|
return riscvBranches[upper] || upper == "JMP" || upper == "JAL"
|
|
case arch.LOONG64:
|
|
return loong64Branches[upper] || upper == "JMP" || upper == "B" ||
|
|
upper == "JAL" || upper == "BL"
|
|
}
|
|
return false
|
|
}
|
|
|
|
// riscvBranches and loong64Branches are the conditional-branch mnemonics; they
|
|
// are listed explicitly rather than matched by a "B" prefix so that bit-manip
|
|
// instructions (BCLR, BSET, …) are never mistaken for branches. The sets
|
|
// mirror the encoder's own branch cases: the B-type table entries
|
|
// (riscv_encode.go), the branch-zero pseudos and the reversed branches
|
|
// BGT/BGTU/BLE/BLEU (riscv_assemble.go), and for loong64 the 16-bit branch
|
|
// table plus the single-register forms of l64branch21Table (BEQZ/BNEZ and the
|
|
// floating-point branches BFPT/BFPF).
|
|
var riscvBranches = map[string]bool{
|
|
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
|
|
"BEQZ": true, "BNEZ": true, "BLEZ": true, "BGEZ": true, "BLTZ": true, "BGTZ": true,
|
|
"BGT": true, "BGTU": true, "BLE": true, "BLEU": true,
|
|
}
|
|
|
|
var loong64Branches = map[string]bool{
|
|
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
|
|
"BLEZ": true, "BLTZ": true, "BGEZ": true, "BGTZ": true,
|
|
"BEQZ": true, "BNEZ": true, "BFPT": true, "BFPF": true,
|
|
}
|
|
|
|
// isJump reports whether the mnemonic is any branch.
|
|
func isJump(a arch.Arch, upper string) bool {
|
|
switch a {
|
|
case arch.ARM64:
|
|
return upper == "CALL" || upper == "BR" || upper == "BLR" || upper == "JMP" ||
|
|
strings.HasPrefix(upper, "B") ||
|
|
strings.HasPrefix(upper, "CBZ") || strings.HasPrefix(upper, "CBNZ") ||
|
|
strings.HasPrefix(upper, "TBZ") || strings.HasPrefix(upper, "TBNZ")
|
|
case arch.RISCV:
|
|
return upper == "CALL" || riscvBranches[upper] ||
|
|
upper == "JMP" || upper == "J" || upper == "JAL" || upper == "JALR" ||
|
|
upper == "JR" || upper == "BR"
|
|
case arch.LOONG64:
|
|
return upper == "CALL" || loong64Branches[upper] ||
|
|
upper == "JIRL" || upper == "JMP" || upper == "BR" ||
|
|
upper == "B" || upper == "JAL" || upper == "BL"
|
|
default: // amd64
|
|
return upper == "CALL" || strings.HasPrefix(upper, "J")
|
|
}
|
|
}
|
|
|
|
// isUnconditionalJump reports whether the mnemonic is an unconditional branch
|
|
// (used to suppress the missing-RET heuristic for tail calls and loops).
|
|
func isUnconditionalJump(a arch.Arch, upper string) bool {
|
|
switch a {
|
|
case arch.ARM64:
|
|
return upper == "B" || upper == "BR" || upper == "JMP"
|
|
case arch.RISCV:
|
|
return upper == "JMP" || upper == "J" || upper == "JAL" ||
|
|
upper == "JALR" || upper == "JR" || upper == "BR"
|
|
case arch.LOONG64:
|
|
return upper == "JMP" || upper == "JIRL" || upper == "BR" || upper == "B" ||
|
|
upper == "JAL" || upper == "BL"
|
|
default:
|
|
return upper == "JMP"
|
|
}
|
|
}
|
|
|
|
// stackDelta computes the net SP change across a function body.
|
|
// It tracks PUSH/POP and SUB/ADD on SP. Returns the net delta (negative
|
|
// means SP decreased, which is the normal direction for stack growth).
|
|
func stackDelta(t *ast.Text, a arch.Arch) int64 {
|
|
var delta int64
|
|
for _, s := range t.Body {
|
|
in, ok := s.(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
upper := strings.ToUpper(in.Mnemonic.Text)
|
|
switch a {
|
|
case arch.AMD64:
|
|
switch upper {
|
|
case "PUSHQ", "PUSHL", "PUSHW":
|
|
delta -= 8
|
|
case "POPQ", "POPL", "POPW":
|
|
delta += 8
|
|
case "SUBQ", "SUBL":
|
|
if len(in.Operands) >= 2 && isSPReg(in.Operands[1], a) {
|
|
if in.Operands[0].Imm.HasVal {
|
|
delta -= in.Operands[0].Imm.Val
|
|
}
|
|
}
|
|
case "ADDQ", "ADDL":
|
|
if len(in.Operands) >= 2 && isSPReg(in.Operands[1], a) {
|
|
if in.Operands[0].Imm.HasVal {
|
|
delta += in.Operands[0].Imm.Val
|
|
}
|
|
}
|
|
}
|
|
case arch.ARM64:
|
|
switch upper {
|
|
case "SUB":
|
|
if len(in.Operands) >= 3 && isSPReg(in.Operands[2], a) {
|
|
if in.Operands[1].Imm.HasVal {
|
|
delta -= in.Operands[1].Imm.Val
|
|
}
|
|
}
|
|
case "ADD":
|
|
if len(in.Operands) >= 3 && isSPReg(in.Operands[2], a) {
|
|
if in.Operands[1].Imm.HasVal {
|
|
delta += in.Operands[1].Imm.Val
|
|
}
|
|
}
|
|
}
|
|
case arch.RISCV:
|
|
switch upper {
|
|
case "ADDI":
|
|
if len(in.Operands) >= 3 && isSPReg(in.Operands[2], a) {
|
|
if in.Operands[1].Imm.HasVal {
|
|
delta += in.Operands[1].Imm.Val
|
|
}
|
|
}
|
|
}
|
|
case arch.LOONG64:
|
|
switch upper {
|
|
case "ADDI.D", "ADDI.W":
|
|
if len(in.Operands) >= 3 && isSPReg(in.Operands[2], a) {
|
|
if in.Operands[1].Imm.HasVal {
|
|
delta += in.Operands[1].Imm.Val
|
|
}
|
|
}
|
|
case "ADD.D", "ADD.W":
|
|
if len(in.Operands) >= 3 && isSPReg(in.Operands[2], a) {
|
|
if in.Operands[1].Imm.HasVal {
|
|
delta += in.Operands[1].Imm.Val
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
return delta
|
|
}
|
|
|
|
// isSPReg reports whether the operand is the stack pointer register.
|
|
func isSPReg(op *ast.Operand, a arch.Arch) bool {
|
|
if op == nil || op.Addr.Sym == nil {
|
|
return false
|
|
}
|
|
name := strings.ToUpper(op.Addr.Sym.Name)
|
|
switch a {
|
|
case arch.AMD64:
|
|
return name == "RSP" || name == "ESP" || name == "SP"
|
|
case arch.ARM64:
|
|
return name == "R31" || name == "SP"
|
|
case arch.RISCV:
|
|
return name == "X2" || name == "SP"
|
|
case arch.LOONG64:
|
|
return name == "R3" || name == "SP"
|
|
}
|
|
return false
|
|
}
|
|
|
|
// shiftRotateBases are the shift and rotate mnemonics without their width
|
|
// suffix. These are the instructions whose encoder path (encodeShift) reads
|
|
// the count from the first operand.
|
|
var shiftRotateBases = map[string]bool{
|
|
"SHL": true, "SHR": true, "SAR": true, "SAL": true,
|
|
"ROL": true, "ROR": true, "RCL": true, "RCR": true,
|
|
}
|
|
|
|
// isShiftCountOperand reports whether operand i of mnem is the shift count.
|
|
// The ISA fixes the shift/rotate count register at CL: the D2/D3 group (and
|
|
// C0/C1 for immediates) encode the count outside the ModRM register field,
|
|
// so the count operand is 8-bit by definition no matter how wide the data is.
|
|
// The count arrives as the first of the two operands; the one-operand form
|
|
// does not exist.
|
|
func isShiftCountOperand(mnem string, i, nops int) bool {
|
|
if nops != 2 || i != 0 {
|
|
return false
|
|
}
|
|
if shiftRotateBases[mnem] {
|
|
return true
|
|
}
|
|
if len(mnem) > 1 {
|
|
switch mnem[len(mnem)-1] {
|
|
case 'Q', 'L', 'W', 'B':
|
|
return shiftRotateBases[mnem[:len(mnem)-1]]
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// isSetcc reports whether the mnemonic is a SETcc: SET plus a condition code.
|
|
// The membership test is the encoder's own SET dispatch, which asm.Encodable
|
|
// mirrors.
|
|
func isSetcc(mnem string) bool {
|
|
return strings.HasPrefix(mnem, "SET") && asm.Encodable(mnem)
|
|
}
|
|
|
|
// checkRegisterWidth detects amd64 register-width mismatches. The naming
|
|
// truth of the Go assembler governs: AX, BX, CX, DX, SI, DI, BP, SP and
|
|
// R8-R15 ARE the 64-bit register names (there are no separate EAX/RAX
|
|
// spellings in go tool asm), and AL-DH are the byte forms. The width comes
|
|
// from the opcode suffix, so an L/W operation over a canonical 64-bit name is
|
|
// the normal, correct spelling, flagging it is pure noise on real kernels.
|
|
// What remains worth flagging: a Q (64-bit) operation over a narrower spelled
|
|
// register (EAX under the gasm alias extension, or a byte form), and byte
|
|
// registers in L/W operations.
|
|
func checkRegisterWidth(mnem string, ops []*ast.Operand) string {
|
|
// A SETcc stores one byte: the destination is an 8-bit register or an
|
|
// 8-bit memory location by definition (0F 90+cc), whichever condition it
|
|
// tests. The trailing letter of spellings like SETPL or SETEQ is part of
|
|
// the condition code, not an operand width, so the whole family is
|
|
// exempt from the suffix logic.
|
|
if isSetcc(mnem) {
|
|
return ""
|
|
}
|
|
// Determine expected width from mnemonic suffix.
|
|
var expected int // 0=unknown, 8/4/2/1=bytes
|
|
switch {
|
|
case strings.HasSuffix(mnem, "Q"):
|
|
expected = 8
|
|
case strings.HasSuffix(mnem, "L"):
|
|
expected = 4
|
|
case strings.HasSuffix(mnem, "W"):
|
|
expected = 2
|
|
case strings.HasSuffix(mnem, "B"):
|
|
expected = 1
|
|
default:
|
|
return "" // no suffix, can't determine width
|
|
}
|
|
for i, op := range ops {
|
|
if op.Kind != ast.OpAddr || op.Addr.Sym == nil {
|
|
continue
|
|
}
|
|
// Only a bare register carries a width to compare: frame and static
|
|
// symbol references (ch+8(FP), foo(SB)) and memory operands are not
|
|
// registers even when their name collides with one.
|
|
if op.Addr.Sym.Pseudo != "" || op.Addr.Base != "" || op.Addr.Index != "" {
|
|
continue
|
|
}
|
|
// The shift/rotate count is exempt: fixed at 8 bits by the ISA.
|
|
if isShiftCountOperand(mnem, i, len(ops)) {
|
|
continue
|
|
}
|
|
name := strings.ToLower(op.Addr.Sym.Name)
|
|
regWidth := amd64RegWidth(name)
|
|
if regWidth == 0 {
|
|
continue // not a register or unknown
|
|
}
|
|
if expected == 8 && regWidth < 8 {
|
|
return fmt.Sprintf("%s uses %d-bit register %s (expected 64-bit)", mnem, regWidth*8, op.Addr.Sym.Name)
|
|
}
|
|
if (expected == 4 || expected == 2) && regWidth == 1 {
|
|
return fmt.Sprintf("%s uses 8-bit register %s (expected a wider register)", mnem, op.Addr.Sym.Name)
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// amd64RegWidth returns the width in bytes of an amd64 register name under
|
|
// the Go assembler's naming model: the canonical word names (AX…SP, R8-R15)
|
|
// are 64-bit, AL-DH are the 8-bit forms, and the R/E-prefixed spellings are
|
|
// the gasm alias extension with their intuitive widths.
|
|
func amd64RegWidth(name string) int {
|
|
switch name {
|
|
case "ax", "bx", "cx", "dx", "si", "di", "bp", "sp",
|
|
"r8", "r9", "r10", "r11", "r12", "r13", "r14", "r15",
|
|
"rax", "rbx", "rcx", "rdx", "rsi", "rdi", "rbp", "rsp":
|
|
return 8
|
|
case "al", "bl", "cl", "dl", "ah", "bh", "ch", "dh":
|
|
return 1
|
|
case "eax", "ebx", "ecx", "edx", "esi", "edi", "ebp", "esp":
|
|
return 4
|
|
}
|
|
return 0
|
|
}
|
|
|
|
func countRange(min, max int) string {
|
|
if min == max {
|
|
return fmt.Sprintf("%d operand(s)", min)
|
|
}
|
|
return fmt.Sprintf("%d-%d operands", min, max)
|
|
}
|
|
|
|
// sortDiagnostics orders diagnostics by line, then column, then code.
|
|
func sortDiagnostics(d []Diagnostic) {
|
|
for i := 1; i < len(d); i++ {
|
|
for j := i; j > 0 && lessDiag(d[j], d[j-1]); j-- {
|
|
d[j], d[j-1] = d[j-1], d[j]
|
|
}
|
|
}
|
|
}
|
|
|
|
func lessDiag(a, b Diagnostic) bool {
|
|
if a.Pos.Line != b.Pos.Line {
|
|
return a.Pos.Line < b.Pos.Line
|
|
}
|
|
if a.Pos.Column != b.Pos.Column {
|
|
return a.Pos.Column < b.Pos.Column
|
|
}
|
|
return a.Code < b.Code
|
|
}
|