feat: gasm-devkit 0.1.0 — GAsm lexer, parser, linter, formatter, LSP and amd64 assembler
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
@@ -0,0 +1,619 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package parser turns a GAsm token stream into an abstract syntax tree. It
|
||||
// is line-oriented, matching how the Plan 9 assembler itself reads a file, and
|
||||
// tolerant: a malformed line is reported as an error but never aborts the
|
||||
// parse of the rest of the file.
|
||||
package parser
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// Error is a single parse diagnostic.
|
||||
type Error struct {
|
||||
Pos token.Position
|
||||
Msg string
|
||||
}
|
||||
|
||||
func (e Error) Error() string {
|
||||
return e.Pos.String() + ": " + e.Msg
|
||||
}
|
||||
|
||||
// Parse scans and parses src, returning the file and any diagnostics. The
|
||||
// returned file is usable even when errors is non-empty.
|
||||
func Parse(path, src string) (*ast.File, []error) {
|
||||
toks := lexer.Tokenize(src)
|
||||
lines := splitLines(toks)
|
||||
p := &state{path: path}
|
||||
p.parse(lines)
|
||||
return p.file, p.errs
|
||||
}
|
||||
|
||||
// state carries the mutable context for one parse.
|
||||
type state struct {
|
||||
path string
|
||||
file *ast.File
|
||||
errs []error
|
||||
|
||||
curText *ast.Text // the TEXT body labels/instructions attach to
|
||||
pending []string // comment lines awaiting a TEXT to become its Doc
|
||||
}
|
||||
|
||||
func (p *state) errorf(pos token.Position, format string, args ...any) {
|
||||
p.errs = append(p.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)})
|
||||
}
|
||||
|
||||
// splitLines groups the token stream into lines, dropping the Newline tokens.
|
||||
func splitLines(toks []token.Token) [][]token.Token {
|
||||
var lines [][]token.Token
|
||||
var cur []token.Token
|
||||
for _, t := range toks {
|
||||
if t.Kind == token.EOF {
|
||||
break
|
||||
}
|
||||
if t.Kind == token.Newline {
|
||||
lines = append(lines, cur)
|
||||
cur = nil
|
||||
continue
|
||||
}
|
||||
cur = append(cur, t)
|
||||
}
|
||||
if len(cur) > 0 {
|
||||
lines = append(lines, cur)
|
||||
}
|
||||
return lines
|
||||
}
|
||||
|
||||
func (p *state) parse(lines [][]token.Token) {
|
||||
p.file = &ast.File{Path: p.path, Macros: map[string]bool{}}
|
||||
for _, line := range lines {
|
||||
line = trimSpace(line)
|
||||
if len(line) == 0 {
|
||||
// Blank line: a comment block ends here only if it was not
|
||||
// directly preceding a declaration; keep pending doc intact
|
||||
// across a single blank line is not desired, so reset.
|
||||
p.pending = nil
|
||||
continue
|
||||
}
|
||||
p.parseLine(line)
|
||||
}
|
||||
}
|
||||
|
||||
// trimSpace is a no-op placeholder kept for symmetry; the lexer already drops
|
||||
// horizontal whitespace, but this documents the intent.
|
||||
func trimSpace(line []token.Token) []token.Token { return line }
|
||||
|
||||
func (p *state) parseLine(line []token.Token) {
|
||||
first := line[0]
|
||||
|
||||
// A lone comment accumulates as documentation for a following TEXT.
|
||||
if len(line) == 1 && first.Kind == token.Comment {
|
||||
p.pending = append(p.pending, commentText(first.Text))
|
||||
return
|
||||
}
|
||||
|
||||
// Preprocessor line.
|
||||
if first.Kind == token.Hash {
|
||||
p.parsePreproc(line)
|
||||
p.pending = nil
|
||||
return
|
||||
}
|
||||
|
||||
// Directives and instructions are identified by a leading identifier.
|
||||
if first.Kind == token.Ident {
|
||||
switch first.Text {
|
||||
case "TEXT":
|
||||
p.parseText(line)
|
||||
return
|
||||
case "GLOBL":
|
||||
p.file.Decls = append(p.file.Decls, p.parseGlobl(line))
|
||||
p.curText = nil
|
||||
p.pending = nil
|
||||
return
|
||||
case "DATA":
|
||||
p.file.Decls = append(p.file.Decls, p.parseData(line))
|
||||
p.curText = nil
|
||||
p.pending = nil
|
||||
return
|
||||
}
|
||||
|
||||
// Label (ident immediately followed by a colon).
|
||||
if len(line) >= 2 && line[1].Kind == token.Colon {
|
||||
lbl := &ast.Label{Name: line[0], Colon: line[1]}
|
||||
p.addStmt(lbl)
|
||||
// A label may share its line with an instruction: "loop: MOVQ …".
|
||||
if rest := dropColon(line); len(rest) > 0 {
|
||||
p.parseInstr(rest)
|
||||
}
|
||||
p.pending = nil
|
||||
return
|
||||
}
|
||||
|
||||
// Otherwise it is an instruction.
|
||||
p.parseInstr(line)
|
||||
p.pending = nil
|
||||
return
|
||||
}
|
||||
|
||||
p.errorf(first.Pos, "unexpected token %s at start of line", first.Kind)
|
||||
p.pending = nil
|
||||
}
|
||||
|
||||
// dropColon removes the leading "ident :" of a label, returning the remainder.
|
||||
func dropColon(line []token.Token) []token.Token {
|
||||
if len(line) >= 2 && line[1].Kind == token.Colon {
|
||||
return line[2:]
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (p *state) addStmt(s ast.Stmt) {
|
||||
if p.curText != nil {
|
||||
p.curText.Body = append(p.curText.Body, s)
|
||||
return
|
||||
}
|
||||
p.file.Orphans = append(p.file.Orphans, s)
|
||||
}
|
||||
|
||||
func (p *state) parsePreproc(line []token.Token) {
|
||||
hash := line[0]
|
||||
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
|
||||
line[2].Kind == token.String {
|
||||
p.file.Decls = append(p.file.Decls, &ast.Include{
|
||||
Hash: hash,
|
||||
Name: line[1],
|
||||
Header: line[2],
|
||||
})
|
||||
return
|
||||
}
|
||||
// Record macro names so the linter can recognise their invocations.
|
||||
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "define" &&
|
||||
line[2].Kind == token.Ident {
|
||||
p.file.Macros[line[2].Text] = true
|
||||
}
|
||||
p.file.Decls = append(p.file.Decls, &ast.Preproc{
|
||||
Hash: hash,
|
||||
Raw: joinRaw(line[1:]),
|
||||
})
|
||||
}
|
||||
|
||||
func (p *state) parseText(line []token.Token) {
|
||||
text := &ast.Text{Keyword: line[0]}
|
||||
if len(p.pending) > 0 {
|
||||
text.Doc = strings.Join(p.pending, "\n")
|
||||
}
|
||||
p.pending = nil
|
||||
|
||||
rest := line[1:]
|
||||
sym, n := parseSymbolPrefix(rest)
|
||||
if sym == nil {
|
||||
p.errorf(line[0].Pos, "TEXT missing a symbol name")
|
||||
}
|
||||
text.Name = sym
|
||||
rest = rest[n:]
|
||||
|
||||
// Consume flags (identifiers, possibly '|' joined) up to the frame '$'.
|
||||
rest = skipComma(rest)
|
||||
for len(rest) > 0 && rest[0].Kind != token.Dollar {
|
||||
if rest[0].Kind == token.Ident {
|
||||
text.Flags = append(text.Flags, rest[0].Text)
|
||||
}
|
||||
// Commas, '|' (Illegal) and anything else between flags is skipped.
|
||||
rest = rest[1:]
|
||||
}
|
||||
|
||||
// Frame: $number ; optional args: -number.
|
||||
if len(rest) > 0 && rest[0].Kind == token.Dollar {
|
||||
text.Frame = parseOperand(rest[:2]) // "$" "number"
|
||||
rest = rest[2:]
|
||||
if len(rest) >= 2 && rest[0].Kind == token.Minus && rest[1].Kind == token.Number {
|
||||
text.Args = &ast.Operand{
|
||||
Kind: ast.OpImmediate,
|
||||
Imm: ast.Immediate{Val: parseInt(rest[1].Text), HasVal: true},
|
||||
Raw: "-" + rest[1].Text,
|
||||
Pos: rest[0].Pos,
|
||||
}
|
||||
rest = rest[2:]
|
||||
}
|
||||
}
|
||||
|
||||
p.file.Decls = append(p.file.Decls, text)
|
||||
p.curText = text
|
||||
}
|
||||
|
||||
func (p *state) parseGlobl(line []token.Token) *ast.Globl {
|
||||
g := &ast.Globl{Keyword: line[0]}
|
||||
rest := skipComma(line[1:])
|
||||
sym, n := parseSymbolPrefix(rest)
|
||||
g.Name = sym
|
||||
rest = skipComma(rest[n:])
|
||||
for len(rest) > 0 && rest[0].Kind != token.Dollar {
|
||||
if rest[0].Kind == token.Ident {
|
||||
g.Flags = append(g.Flags, rest[0].Text)
|
||||
}
|
||||
rest = rest[1:]
|
||||
}
|
||||
if len(rest) > 0 && rest[0].Kind == token.Dollar {
|
||||
g.Size = parseOperand(rest)
|
||||
}
|
||||
return g
|
||||
}
|
||||
|
||||
func (p *state) parseData(line []token.Token) *ast.Data {
|
||||
d := &ast.Data{Keyword: line[0]}
|
||||
rest := line[1:]
|
||||
|
||||
// The name may carry a /width suffix: ·idx+0(SB)/4. Split it off the
|
||||
// symbol group that precedes the first top-level comma.
|
||||
nameGroup, valuePart := splitFirstComma(rest)
|
||||
nameGroup, width := splitTrailingWidth(nameGroup)
|
||||
sym, _ := parseSymbolPrefix(nameGroup)
|
||||
d.Name = sym
|
||||
d.Width = width
|
||||
if len(valuePart) > 0 {
|
||||
d.Value = parseOperand(stripComment(valuePart))
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
func (p *state) parseInstr(line []token.Token) {
|
||||
body, comment := splitTrailingComment(line)
|
||||
if len(body) == 0 {
|
||||
return
|
||||
}
|
||||
instr := &ast.Instr{Mnemonic: body[0], Comment: comment}
|
||||
for _, grp := range splitOperands(body[1:]) {
|
||||
if op := parseOperand(grp); op != nil {
|
||||
instr.Operands = append(instr.Operands, op)
|
||||
}
|
||||
}
|
||||
p.addStmt(instr)
|
||||
}
|
||||
|
||||
// --- symbol parsing ---------------------------------------------------------
|
||||
|
||||
var pseudoRegs = map[string]bool{"FP": true, "SP": true, "SB": true, "PC": true}
|
||||
|
||||
// parseSymbolPrefix parses a leading symbol reference from g and returns it
|
||||
// together with the number of tokens consumed. It returns (nil, 0) when no
|
||||
// symbol is present.
|
||||
func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) {
|
||||
if len(g) == 0 || g[0].Kind != token.Ident {
|
||||
return nil, 0
|
||||
}
|
||||
sym := &ast.Symbol{Pos: g[0].Pos}
|
||||
i := 0
|
||||
setName(g[0].Text, sym)
|
||||
i++
|
||||
|
||||
if i+1 < len(g) && g[i].Kind == token.LAngle && g[i+1].Kind == token.RAngle {
|
||||
sym.Static = true
|
||||
i += 2
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Plus {
|
||||
i++
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
sym.Offset, sym.HasOff = parseInt(g[i].Text), true
|
||||
i++
|
||||
}
|
||||
}
|
||||
if i+2 < len(g) && g[i].Kind == token.LParen && g[i+1].Kind == token.Ident &&
|
||||
pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen {
|
||||
sym.Pseudo = g[i+1].Text
|
||||
i += 3
|
||||
}
|
||||
sym.Raw = joinRaw(g[:i])
|
||||
return sym, i
|
||||
}
|
||||
|
||||
// setName splits a raw identifier on the middle dot into package and name.
|
||||
func setName(raw string, sym *ast.Symbol) {
|
||||
const dot = "\u00B7"
|
||||
switch {
|
||||
case strings.HasPrefix(raw, dot):
|
||||
sym.Pkg = ""
|
||||
sym.Name = strings.TrimPrefix(raw, dot)
|
||||
case strings.Contains(raw, dot):
|
||||
parts := strings.SplitN(raw, dot, 2)
|
||||
sym.Pkg = parts[0]
|
||||
sym.Name = parts[1]
|
||||
default:
|
||||
sym.Name = raw
|
||||
}
|
||||
}
|
||||
|
||||
// --- operand parsing --------------------------------------------------------
|
||||
|
||||
// parseOperand parses one operand group into an Operand.
|
||||
func parseOperand(g []token.Token) *ast.Operand {
|
||||
g = stripComment(g)
|
||||
if len(g) == 0 {
|
||||
return nil
|
||||
}
|
||||
op := &ast.Operand{Raw: joinRaw(g), Pos: g[0].Pos}
|
||||
if g[0].Kind == token.Dollar {
|
||||
op.Kind = ast.OpImmediate
|
||||
op.Imm = parseImmediate(g[1:])
|
||||
return op
|
||||
}
|
||||
op.Kind = ast.OpAddr
|
||||
op.Addr = parseAddress(g)
|
||||
return op
|
||||
}
|
||||
|
||||
// parseImmediate parses the tokens following a '$'.
|
||||
func parseImmediate(g []token.Token) ast.Immediate {
|
||||
var imm ast.Immediate
|
||||
if len(g) == 0 {
|
||||
return imm
|
||||
}
|
||||
// $sym(…) form.
|
||||
if findPseudoParen(g) >= 0 || (g[0].Kind == token.Ident) {
|
||||
if sym, n := parseSymbolPrefix(g); sym != nil && (sym.Pseudo != "" || sym.Static) {
|
||||
imm.Sym = sym
|
||||
_ = n
|
||||
return imm
|
||||
}
|
||||
}
|
||||
i := 0
|
||||
if g[i].Kind == token.Minus {
|
||||
imm.Neg = true
|
||||
i++
|
||||
} else if g[i].Kind == token.Plus {
|
||||
i++
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
text := g[i].Text
|
||||
if v, ok := tryInt(text); ok {
|
||||
imm.Val = v
|
||||
imm.HasVal = true
|
||||
} else {
|
||||
imm.Float = text
|
||||
}
|
||||
i++
|
||||
} else if i < len(g) && (g[i].Kind == token.String || g[i].Kind == token.Rune) {
|
||||
imm.Str = g[i].Text
|
||||
i++
|
||||
}
|
||||
return imm
|
||||
}
|
||||
|
||||
// parseAddress parses a non-immediate operand.
|
||||
func parseAddress(g []token.Token) ast.Address {
|
||||
var addr ast.Address
|
||||
if len(g) == 0 {
|
||||
return addr
|
||||
}
|
||||
// Symbol-with-pseudo form: name[<>][+off](PSEUDO).
|
||||
if idx := findPseudoParen(g); idx >= 0 {
|
||||
sym, _ := parseSymbolPrefix(g[:idx+3])
|
||||
addr.Sym = sym
|
||||
return addr
|
||||
}
|
||||
|
||||
i := 0
|
||||
// Optional leading displacement before a '(' base group.
|
||||
if isSignedNumber(g, i) && i+1 < len(g) && g[i+1].Kind == token.LParen {
|
||||
neg := false
|
||||
if g[i].Kind == token.Minus {
|
||||
neg = true
|
||||
i++
|
||||
} else if g[i].Kind == token.Plus {
|
||||
i++
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
addr.Offset = parseInt(g[i].Text)
|
||||
addr.HasOff = true
|
||||
if neg {
|
||||
addr.Offset = -addr.Offset
|
||||
}
|
||||
i++
|
||||
}
|
||||
}
|
||||
// First parenthesised group: the base register.
|
||||
if i < len(g) && g[i].Kind == token.LParen {
|
||||
i++
|
||||
if i < len(g) && g[i].Kind == token.Ident {
|
||||
addr.Base = g[i].Text
|
||||
i++
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.RParen {
|
||||
i++
|
||||
}
|
||||
}
|
||||
// Optional second group: (index*scale) or (index).
|
||||
if i < len(g) && g[i].Kind == token.LParen {
|
||||
i++
|
||||
if i < len(g) && g[i].Kind == token.Ident {
|
||||
addr.Index = g[i].Text
|
||||
i++
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.Star {
|
||||
i++
|
||||
if i < len(g) && g[i].Kind == token.Number {
|
||||
addr.Scale = int(parseInt(g[i].Text))
|
||||
i++
|
||||
}
|
||||
}
|
||||
if i < len(g) && g[i].Kind == token.RParen {
|
||||
i++
|
||||
}
|
||||
}
|
||||
// Bare name (register, label or symbol) possibly with an arm64 shift.
|
||||
if addr.Base == "" && addr.Sym == nil && g[0].Kind == token.Ident {
|
||||
sym := &ast.Symbol{Pos: g[0].Pos}
|
||||
setName(g[0].Text, sym)
|
||||
sym.Raw = g[0].Text
|
||||
addr.Sym = sym
|
||||
i = 1
|
||||
}
|
||||
// Any remaining tokens form a verbatim shift/extension suffix (arm64).
|
||||
if i > 0 && i < len(g) {
|
||||
addr.Shift = joinRaw(g[i:])
|
||||
}
|
||||
return addr
|
||||
}
|
||||
|
||||
// findPseudoParen returns the index of the '(' that begins a (PSEUDO) group,
|
||||
// or -1 when none is present.
|
||||
func findPseudoParen(g []token.Token) int {
|
||||
for i := 0; i+2 < len(g); i++ {
|
||||
if g[i].Kind == token.LParen && g[i+1].Kind == token.Ident &&
|
||||
pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// --- token helpers ----------------------------------------------------------
|
||||
|
||||
// splitOperands splits a token slice on top-level commas (commas outside any
|
||||
// parenthesis group).
|
||||
func splitOperands(g []token.Token) [][]token.Token {
|
||||
var out [][]token.Token
|
||||
var cur []token.Token
|
||||
depth := 0
|
||||
for _, t := range g {
|
||||
switch t.Kind {
|
||||
case token.LParen:
|
||||
depth++
|
||||
cur = append(cur, t)
|
||||
case token.RParen:
|
||||
depth--
|
||||
cur = append(cur, t)
|
||||
case token.Comma:
|
||||
if depth == 0 {
|
||||
if len(cur) > 0 {
|
||||
out = append(out, cur)
|
||||
}
|
||||
cur = nil
|
||||
} else {
|
||||
cur = append(cur, t)
|
||||
}
|
||||
case token.Comment:
|
||||
// A comment terminates the operand list.
|
||||
if len(cur) > 0 {
|
||||
out = append(out, cur)
|
||||
}
|
||||
return out
|
||||
default:
|
||||
cur = append(cur, t)
|
||||
}
|
||||
}
|
||||
if len(cur) > 0 {
|
||||
out = append(out, cur)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// splitFirstComma splits g at the first top-level comma.
|
||||
func splitFirstComma(g []token.Token) (before, after []token.Token) {
|
||||
depth := 0
|
||||
for i, t := range g {
|
||||
switch t.Kind {
|
||||
case token.LParen:
|
||||
depth++
|
||||
case token.RParen:
|
||||
depth--
|
||||
case token.Comma:
|
||||
if depth == 0 {
|
||||
return g[:i], g[i+1:]
|
||||
}
|
||||
}
|
||||
}
|
||||
return g, nil
|
||||
}
|
||||
|
||||
// splitTrailingComment separates a trailing comment from the line body.
|
||||
func splitTrailingComment(g []token.Token) (body []token.Token, comment string) {
|
||||
for i, t := range g {
|
||||
if t.Kind == token.Comment {
|
||||
return g[:i], commentText(t.Text)
|
||||
}
|
||||
}
|
||||
return g, ""
|
||||
}
|
||||
|
||||
// stripComment removes a trailing comment token from a group.
|
||||
func stripComment(g []token.Token) []token.Token {
|
||||
for i, t := range g {
|
||||
if t.Kind == token.Comment {
|
||||
return g[:i]
|
||||
}
|
||||
}
|
||||
return g
|
||||
}
|
||||
|
||||
// splitTrailingWidth removes a "/width" suffix from a DATA name group.
|
||||
func splitTrailingWidth(g []token.Token) ([]token.Token, int) {
|
||||
for i := 0; i+1 < len(g); i++ {
|
||||
if g[i].Kind == token.Slash && g[i+1].Kind == token.Number {
|
||||
return g[:i], int(parseInt(g[i+1].Text))
|
||||
}
|
||||
}
|
||||
return g, 0
|
||||
}
|
||||
|
||||
func skipComma(g []token.Token) []token.Token {
|
||||
if len(g) > 0 && g[0].Kind == token.Comma {
|
||||
return g[1:]
|
||||
}
|
||||
return g
|
||||
}
|
||||
|
||||
func isSignedNumber(g []token.Token, i int) bool {
|
||||
if i >= len(g) {
|
||||
return false
|
||||
}
|
||||
if g[i].Kind == token.Number {
|
||||
return true
|
||||
}
|
||||
if (g[i].Kind == token.Minus || g[i].Kind == token.Plus) &&
|
||||
i+1 < len(g) && g[i+1].Kind == token.Number {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func joinRaw(g []token.Token) string {
|
||||
parts := make([]string, len(g))
|
||||
for i, t := range g {
|
||||
parts[i] = t.Text
|
||||
}
|
||||
return strings.Join(parts, " ")
|
||||
}
|
||||
|
||||
// commentText removes a leading // or /* marker from a comment token's text.
|
||||
func commentText(s string) string {
|
||||
if strings.HasPrefix(s, "//") {
|
||||
return strings.TrimSpace(strings.TrimPrefix(s, "//"))
|
||||
}
|
||||
if strings.HasPrefix(s, "/*") {
|
||||
s = strings.TrimPrefix(s, "/*")
|
||||
s = strings.TrimSuffix(s, "*/")
|
||||
return strings.TrimSpace(s)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func parseInt(text string) int64 {
|
||||
v, _ := tryInt(text)
|
||||
return v
|
||||
}
|
||||
|
||||
func tryInt(text string) (int64, bool) {
|
||||
v, err := strconv.ParseInt(text, 0, 64)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
return v, true
|
||||
}
|
||||
Reference in New Issue
Block a user