feat: gasm-devkit 0.1.0 — GAsm lexer, parser, linter, formatter, LSP and amd64 assembler
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
+395
@@ -0,0 +1,395 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package lexer implements a hand-written scanner for Go's Plan 9 assembler
|
||||
// (GAsm). It turns a source string into a flat token stream that the parser,
|
||||
// formatter and language server all build on. The scanner is deliberately
|
||||
// permissive: it never panics and maps anything it cannot classify to an
|
||||
// Illegal token so that downstream tools can still operate on malformed input.
|
||||
package lexer
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"unicode"
|
||||
"unicode/utf8"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// middleDot is the Plan 9 symbol separator (U+00B7), used in ·funcName(SB).
|
||||
const middleDot = '\u00B7'
|
||||
|
||||
// Lexer scans a source string one token at a time.
|
||||
type Lexer struct {
|
||||
src []rune
|
||||
off []int // off[i] is the byte offset of src[i]; off[len(src)] is len(bytes)
|
||||
i int // index of the current rune
|
||||
line int // one-based line of src[i]
|
||||
col int // one-based rune column of src[i]
|
||||
}
|
||||
|
||||
// New returns a Lexer over src.
|
||||
func New(src string) *Lexer {
|
||||
runes := []rune(src)
|
||||
off := make([]int, len(runes)+1)
|
||||
b := 0
|
||||
for i, r := range runes {
|
||||
off[i] = b
|
||||
b += utf8.RuneLen(r)
|
||||
}
|
||||
off[len(runes)] = b
|
||||
return &Lexer{src: runes, off: off, line: 1, col: 1}
|
||||
}
|
||||
|
||||
// Tokenize scans src fully and returns every token up to and including the
|
||||
// trailing EOF token.
|
||||
func Tokenize(src string) []token.Token {
|
||||
l := New(src)
|
||||
var out []token.Token
|
||||
for {
|
||||
tok := l.Next()
|
||||
out = append(out, tok)
|
||||
if tok.Kind == token.EOF {
|
||||
return out
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// cur returns the current rune, or 0 at end of input.
|
||||
func (l *Lexer) cur() rune {
|
||||
if l.i >= len(l.src) {
|
||||
return 0
|
||||
}
|
||||
return l.src[l.i]
|
||||
}
|
||||
|
||||
// peek returns the rune k positions ahead, or 0 past the end.
|
||||
func (l *Lexer) peek(k int) rune {
|
||||
if l.i+k >= len(l.src) || l.i+k < 0 {
|
||||
return 0
|
||||
}
|
||||
return l.src[l.i+k]
|
||||
}
|
||||
|
||||
// pos snapshots the current source position.
|
||||
func (l *Lexer) pos() token.Position {
|
||||
return token.Position{Offset: l.off[l.i], Line: l.line, Column: l.col}
|
||||
}
|
||||
|
||||
// advance consumes one rune, updating line and column bookkeeping.
|
||||
func (l *Lexer) advance() {
|
||||
if l.i >= len(l.src) {
|
||||
return
|
||||
}
|
||||
if l.src[l.i] == '\n' {
|
||||
l.line++
|
||||
l.col = 1
|
||||
} else {
|
||||
l.col++
|
||||
}
|
||||
l.i++
|
||||
}
|
||||
|
||||
// make builds a token of the given kind spanning [start, current position).
|
||||
func (l *Lexer) make(kind token.Kind, start token.Position, text string) token.Token {
|
||||
return token.Token{Kind: kind, Text: text, Pos: start, End: l.pos()}
|
||||
}
|
||||
|
||||
// Next returns the next token, skipping spaces and tabs. Newlines are
|
||||
// significant and returned as Newline tokens so the parser can treat the
|
||||
// stream line by line.
|
||||
func (l *Lexer) Next() token.Token {
|
||||
for {
|
||||
// Skip horizontal whitespace. A backslash immediately before a newline
|
||||
// is a C-preprocessor line continuation (used by #define macros in the
|
||||
// runtime .s files): splice the lines together by consuming both, so
|
||||
// the whole macro becomes one logical line that the parser treats as an
|
||||
// opaque preprocessor directive.
|
||||
for {
|
||||
c := l.cur()
|
||||
if c == ' ' || c == '\t' || c == '\r' {
|
||||
l.advance()
|
||||
continue
|
||||
}
|
||||
if c == '\\' && (l.peek(1) == '\n' || l.peek(1) == '\r') {
|
||||
l.advance() // backslash
|
||||
if l.cur() == '\r' {
|
||||
l.advance()
|
||||
}
|
||||
if l.cur() == '\n' {
|
||||
l.advance()
|
||||
}
|
||||
continue
|
||||
}
|
||||
break
|
||||
}
|
||||
|
||||
start := l.pos()
|
||||
r := l.cur()
|
||||
|
||||
switch {
|
||||
case r == 0:
|
||||
return l.make(token.EOF, start, "")
|
||||
|
||||
case r == '\n':
|
||||
l.advance()
|
||||
return l.make(token.Newline, start, "\n")
|
||||
|
||||
case r == '/':
|
||||
switch l.peek(1) {
|
||||
case '/':
|
||||
return l.lineComment(start)
|
||||
case '*':
|
||||
return l.blockComment(start)
|
||||
default:
|
||||
l.advance()
|
||||
return l.make(token.Slash, start, "/")
|
||||
}
|
||||
|
||||
case r == '"':
|
||||
return l.string(start)
|
||||
|
||||
case r == '\'':
|
||||
return l.runeLit(start)
|
||||
|
||||
case isIdentStart(r):
|
||||
return l.ident(start)
|
||||
|
||||
case isDigit(r):
|
||||
return l.number(start)
|
||||
|
||||
default:
|
||||
return l.punct(start)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// lineComment consumes a // comment up to, but not including, the newline.
|
||||
func (l *Lexer) lineComment(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
for l.cur() != 0 && l.cur() != '\n' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Comment, start, b.String())
|
||||
}
|
||||
|
||||
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
|
||||
func (l *Lexer) blockComment(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
b.WriteRune(l.cur()) // '/'
|
||||
l.advance()
|
||||
b.WriteRune(l.cur()) // '*'
|
||||
l.advance()
|
||||
for l.cur() != 0 {
|
||||
if l.cur() == '*' && l.peek(1) == '/' {
|
||||
b.WriteString("*/")
|
||||
l.advance()
|
||||
l.advance()
|
||||
break
|
||||
}
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Comment, start, b.String())
|
||||
}
|
||||
|
||||
// string consumes a double-quoted string literal, honouring backslash escapes.
|
||||
func (l *Lexer) string(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
b.WriteRune('"')
|
||||
l.advance() // opening quote
|
||||
for l.cur() != 0 && l.cur() != '\n' {
|
||||
r := l.cur()
|
||||
b.WriteRune(r)
|
||||
l.advance()
|
||||
if r == '\\' {
|
||||
if l.cur() != 0 && l.cur() != '\n' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
continue
|
||||
}
|
||||
if r == '"' {
|
||||
return l.make(token.String, start, b.String())
|
||||
}
|
||||
}
|
||||
// Unterminated string: return what we have rather than failing.
|
||||
return l.make(token.String, start, b.String())
|
||||
}
|
||||
|
||||
// runeLit consumes a single-quoted rune literal such as 'a' or '\n'.
|
||||
func (l *Lexer) runeLit(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
b.WriteRune('\'')
|
||||
l.advance() // opening quote
|
||||
for l.cur() != 0 && l.cur() != '\n' {
|
||||
r := l.cur()
|
||||
b.WriteRune(r)
|
||||
l.advance()
|
||||
if r == '\\' {
|
||||
if l.cur() != 0 && l.cur() != '\n' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
continue
|
||||
}
|
||||
if r == '\'' {
|
||||
return l.make(token.Rune, start, b.String())
|
||||
}
|
||||
}
|
||||
return l.make(token.Rune, start, b.String())
|
||||
}
|
||||
|
||||
// ident consumes an identifier: letters, digits, '_', '.', and the middle dot.
|
||||
func (l *Lexer) ident(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
for isIdentChar(l.cur()) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Ident, start, b.String())
|
||||
}
|
||||
|
||||
// number consumes an integer or floating-point literal. The sign is never
|
||||
// part of the literal; it is scanned separately as a Minus or Plus token.
|
||||
func (l *Lexer) number(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
// Base prefixes.
|
||||
if l.cur() == '0' && (l.peek(1) == 'x' || l.peek(1) == 'X') {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
for isHexDigit(l.cur()) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Number, start, b.String())
|
||||
}
|
||||
if l.cur() == '0' && (l.peek(1) == 'b' || l.peek(1) == 'B') {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
for l.cur() == '0' || l.cur() == '1' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Number, start, b.String())
|
||||
}
|
||||
if l.cur() == '0' && (l.peek(1) == 'o' || l.peek(1) == 'O') {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
for l.cur() >= '0' && l.cur() <= '7' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Number, start, b.String())
|
||||
}
|
||||
// Decimal, possibly fractional and/or with an exponent.
|
||||
for isDigit(l.cur()) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
if l.cur() == '.' && isDigit(l.peek(1)) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
for isDigit(l.cur()) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
}
|
||||
if l.cur() == 'e' || l.cur() == 'E' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
if l.cur() == '+' || l.cur() == '-' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
for isDigit(l.cur()) {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
}
|
||||
return l.make(token.Number, start, b.String())
|
||||
}
|
||||
|
||||
// punct consumes a single punctuation or operator token, handling the
|
||||
// multi-character operators <<, >> and ->.
|
||||
func (l *Lexer) punct(start token.Position) token.Token {
|
||||
r := l.cur()
|
||||
switch r {
|
||||
case '(':
|
||||
l.advance()
|
||||
return l.make(token.LParen, start, "(")
|
||||
case ')':
|
||||
l.advance()
|
||||
return l.make(token.RParen, start, ")")
|
||||
case ',':
|
||||
l.advance()
|
||||
return l.make(token.Comma, start, ",")
|
||||
case '+':
|
||||
l.advance()
|
||||
return l.make(token.Plus, start, "+")
|
||||
case '-':
|
||||
if l.peek(1) == '>' {
|
||||
l.advance()
|
||||
l.advance()
|
||||
return l.make(token.Arrow, start, "->")
|
||||
}
|
||||
l.advance()
|
||||
return l.make(token.Minus, start, "-")
|
||||
case '*':
|
||||
l.advance()
|
||||
return l.make(token.Star, start, "*")
|
||||
case ':':
|
||||
l.advance()
|
||||
return l.make(token.Colon, start, ":")
|
||||
case '$':
|
||||
l.advance()
|
||||
return l.make(token.Dollar, start, "$")
|
||||
case '<':
|
||||
if l.peek(1) == '<' {
|
||||
l.advance()
|
||||
l.advance()
|
||||
return l.make(token.LShift, start, "<<")
|
||||
}
|
||||
l.advance()
|
||||
return l.make(token.LAngle, start, "<")
|
||||
case '>':
|
||||
if l.peek(1) == '>' {
|
||||
l.advance()
|
||||
l.advance()
|
||||
return l.make(token.RShift, start, ">>")
|
||||
}
|
||||
l.advance()
|
||||
return l.make(token.RAngle, start, ">")
|
||||
case '@':
|
||||
l.advance()
|
||||
return l.make(token.At, start, "@")
|
||||
case '#':
|
||||
l.advance()
|
||||
return l.make(token.Hash, start, "#")
|
||||
default:
|
||||
// Unknown rune: emit it as Illegal and move on.
|
||||
l.advance()
|
||||
return l.make(token.Illegal, start, string(r))
|
||||
}
|
||||
}
|
||||
|
||||
func isDigit(r rune) bool { return r >= '0' && r <= '9' }
|
||||
|
||||
func isHexDigit(r rune) bool {
|
||||
return isDigit(r) || (r >= 'a' && r <= 'f') || (r >= 'A' && r <= 'F')
|
||||
}
|
||||
|
||||
func isIdentStart(r rune) bool {
|
||||
return r == '_' || r == middleDot || unicode.IsLetter(r)
|
||||
}
|
||||
|
||||
func isIdentChar(r rune) bool {
|
||||
return isIdentStart(r) || isDigit(r) || r == '.'
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package lexer
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// kinds tokenizes src and returns the kind sequence, dropping Newline/EOF.
|
||||
func kinds(src string) []token.Kind {
|
||||
var out []token.Kind
|
||||
for _, t := range Tokenize(src) {
|
||||
if t.Kind == token.Newline || t.Kind == token.EOF {
|
||||
continue
|
||||
}
|
||||
out = append(out, t.Kind)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// texts tokenizes src and returns the literal text of each significant token.
|
||||
func texts(src string) []string {
|
||||
var out []string
|
||||
for _, t := range Tokenize(src) {
|
||||
if t.Kind == token.Newline || t.Kind == token.EOF {
|
||||
continue
|
||||
}
|
||||
out = append(out, t.Text)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func eq[T comparable](t *testing.T, got, want []T) {
|
||||
t.Helper()
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("length mismatch:\n got %v\n want %v", got, want)
|
||||
}
|
||||
for i := range got {
|
||||
if got[i] != want[i] {
|
||||
t.Fatalf("index %d:\n got %v\n want %v", i, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextDirective(t *testing.T) {
|
||||
eq(t, texts("TEXT ·analyzeO1RangeAVX2(SB), NOSPLIT, $0-65"),
|
||||
[]string{"TEXT", "·analyzeO1RangeAVX2", "(", "SB", ")", ",", "NOSPLIT", ",", "$", "0", "-", "65"})
|
||||
}
|
||||
|
||||
func TestDataAndGlobl(t *testing.T) {
|
||||
eq(t, texts("GLOBL ·idx16(SB), RODATA, $64"),
|
||||
[]string{"GLOBL", "·idx16", "(", "SB", ")", ",", "RODATA", ",", "$", "64"})
|
||||
eq(t, texts("DATA ·idx16+0(SB)/4, $1"),
|
||||
[]string{"DATA", "·idx16", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "1"})
|
||||
}
|
||||
|
||||
func TestStaticSymbol(t *testing.T) {
|
||||
// mask24<> is a file-local symbol; <> must lex as two angle tokens.
|
||||
eq(t, texts("GLOBL mask24<>(SB), RODATA, $16"),
|
||||
[]string{"GLOBL", "mask24", "<", ">", "(", "SB", ")", ",", "RODATA", ",", "$", "16"})
|
||||
}
|
||||
|
||||
func TestNegativeImmediate(t *testing.T) {
|
||||
eq(t, texts("ANDQ $-8, R10"),
|
||||
[]string{"ANDQ", "$", "-", "8", ",", "R10"})
|
||||
}
|
||||
|
||||
func TestHexImmediate(t *testing.T) {
|
||||
eq(t, texts("DATA mask24<>+0(SB)/4, $0x80020100"),
|
||||
[]string{"DATA", "mask24", "<", ">", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "0x80020100"})
|
||||
}
|
||||
|
||||
func TestMemoryAddressing(t *testing.T) {
|
||||
eq(t, texts("LEAQ (SI)(BX*4), R9"),
|
||||
[]string{"LEAQ", "(", "SI", ")", "(", "BX", "*", "4", ")", ",", "R9"})
|
||||
eq(t, texts("VMOVDQU32 Z0, 4(SI)(AX*1)"),
|
||||
[]string{"VMOVDQU32", "Z0", ",", "4", "(", "SI", ")", "(", "AX", "*", "1", ")"})
|
||||
}
|
||||
|
||||
func TestLabelAndComment(t *testing.T) {
|
||||
eq(t, kinds("vec1:\n\tJMP vec1 // loop"),
|
||||
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
|
||||
}
|
||||
|
||||
func TestAVX512Mnemonics(t *testing.T) {
|
||||
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
|
||||
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
|
||||
eq(t, texts("KTESTW K1, K1"),
|
||||
[]string{"KTESTW", "K1", ",", "K1"})
|
||||
}
|
||||
|
||||
func TestArm64Shifts(t *testing.T) {
|
||||
eq(t, texts("ADD R0<<2, R1, R2"),
|
||||
[]string{"ADD", "R0", "<<", "2", ",", "R1", ",", "R2"})
|
||||
eq(t, texts("MOVD R3->4, R5"),
|
||||
[]string{"MOVD", "R3", "->", "4", ",", "R5"})
|
||||
}
|
||||
|
||||
func TestInclude(t *testing.T) {
|
||||
eq(t, texts(`#include "textflag.h"`),
|
||||
[]string{"#", "include", `"textflag.h"`})
|
||||
}
|
||||
|
||||
func TestPositions(t *testing.T) {
|
||||
toks := Tokenize("MOVQ AX, BX\nRET")
|
||||
// Find RET and check it landed on line 2.
|
||||
var ret token.Token
|
||||
for _, tok := range toks {
|
||||
if tok.Text == "RET" {
|
||||
ret = tok
|
||||
}
|
||||
}
|
||||
if ret.Pos.Line != 2 || ret.Pos.Column != 1 {
|
||||
t.Fatalf("RET position = %v, want 2:1", ret.Pos)
|
||||
}
|
||||
}
|
||||
|
||||
func TestIllegalNeverPanics(t *testing.T) {
|
||||
// A stray backtick and NUL-ish garbage must not crash the scanner.
|
||||
toks := Tokenize("MOVQ ` , \x01 AX")
|
||||
if len(toks) == 0 {
|
||||
t.Fatal("expected tokens")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBlockComment(t *testing.T) {
|
||||
eq(t, kinds("MOVQ /* inline */ AX"),
|
||||
[]token.Kind{token.Ident, token.Comment, token.Ident})
|
||||
// An unterminated block comment is tolerated.
|
||||
toks := Tokenize("MOVQ /* never closed")
|
||||
if toks[len(toks)-2].Kind != token.Comment {
|
||||
t.Errorf("expected a comment token, got %v", toks)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRuneLiteral(t *testing.T) {
|
||||
eq(t, texts("MOVL $'a', AX"),
|
||||
[]string{"MOVL", "$", "'a'", ",", "AX"})
|
||||
}
|
||||
|
||||
func TestFloatAndBases(t *testing.T) {
|
||||
eq(t, texts("$1.5"), []string{"$", "1.5"})
|
||||
eq(t, texts("$0b1010"), []string{"$", "0b1010"})
|
||||
eq(t, texts("$0o755"), []string{"$", "0o755"})
|
||||
eq(t, texts("$1e3"), []string{"$", "1e3"})
|
||||
}
|
||||
|
||||
func TestOperatorVariants(t *testing.T) {
|
||||
eq(t, texts("R0>>2"), []string{"R0", ">>", "2"})
|
||||
eq(t, texts("@>"), []string{"@", ">"})
|
||||
eq(t, texts("a/b"), []string{"a", "/", "b"})
|
||||
}
|
||||
Reference in New Issue
Block a user