469 lines
12 KiB
Go
469 lines
12 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
// Package lexer implements a hand-written scanner for Go's Plan 9 assembler
|
|
// (GAsm). It turns a source string into a flat token stream that the parser,
|
|
// formatter and language server all build on. The scanner is deliberately
|
|
// permissive: it never panics and maps anything it cannot classify to an
|
|
// Illegal token so that downstream tools can still operate on malformed input.
|
|
package lexer
|
|
|
|
import (
|
|
"strings"
|
|
"unicode"
|
|
"unicode/utf8"
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
|
)
|
|
|
|
// middleDot is the Plan 9 symbol separator (U+00B7), used in ·funcName(SB).
|
|
const middleDot = '\u00B7'
|
|
|
|
// Lexer scans a source string one token at a time.
|
|
type Lexer struct {
|
|
src []rune
|
|
off []int // off[i] is the byte offset of src[i]; off[len(src)] is len(bytes)
|
|
i int // index of the current rune
|
|
line int // one-based line of src[i]
|
|
col int // one-based rune column of src[i]
|
|
}
|
|
|
|
// New returns a Lexer over src.
|
|
func New(src string) *Lexer {
|
|
// Decode over the raw bytes rather than converting with []rune(src): a
|
|
// lone invalid byte converts to U+FFFD, whose RuneLen is three, and the
|
|
// offset table would then count three bytes where the source has one,
|
|
// inflating every later Position.Offset against the original source.
|
|
// Decoding advances by the true byte width (one for an invalid byte)
|
|
// while the rune stream still carries RuneError, so token text keeps the
|
|
// replacement character.
|
|
runes := make([]rune, 0, len(src))
|
|
off := make([]int, 0, len(src)+1)
|
|
for b := 0; b < len(src); {
|
|
r, size := utf8.DecodeRuneInString(src[b:])
|
|
runes = append(runes, r)
|
|
off = append(off, b)
|
|
b += size
|
|
}
|
|
off = append(off, len(src))
|
|
return &Lexer{src: runes, off: off, line: 1, col: 1}
|
|
}
|
|
|
|
// Tokenize scans src fully and returns every token up to and including the
|
|
// trailing EOF token.
|
|
func Tokenize(src string) []token.Token {
|
|
l := New(src)
|
|
var out []token.Token
|
|
for {
|
|
tok := l.Next()
|
|
out = append(out, tok)
|
|
if tok.Kind == token.EOF {
|
|
return out
|
|
}
|
|
}
|
|
}
|
|
|
|
// atEnd reports whether the scanner sits past the last rune. Only the index
|
|
// decides: a literal NUL rune in the source is a real character, not the end
|
|
// of input, even though cur() returns 0 for both.
|
|
func (l *Lexer) atEnd() bool { return l.i >= len(l.src) }
|
|
|
|
// cur returns the current rune, or 0 at end of input. A real NUL rune in the
|
|
// source is indistinguishable here; callers that must tell them apart use
|
|
// atEnd.
|
|
func (l *Lexer) cur() rune {
|
|
if l.atEnd() {
|
|
return 0
|
|
}
|
|
return l.src[l.i]
|
|
}
|
|
|
|
// peek returns the rune k positions ahead, or 0 past the end.
|
|
func (l *Lexer) peek(k int) rune {
|
|
if l.i+k >= len(l.src) || l.i+k < 0 {
|
|
return 0
|
|
}
|
|
return l.src[l.i+k]
|
|
}
|
|
|
|
// pos snapshots the current source position.
|
|
func (l *Lexer) pos() token.Position {
|
|
return token.Position{Offset: l.off[l.i], Line: l.line, Column: l.col}
|
|
}
|
|
|
|
// advance consumes one rune, updating line and column bookkeeping.
|
|
func (l *Lexer) advance() {
|
|
if l.i >= len(l.src) {
|
|
return
|
|
}
|
|
if l.src[l.i] == '\n' {
|
|
l.line++
|
|
l.col = 1
|
|
} else {
|
|
l.col++
|
|
}
|
|
l.i++
|
|
}
|
|
|
|
// make builds a token of the given kind spanning [start, current position).
|
|
func (l *Lexer) make(kind token.Kind, start token.Position, text string) token.Token {
|
|
return token.Token{Kind: kind, Text: text, Pos: start, End: l.pos()}
|
|
}
|
|
|
|
// Next returns the next token, skipping spaces and tabs. Newlines are
|
|
// significant and returned as Newline tokens so the parser can treat the
|
|
// stream line by line.
|
|
func (l *Lexer) Next() token.Token {
|
|
for {
|
|
// Skip horizontal whitespace. A backslash immediately before a newline
|
|
// is a C-preprocessor line continuation (used by #define macros in the
|
|
// runtime .s files): splice the lines together by consuming both, so
|
|
// the whole macro becomes one logical line that the parser treats as an
|
|
// opaque preprocessor directive. The backslash may also reach its
|
|
// newline across whitespace and a trailing comment ("…; \ // note\n"),
|
|
// which the toolchain's scanner skips the same way.
|
|
for {
|
|
c := l.cur()
|
|
if c == ' ' || c == '\t' || c == '\r' {
|
|
l.advance()
|
|
continue
|
|
}
|
|
if c == '\\' && (l.peek(1) == '\n' || l.peek(1) == '\r') {
|
|
l.advance() // backslash
|
|
if l.cur() == '\r' {
|
|
l.advance()
|
|
}
|
|
if l.cur() == '\n' {
|
|
l.advance()
|
|
}
|
|
continue
|
|
}
|
|
if c == '\\' && l.continuationAhead() {
|
|
l.advance() // backslash, then the runes the scan saw
|
|
for !l.atEnd() && l.cur() != '\n' {
|
|
l.advance()
|
|
}
|
|
if !l.atEnd() {
|
|
l.advance() // the newline that closes the continuation
|
|
}
|
|
continue
|
|
}
|
|
break
|
|
}
|
|
|
|
start := l.pos()
|
|
r := l.cur()
|
|
|
|
switch {
|
|
case l.atEnd():
|
|
return l.make(token.EOF, start, "")
|
|
|
|
case r == 0:
|
|
// A real NUL rune (atEnd is false): fall through to punct, which
|
|
// emits it as an Illegal token and advances, so nothing after it
|
|
// is silently dropped.
|
|
return l.punct(start)
|
|
|
|
case r == '\n':
|
|
l.advance()
|
|
return l.make(token.Newline, start, "\n")
|
|
|
|
case r == '/':
|
|
switch l.peek(1) {
|
|
case '/':
|
|
return l.lineComment(start)
|
|
case '*':
|
|
return l.blockComment(start)
|
|
default:
|
|
l.advance()
|
|
return l.make(token.Slash, start, "/")
|
|
}
|
|
|
|
case r == '"':
|
|
return l.string(start)
|
|
|
|
case r == '\'':
|
|
return l.runeLit(start)
|
|
|
|
case isIdentStart(r):
|
|
return l.ident(start)
|
|
|
|
case isDigit(r):
|
|
return l.number(start)
|
|
|
|
default:
|
|
return l.punct(start)
|
|
}
|
|
}
|
|
}
|
|
|
|
// continuationAhead reports, without consuming anything, whether the
|
|
// backslash at the current position closes onto a newline through nothing
|
|
// but horizontal whitespace and one line comment. Positions after the
|
|
// backslash are inspected directly on the rune slice so a non-match leaves
|
|
// the scanner state untouched.
|
|
func (l *Lexer) continuationAhead() bool {
|
|
i := l.i + 1
|
|
for i < len(l.src) {
|
|
switch r := l.src[i]; {
|
|
case r == ' ' || r == '\t' || r == '\r':
|
|
i++
|
|
case r == '/' && i+1 < len(l.src) && l.src[i+1] == '/':
|
|
for i < len(l.src) && l.src[i] != '\n' {
|
|
i++
|
|
}
|
|
default:
|
|
return r == '\n'
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// lineComment consumes a // comment up to, but not including, the newline. A
|
|
// trailing run of \r, spaces and tabs is line-ending whitespace rather than
|
|
// comment content, so it never enters the token text. Trimming only a \r
|
|
// directly before the token's end would make the text depend on what follows
|
|
// the comment (a newline or the end of the input): "//x\r " would carry the
|
|
// "\r " while "//x\r\n" would not, and a formatter that terminates the line
|
|
// with \n would then re-lex its own output to a shorter comment.
|
|
func (l *Lexer) lineComment(start token.Position) token.Token {
|
|
var b strings.Builder
|
|
for !l.atEnd() && l.cur() != '\n' {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
return l.make(token.Comment, start, strings.TrimRight(b.String(), " \t\r"))
|
|
}
|
|
|
|
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
|
|
func (l *Lexer) blockComment(start token.Position) token.Token {
|
|
var b strings.Builder
|
|
b.WriteRune(l.cur()) // '/'
|
|
l.advance()
|
|
b.WriteRune(l.cur()) // '*'
|
|
l.advance()
|
|
for !l.atEnd() {
|
|
if l.cur() == '*' && l.peek(1) == '/' {
|
|
b.WriteString("*/")
|
|
l.advance()
|
|
l.advance()
|
|
break
|
|
}
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
return l.make(token.Comment, start, b.String())
|
|
}
|
|
|
|
// string consumes a double-quoted string literal, honouring backslash escapes.
|
|
func (l *Lexer) string(start token.Position) token.Token {
|
|
var b strings.Builder
|
|
b.WriteRune('"')
|
|
l.advance() // opening quote
|
|
for !l.atEnd() && l.cur() != '\n' {
|
|
r := l.cur()
|
|
b.WriteRune(r)
|
|
l.advance()
|
|
if r == '\\' {
|
|
if !l.atEnd() && l.cur() != '\n' {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
continue
|
|
}
|
|
if r == '"' {
|
|
return l.make(token.String, start, b.String())
|
|
}
|
|
}
|
|
// Unterminated string: return what we have rather than failing.
|
|
return l.make(token.String, start, b.String())
|
|
}
|
|
|
|
// runeLit consumes a single-quoted rune literal such as 'a' or '\n'.
|
|
func (l *Lexer) runeLit(start token.Position) token.Token {
|
|
var b strings.Builder
|
|
b.WriteRune('\'')
|
|
l.advance() // opening quote
|
|
for !l.atEnd() && l.cur() != '\n' {
|
|
r := l.cur()
|
|
b.WriteRune(r)
|
|
l.advance()
|
|
if r == '\\' {
|
|
if !l.atEnd() && l.cur() != '\n' {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
continue
|
|
}
|
|
if r == '\'' {
|
|
return l.make(token.Rune, start, b.String())
|
|
}
|
|
}
|
|
return l.make(token.Rune, start, b.String())
|
|
}
|
|
|
|
// ident consumes an identifier: letters, digits, '_', '.', and the middle dot.
|
|
func (l *Lexer) ident(start token.Position) token.Token {
|
|
var b strings.Builder
|
|
for isIdentChar(l.cur()) {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
return l.make(token.Ident, start, b.String())
|
|
}
|
|
|
|
// number consumes an integer or floating-point literal. The sign is never
|
|
// part of the literal; it is scanned separately as a Minus or Plus token.
|
|
func (l *Lexer) number(start token.Position) token.Token {
|
|
var b strings.Builder
|
|
// Base prefixes.
|
|
if l.cur() == '0' && (l.peek(1) == 'x' || l.peek(1) == 'X') {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
for isHexDigit(l.cur()) {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
return l.make(token.Number, start, b.String())
|
|
}
|
|
if l.cur() == '0' && (l.peek(1) == 'b' || l.peek(1) == 'B') {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
for l.cur() == '0' || l.cur() == '1' {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
return l.make(token.Number, start, b.String())
|
|
}
|
|
if l.cur() == '0' && (l.peek(1) == 'o' || l.peek(1) == 'O') {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
for l.cur() >= '0' && l.cur() <= '7' {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
return l.make(token.Number, start, b.String())
|
|
}
|
|
// Decimal, possibly fractional and/or with an exponent.
|
|
for isDigit(l.cur()) {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
if l.cur() == '.' && isDigit(l.peek(1)) {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
for isDigit(l.cur()) {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
}
|
|
if l.cur() == 'e' || l.cur() == 'E' {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
if l.cur() == '+' || l.cur() == '-' {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
for isDigit(l.cur()) {
|
|
b.WriteRune(l.cur())
|
|
l.advance()
|
|
}
|
|
}
|
|
return l.make(token.Number, start, b.String())
|
|
}
|
|
|
|
// punct consumes a single punctuation or operator token, handling the
|
|
// multi-character operators <<, >> and ->.
|
|
func (l *Lexer) punct(start token.Position) token.Token {
|
|
r := l.cur()
|
|
switch r {
|
|
case '(':
|
|
l.advance()
|
|
return l.make(token.LParen, start, "(")
|
|
case ')':
|
|
l.advance()
|
|
return l.make(token.RParen, start, ")")
|
|
case ',':
|
|
l.advance()
|
|
return l.make(token.Comma, start, ",")
|
|
case '+':
|
|
l.advance()
|
|
return l.make(token.Plus, start, "+")
|
|
case '-':
|
|
if l.peek(1) == '>' {
|
|
l.advance()
|
|
l.advance()
|
|
return l.make(token.Arrow, start, "->")
|
|
}
|
|
l.advance()
|
|
return l.make(token.Minus, start, "-")
|
|
case '*':
|
|
l.advance()
|
|
return l.make(token.Star, start, "*")
|
|
case ':':
|
|
l.advance()
|
|
return l.make(token.Colon, start, ":")
|
|
case '$':
|
|
l.advance()
|
|
return l.make(token.Dollar, start, "$")
|
|
case '<':
|
|
if l.peek(1) == '<' {
|
|
l.advance()
|
|
l.advance()
|
|
return l.make(token.LShift, start, "<<")
|
|
}
|
|
l.advance()
|
|
return l.make(token.LAngle, start, "<")
|
|
case '>':
|
|
if l.peek(1) == '>' {
|
|
l.advance()
|
|
l.advance()
|
|
return l.make(token.RShift, start, ">>")
|
|
}
|
|
l.advance()
|
|
return l.make(token.RAngle, start, ">")
|
|
case '@':
|
|
l.advance()
|
|
return l.make(token.At, start, "@")
|
|
case '#':
|
|
l.advance()
|
|
return l.make(token.Hash, start, "#")
|
|
case '|':
|
|
l.advance()
|
|
return l.make(token.Pipe, start, "|")
|
|
case ';':
|
|
l.advance()
|
|
return l.make(token.Semicolon, start, ";")
|
|
case '&':
|
|
l.advance()
|
|
return l.make(token.Ampersand, start, "&")
|
|
case '~':
|
|
l.advance()
|
|
return l.make(token.Tilde, start, "~")
|
|
default:
|
|
// Unknown rune: emit it as Illegal and move on.
|
|
l.advance()
|
|
return l.make(token.Illegal, start, string(r))
|
|
}
|
|
}
|
|
|
|
func isDigit(r rune) bool { return r >= '0' && r <= '9' }
|
|
|
|
func isHexDigit(r rune) bool {
|
|
return isDigit(r) || (r >= 'a' && r <= 'f') || (r >= 'A' && r <= 'F')
|
|
}
|
|
|
|
func isIdentStart(r rune) bool {
|
|
return r == '_' || r == middleDot || unicode.IsLetter(r)
|
|
}
|
|
|
|
func isIdentChar(r rune) bool {
|
|
return isIdentStart(r) || isDigit(r) || r == '.'
|
|
}
|