Files

477 lines
13 KiB
Go
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package lexer implements a hand-written scanner for Go's Plan 9 assembler
// (GAsm). It turns a source string into a flat token stream that the parser,
// formatter and language server all build on. The scanner is deliberately
// permissive: it never panics and maps anything it cannot classify to an
// Illegal token so that downstream tools can still operate on malformed input.
package lexer
import (
"strings"
"unicode"
"unicode/utf8"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
const (
// middleDot is the Plan 9 symbol separator (U+00B7), used in
// ·funcName(SB): it stands for the period between package path and name.
middleDot = '\u00B7'
// divisionSlash is the Plan 9 path separator (U+2215), used inside the
// package path of a symbol: internal∕runtime∕atomic·Xchg. Like the
// middle dot it is an identifier character, so a package path containing
// it lexes as one name; the ordinary slash (U+002F) stays punctuation.
divisionSlash = '\u2215'
)
// Lexer scans a source string one token at a time.
type Lexer struct {
src []rune
off []int // off[i] is the byte offset of src[i]; off[len(src)] is len(bytes)
i int // index of the current rune
line int // one-based line of src[i]
col int // one-based rune column of src[i]
}
// New returns a Lexer over src.
func New(src string) *Lexer {
// Decode over the raw bytes rather than converting with []rune(src): a
// lone invalid byte converts to U+FFFD, whose RuneLen is three, and the
// offset table would then count three bytes where the source has one,
// inflating every later Position.Offset against the original source.
// Decoding advances by the true byte width (one for an invalid byte)
// while the rune stream still carries RuneError, so token text keeps the
// replacement character.
runes := make([]rune, 0, len(src))
off := make([]int, 0, len(src)+1)
for b := 0; b < len(src); {
r, size := utf8.DecodeRuneInString(src[b:])
runes = append(runes, r)
off = append(off, b)
b += size
}
off = append(off, len(src))
return &Lexer{src: runes, off: off, line: 1, col: 1}
}
// Tokenize scans src fully and returns every token up to and including the
// trailing EOF token.
func Tokenize(src string) []token.Token {
l := New(src)
var out []token.Token
for {
tok := l.Next()
out = append(out, tok)
if tok.Kind == token.EOF {
return out
}
}
}
// atEnd reports whether the scanner sits past the last rune. Only the index
// decides: a literal NUL rune in the source is a real character, not the end
// of input, even though cur() returns 0 for both.
func (l *Lexer) atEnd() bool { return l.i >= len(l.src) }
// cur returns the current rune, or 0 at end of input. A real NUL rune in the
// source is indistinguishable here; callers that must tell them apart use
// atEnd.
func (l *Lexer) cur() rune {
if l.atEnd() {
return 0
}
return l.src[l.i]
}
// peek returns the rune k positions ahead, or 0 past the end.
func (l *Lexer) peek(k int) rune {
if l.i+k >= len(l.src) || l.i+k < 0 {
return 0
}
return l.src[l.i+k]
}
// pos snapshots the current source position.
func (l *Lexer) pos() token.Position {
return token.Position{Offset: l.off[l.i], Line: l.line, Column: l.col}
}
// advance consumes one rune, updating line and column bookkeeping.
func (l *Lexer) advance() {
if l.i >= len(l.src) {
return
}
if l.src[l.i] == '\n' {
l.line++
l.col = 1
} else {
l.col++
}
l.i++
}
// make builds a token of the given kind spanning [start, current position).
func (l *Lexer) make(kind token.Kind, start token.Position, text string) token.Token {
return token.Token{Kind: kind, Text: text, Pos: start, End: l.pos()}
}
// Next returns the next token, skipping spaces and tabs. Newlines are
// significant and returned as Newline tokens so the parser can treat the
// stream line by line.
func (l *Lexer) Next() token.Token {
for {
// Skip horizontal whitespace. A backslash immediately before a newline
// is a C-preprocessor line continuation (used by #define macros in the
// runtime .s files): splice the lines together by consuming both, so
// the whole macro becomes one logical line that the parser treats as an
// opaque preprocessor directive. The backslash may also reach its
// newline across whitespace and a trailing comment ("…; \ // note\n"),
// which the toolchain's scanner skips the same way.
for {
c := l.cur()
if c == ' ' || c == '\t' || c == '\r' {
l.advance()
continue
}
if c == '\\' && (l.peek(1) == '\n' || l.peek(1) == '\r') {
l.advance() // backslash
if l.cur() == '\r' {
l.advance()
}
if l.cur() == '\n' {
l.advance()
}
continue
}
if c == '\\' && l.continuationAhead() {
l.advance() // backslash, then the runes the scan saw
for !l.atEnd() && l.cur() != '\n' {
l.advance()
}
if !l.atEnd() {
l.advance() // the newline that closes the continuation
}
continue
}
break
}
start := l.pos()
r := l.cur()
switch {
case l.atEnd():
return l.make(token.EOF, start, "")
case r == 0:
// A real NUL rune (atEnd is false): fall through to punct, which
// emits it as an Illegal token and advances, so nothing after it
// is silently dropped.
return l.punct(start)
case r == '\n':
l.advance()
return l.make(token.Newline, start, "\n")
case r == '/':
switch l.peek(1) {
case '/':
return l.lineComment(start)
case '*':
return l.blockComment(start)
default:
l.advance()
return l.make(token.Slash, start, "/")
}
case r == '"':
return l.string(start)
case r == '\'':
return l.runeLit(start)
case isIdentStart(r):
return l.ident(start)
case isDigit(r):
return l.number(start)
default:
return l.punct(start)
}
}
}
// continuationAhead reports, without consuming anything, whether the
// backslash at the current position closes onto a newline through nothing
// but horizontal whitespace and one line comment. Positions after the
// backslash are inspected directly on the rune slice so a non-match leaves
// the scanner state untouched.
func (l *Lexer) continuationAhead() bool {
i := l.i + 1
for i < len(l.src) {
switch r := l.src[i]; {
case r == ' ' || r == '\t' || r == '\r':
i++
case r == '/' && i+1 < len(l.src) && l.src[i+1] == '/':
for i < len(l.src) && l.src[i] != '\n' {
i++
}
default:
return r == '\n'
}
}
return false
}
// lineComment consumes a // comment up to, but not including, the newline. A
// trailing run of \r, spaces and tabs is line-ending whitespace rather than
// comment content, so it never enters the token text. Trimming only a \r
// directly before the token's end would make the text depend on what follows
// the comment (a newline or the end of the input): "//x\r " would carry the
// "\r " while "//x\r\n" would not, and a formatter that terminates the line
// with \n would then re-lex its own output to a shorter comment.
func (l *Lexer) lineComment(start token.Position) token.Token {
var b strings.Builder
for !l.atEnd() && l.cur() != '\n' {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Comment, start, strings.TrimRight(b.String(), " \t\r"))
}
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
func (l *Lexer) blockComment(start token.Position) token.Token {
var b strings.Builder
b.WriteRune(l.cur()) // '/'
l.advance()
b.WriteRune(l.cur()) // '*'
l.advance()
for !l.atEnd() {
if l.cur() == '*' && l.peek(1) == '/' {
b.WriteString("*/")
l.advance()
l.advance()
break
}
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Comment, start, b.String())
}
// string consumes a double-quoted string literal, honouring backslash escapes.
func (l *Lexer) string(start token.Position) token.Token {
var b strings.Builder
b.WriteRune('"')
l.advance() // opening quote
for !l.atEnd() && l.cur() != '\n' {
r := l.cur()
b.WriteRune(r)
l.advance()
if r == '\\' {
if !l.atEnd() && l.cur() != '\n' {
b.WriteRune(l.cur())
l.advance()
}
continue
}
if r == '"' {
return l.make(token.String, start, b.String())
}
}
// Unterminated string: return what we have rather than failing.
return l.make(token.String, start, b.String())
}
// runeLit consumes a single-quoted rune literal such as 'a' or '\n'.
func (l *Lexer) runeLit(start token.Position) token.Token {
var b strings.Builder
b.WriteRune('\'')
l.advance() // opening quote
for !l.atEnd() && l.cur() != '\n' {
r := l.cur()
b.WriteRune(r)
l.advance()
if r == '\\' {
if !l.atEnd() && l.cur() != '\n' {
b.WriteRune(l.cur())
l.advance()
}
continue
}
if r == '\'' {
return l.make(token.Rune, start, b.String())
}
}
return l.make(token.Rune, start, b.String())
}
// ident consumes an identifier: letters, digits, '_', '.', and the middle dot.
func (l *Lexer) ident(start token.Position) token.Token {
var b strings.Builder
for isIdentChar(l.cur()) {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Ident, start, b.String())
}
// number consumes an integer or floating-point literal. The sign is never
// part of the literal; it is scanned separately as a Minus or Plus token.
func (l *Lexer) number(start token.Position) token.Token {
var b strings.Builder
// Base prefixes.
if l.cur() == '0' && (l.peek(1) == 'x' || l.peek(1) == 'X') {
b.WriteRune(l.cur())
l.advance()
b.WriteRune(l.cur())
l.advance()
for isHexDigit(l.cur()) {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Number, start, b.String())
}
if l.cur() == '0' && (l.peek(1) == 'b' || l.peek(1) == 'B') {
b.WriteRune(l.cur())
l.advance()
b.WriteRune(l.cur())
l.advance()
for l.cur() == '0' || l.cur() == '1' {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Number, start, b.String())
}
if l.cur() == '0' && (l.peek(1) == 'o' || l.peek(1) == 'O') {
b.WriteRune(l.cur())
l.advance()
b.WriteRune(l.cur())
l.advance()
for l.cur() >= '0' && l.cur() <= '7' {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Number, start, b.String())
}
// Decimal, possibly fractional and/or with an exponent.
for isDigit(l.cur()) {
b.WriteRune(l.cur())
l.advance()
}
if l.cur() == '.' && isDigit(l.peek(1)) {
b.WriteRune(l.cur())
l.advance()
for isDigit(l.cur()) {
b.WriteRune(l.cur())
l.advance()
}
}
if l.cur() == 'e' || l.cur() == 'E' {
b.WriteRune(l.cur())
l.advance()
if l.cur() == '+' || l.cur() == '-' {
b.WriteRune(l.cur())
l.advance()
}
for isDigit(l.cur()) {
b.WriteRune(l.cur())
l.advance()
}
}
return l.make(token.Number, start, b.String())
}
// punct consumes a single punctuation or operator token, handling the
// multi-character operators <<, >> and ->.
func (l *Lexer) punct(start token.Position) token.Token {
r := l.cur()
switch r {
case '(':
l.advance()
return l.make(token.LParen, start, "(")
case ')':
l.advance()
return l.make(token.RParen, start, ")")
case ',':
l.advance()
return l.make(token.Comma, start, ",")
case '+':
l.advance()
return l.make(token.Plus, start, "+")
case '-':
if l.peek(1) == '>' {
l.advance()
l.advance()
return l.make(token.Arrow, start, "->")
}
l.advance()
return l.make(token.Minus, start, "-")
case '*':
l.advance()
return l.make(token.Star, start, "*")
case ':':
l.advance()
return l.make(token.Colon, start, ":")
case '$':
l.advance()
return l.make(token.Dollar, start, "$")
case '<':
if l.peek(1) == '<' {
l.advance()
l.advance()
return l.make(token.LShift, start, "<<")
}
l.advance()
return l.make(token.LAngle, start, "<")
case '>':
if l.peek(1) == '>' {
l.advance()
l.advance()
return l.make(token.RShift, start, ">>")
}
l.advance()
return l.make(token.RAngle, start, ">")
case '@':
l.advance()
return l.make(token.At, start, "@")
case '#':
l.advance()
return l.make(token.Hash, start, "#")
case '|':
l.advance()
return l.make(token.Pipe, start, "|")
case ';':
l.advance()
return l.make(token.Semicolon, start, ";")
case '&':
l.advance()
return l.make(token.Ampersand, start, "&")
case '~':
l.advance()
return l.make(token.Tilde, start, "~")
default:
// Unknown rune: emit it as Illegal and move on.
l.advance()
return l.make(token.Illegal, start, string(r))
}
}
func isDigit(r rune) bool { return r >= '0' && r <= '9' }
func isHexDigit(r rune) bool {
return isDigit(r) || (r >= 'a' && r <= 'f') || (r >= 'A' && r <= 'F')
}
func isIdentStart(r rune) bool {
return r == '_' || r == middleDot || r == divisionSlash || unicode.IsLetter(r)
}
func isIdentChar(r rune) bool {
return isIdentStart(r) || isDigit(r) || r == '.'
}