// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause // Package lexer implements a hand-written scanner for Go's Plan 9 assembler // (GAsm). It turns a source string into a flat token stream that the parser, // formatter and language server all build on. The scanner is deliberately // permissive: it never panics and maps anything it cannot classify to an // Illegal token so that downstream tools can still operate on malformed input. package lexer import ( "strings" "unicode" "unicode/utf8" "sourcedock.dev/petrbalvin/gasm-devkit/token" ) // middleDot is the Plan 9 symbol separator (U+00B7), used in ·funcName(SB). const middleDot = '\u00B7' // Lexer scans a source string one token at a time. type Lexer struct { src []rune off []int // off[i] is the byte offset of src[i]; off[len(src)] is len(bytes) i int // index of the current rune line int // one-based line of src[i] col int // one-based rune column of src[i] } // New returns a Lexer over src. func New(src string) *Lexer { runes := []rune(src) off := make([]int, len(runes)+1) b := 0 for i, r := range runes { off[i] = b b += utf8.RuneLen(r) } off[len(runes)] = b return &Lexer{src: runes, off: off, line: 1, col: 1} } // Tokenize scans src fully and returns every token up to and including the // trailing EOF token. func Tokenize(src string) []token.Token { l := New(src) var out []token.Token for { tok := l.Next() out = append(out, tok) if tok.Kind == token.EOF { return out } } } // cur returns the current rune, or 0 at end of input. func (l *Lexer) cur() rune { if l.i >= len(l.src) { return 0 } return l.src[l.i] } // peek returns the rune k positions ahead, or 0 past the end. func (l *Lexer) peek(k int) rune { if l.i+k >= len(l.src) || l.i+k < 0 { return 0 } return l.src[l.i+k] } // pos snapshots the current source position. func (l *Lexer) pos() token.Position { return token.Position{Offset: l.off[l.i], Line: l.line, Column: l.col} } // advance consumes one rune, updating line and column bookkeeping. func (l *Lexer) advance() { if l.i >= len(l.src) { return } if l.src[l.i] == '\n' { l.line++ l.col = 1 } else { l.col++ } l.i++ } // make builds a token of the given kind spanning [start, current position). func (l *Lexer) make(kind token.Kind, start token.Position, text string) token.Token { return token.Token{Kind: kind, Text: text, Pos: start, End: l.pos()} } // Next returns the next token, skipping spaces and tabs. Newlines are // significant and returned as Newline tokens so the parser can treat the // stream line by line. func (l *Lexer) Next() token.Token { for { // Skip horizontal whitespace. A backslash immediately before a newline // is a C-preprocessor line continuation (used by #define macros in the // runtime .s files): splice the lines together by consuming both, so // the whole macro becomes one logical line that the parser treats as an // opaque preprocessor directive. for { c := l.cur() if c == ' ' || c == '\t' || c == '\r' { l.advance() continue } if c == '\\' && (l.peek(1) == '\n' || l.peek(1) == '\r') { l.advance() // backslash if l.cur() == '\r' { l.advance() } if l.cur() == '\n' { l.advance() } continue } break } start := l.pos() r := l.cur() switch { case r == 0: return l.make(token.EOF, start, "") case r == '\n': l.advance() return l.make(token.Newline, start, "\n") case r == '/': switch l.peek(1) { case '/': return l.lineComment(start) case '*': return l.blockComment(start) default: l.advance() return l.make(token.Slash, start, "/") } case r == '"': return l.string(start) case r == '\'': return l.runeLit(start) case isIdentStart(r): return l.ident(start) case isDigit(r): return l.number(start) default: return l.punct(start) } } } // lineComment consumes a // comment up to, but not including, the newline. func (l *Lexer) lineComment(start token.Position) token.Token { var b strings.Builder for l.cur() != 0 && l.cur() != '\n' { b.WriteRune(l.cur()) l.advance() } return l.make(token.Comment, start, b.String()) } // blockComment consumes a /* ... */ comment, tolerating an unterminated one. func (l *Lexer) blockComment(start token.Position) token.Token { var b strings.Builder b.WriteRune(l.cur()) // '/' l.advance() b.WriteRune(l.cur()) // '*' l.advance() for l.cur() != 0 { if l.cur() == '*' && l.peek(1) == '/' { b.WriteString("*/") l.advance() l.advance() break } b.WriteRune(l.cur()) l.advance() } return l.make(token.Comment, start, b.String()) } // string consumes a double-quoted string literal, honouring backslash escapes. func (l *Lexer) string(start token.Position) token.Token { var b strings.Builder b.WriteRune('"') l.advance() // opening quote for l.cur() != 0 && l.cur() != '\n' { r := l.cur() b.WriteRune(r) l.advance() if r == '\\' { if l.cur() != 0 && l.cur() != '\n' { b.WriteRune(l.cur()) l.advance() } continue } if r == '"' { return l.make(token.String, start, b.String()) } } // Unterminated string: return what we have rather than failing. return l.make(token.String, start, b.String()) } // runeLit consumes a single-quoted rune literal such as 'a' or '\n'. func (l *Lexer) runeLit(start token.Position) token.Token { var b strings.Builder b.WriteRune('\'') l.advance() // opening quote for l.cur() != 0 && l.cur() != '\n' { r := l.cur() b.WriteRune(r) l.advance() if r == '\\' { if l.cur() != 0 && l.cur() != '\n' { b.WriteRune(l.cur()) l.advance() } continue } if r == '\'' { return l.make(token.Rune, start, b.String()) } } return l.make(token.Rune, start, b.String()) } // ident consumes an identifier: letters, digits, '_', '.', and the middle dot. func (l *Lexer) ident(start token.Position) token.Token { var b strings.Builder for isIdentChar(l.cur()) { b.WriteRune(l.cur()) l.advance() } return l.make(token.Ident, start, b.String()) } // number consumes an integer or floating-point literal. The sign is never // part of the literal; it is scanned separately as a Minus or Plus token. func (l *Lexer) number(start token.Position) token.Token { var b strings.Builder // Base prefixes. if l.cur() == '0' && (l.peek(1) == 'x' || l.peek(1) == 'X') { b.WriteRune(l.cur()) l.advance() b.WriteRune(l.cur()) l.advance() for isHexDigit(l.cur()) { b.WriteRune(l.cur()) l.advance() } return l.make(token.Number, start, b.String()) } if l.cur() == '0' && (l.peek(1) == 'b' || l.peek(1) == 'B') { b.WriteRune(l.cur()) l.advance() b.WriteRune(l.cur()) l.advance() for l.cur() == '0' || l.cur() == '1' { b.WriteRune(l.cur()) l.advance() } return l.make(token.Number, start, b.String()) } if l.cur() == '0' && (l.peek(1) == 'o' || l.peek(1) == 'O') { b.WriteRune(l.cur()) l.advance() b.WriteRune(l.cur()) l.advance() for l.cur() >= '0' && l.cur() <= '7' { b.WriteRune(l.cur()) l.advance() } return l.make(token.Number, start, b.String()) } // Decimal, possibly fractional and/or with an exponent. for isDigit(l.cur()) { b.WriteRune(l.cur()) l.advance() } if l.cur() == '.' && isDigit(l.peek(1)) { b.WriteRune(l.cur()) l.advance() for isDigit(l.cur()) { b.WriteRune(l.cur()) l.advance() } } if l.cur() == 'e' || l.cur() == 'E' { b.WriteRune(l.cur()) l.advance() if l.cur() == '+' || l.cur() == '-' { b.WriteRune(l.cur()) l.advance() } for isDigit(l.cur()) { b.WriteRune(l.cur()) l.advance() } } return l.make(token.Number, start, b.String()) } // punct consumes a single punctuation or operator token, handling the // multi-character operators <<, >> and ->. func (l *Lexer) punct(start token.Position) token.Token { r := l.cur() switch r { case '(': l.advance() return l.make(token.LParen, start, "(") case ')': l.advance() return l.make(token.RParen, start, ")") case ',': l.advance() return l.make(token.Comma, start, ",") case '+': l.advance() return l.make(token.Plus, start, "+") case '-': if l.peek(1) == '>' { l.advance() l.advance() return l.make(token.Arrow, start, "->") } l.advance() return l.make(token.Minus, start, "-") case '*': l.advance() return l.make(token.Star, start, "*") case ':': l.advance() return l.make(token.Colon, start, ":") case '$': l.advance() return l.make(token.Dollar, start, "$") case '<': if l.peek(1) == '<' { l.advance() l.advance() return l.make(token.LShift, start, "<<") } l.advance() return l.make(token.LAngle, start, "<") case '>': if l.peek(1) == '>' { l.advance() l.advance() return l.make(token.RShift, start, ">>") } l.advance() return l.make(token.RAngle, start, ">") case '@': l.advance() return l.make(token.At, start, "@") case '#': l.advance() return l.make(token.Hash, start, "#") default: // Unknown rune: emit it as Illegal and move on. l.advance() return l.make(token.Illegal, start, string(r)) } } func isDigit(r rune) bool { return r >= '0' && r <= '9' } func isHexDigit(r rune) bool { return isDigit(r) || (r >= 'a' && r <= 'f') || (r >= 'A' && r <= 'F') } func isIdentStart(r rune) bool { return r == '_' || r == middleDot || unicode.IsLetter(r) } func isIdentChar(r rune) bool { return isIdentStart(r) || isDigit(r) || r == '.' }