// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause // Package lexer implements a hand-written scanner for Go's Plan 9 assembler // (GAsm). It turns a source string into a flat token stream that the parser, // formatter and language server all build on. The scanner is deliberately // permissive: it never panics and maps anything it cannot classify to an // Illegal token so that downstream tools can still operate on malformed input. package lexer import ( "strings" "unicode" "unicode/utf8" "sourcedock.dev/petrbalvin/gasm-devkit/token" ) // middleDot is the Plan 9 symbol separator (U+00B7), used in ·funcName(SB). const middleDot = '\u00B7' // Lexer scans a source string one token at a time. type Lexer struct { src []rune off []int // off[i] is the byte offset of src[i]; off[len(src)] is len(bytes) i int // index of the current rune line int // one-based line of src[i] col int // one-based rune column of src[i] } // New returns a Lexer over src. func New(src string) *Lexer { // Decode over the raw bytes rather than converting with []rune(src): a // lone invalid byte converts to U+FFFD, whose RuneLen is three, and the // offset table would then count three bytes where the source has one, // inflating every later Position.Offset against the original source. // Decoding advances by the true byte width (one for an invalid byte) // while the rune stream still carries RuneError, so token text keeps the // replacement character. runes := make([]rune, 0, len(src)) off := make([]int, 0, len(src)+1) for b := 0; b < len(src); { r, size := utf8.DecodeRuneInString(src[b:]) runes = append(runes, r) off = append(off, b) b += size } off = append(off, len(src)) return &Lexer{src: runes, off: off, line: 1, col: 1} } // Tokenize scans src fully and returns every token up to and including the // trailing EOF token. func Tokenize(src string) []token.Token { l := New(src) var out []token.Token for { tok := l.Next() out = append(out, tok) if tok.Kind == token.EOF { return out } } } // atEnd reports whether the scanner sits past the last rune. Only the index // decides: a literal NUL rune in the source is a real character, not the end // of input, even though cur() returns 0 for both. func (l *Lexer) atEnd() bool { return l.i >= len(l.src) } // cur returns the current rune, or 0 at end of input. A real NUL rune in the // source is indistinguishable here; callers that must tell them apart use // atEnd. func (l *Lexer) cur() rune { if l.atEnd() { return 0 } return l.src[l.i] } // peek returns the rune k positions ahead, or 0 past the end. func (l *Lexer) peek(k int) rune { if l.i+k >= len(l.src) || l.i+k < 0 { return 0 } return l.src[l.i+k] } // pos snapshots the current source position. func (l *Lexer) pos() token.Position { return token.Position{Offset: l.off[l.i], Line: l.line, Column: l.col} } // advance consumes one rune, updating line and column bookkeeping. func (l *Lexer) advance() { if l.i >= len(l.src) { return } if l.src[l.i] == '\n' { l.line++ l.col = 1 } else { l.col++ } l.i++ } // make builds a token of the given kind spanning [start, current position). func (l *Lexer) make(kind token.Kind, start token.Position, text string) token.Token { return token.Token{Kind: kind, Text: text, Pos: start, End: l.pos()} } // Next returns the next token, skipping spaces and tabs. Newlines are // significant and returned as Newline tokens so the parser can treat the // stream line by line. func (l *Lexer) Next() token.Token { for { // Skip horizontal whitespace. A backslash immediately before a newline // is a C-preprocessor line continuation (used by #define macros in the // runtime .s files): splice the lines together by consuming both, so // the whole macro becomes one logical line that the parser treats as an // opaque preprocessor directive. The backslash may also reach its // newline across whitespace and a trailing comment ("…; \ // note\n"), // which the toolchain's scanner skips the same way. for { c := l.cur() if c == ' ' || c == '\t' || c == '\r' { l.advance() continue } if c == '\\' && (l.peek(1) == '\n' || l.peek(1) == '\r') { l.advance() // backslash if l.cur() == '\r' { l.advance() } if l.cur() == '\n' { l.advance() } continue } if c == '\\' && l.continuationAhead() { l.advance() // backslash, then the runes the scan saw for !l.atEnd() && l.cur() != '\n' { l.advance() } if !l.atEnd() { l.advance() // the newline that closes the continuation } continue } break } start := l.pos() r := l.cur() switch { case l.atEnd(): return l.make(token.EOF, start, "") case r == 0: // A real NUL rune (atEnd is false): fall through to punct, which // emits it as an Illegal token and advances, so nothing after it // is silently dropped. return l.punct(start) case r == '\n': l.advance() return l.make(token.Newline, start, "\n") case r == '/': switch l.peek(1) { case '/': return l.lineComment(start) case '*': return l.blockComment(start) default: l.advance() return l.make(token.Slash, start, "/") } case r == '"': return l.string(start) case r == '\'': return l.runeLit(start) case isIdentStart(r): return l.ident(start) case isDigit(r): return l.number(start) default: return l.punct(start) } } } // continuationAhead reports, without consuming anything, whether the // backslash at the current position closes onto a newline through nothing // but horizontal whitespace and one line comment. Positions after the // backslash are inspected directly on the rune slice so a non-match leaves // the scanner state untouched. func (l *Lexer) continuationAhead() bool { i := l.i + 1 for i < len(l.src) { switch r := l.src[i]; { case r == ' ' || r == '\t' || r == '\r': i++ case r == '/' && i+1 < len(l.src) && l.src[i+1] == '/': for i < len(l.src) && l.src[i] != '\n' { i++ } default: return r == '\n' } } return false } // lineComment consumes a // comment up to, but not including, the newline. A // trailing run of \r, spaces and tabs is line-ending whitespace rather than // comment content, so it never enters the token text. Trimming only a \r // directly before the token's end would make the text depend on what follows // the comment (a newline or the end of the input): "//x\r " would carry the // "\r " while "//x\r\n" would not, and a formatter that terminates the line // with \n would then re-lex its own output to a shorter comment. func (l *Lexer) lineComment(start token.Position) token.Token { var b strings.Builder for !l.atEnd() && l.cur() != '\n' { b.WriteRune(l.cur()) l.advance() } return l.make(token.Comment, start, strings.TrimRight(b.String(), " \t\r")) } // blockComment consumes a /* ... */ comment, tolerating an unterminated one. func (l *Lexer) blockComment(start token.Position) token.Token { var b strings.Builder b.WriteRune(l.cur()) // '/' l.advance() b.WriteRune(l.cur()) // '*' l.advance() for !l.atEnd() { if l.cur() == '*' && l.peek(1) == '/' { b.WriteString("*/") l.advance() l.advance() break } b.WriteRune(l.cur()) l.advance() } return l.make(token.Comment, start, b.String()) } // string consumes a double-quoted string literal, honouring backslash escapes. func (l *Lexer) string(start token.Position) token.Token { var b strings.Builder b.WriteRune('"') l.advance() // opening quote for !l.atEnd() && l.cur() != '\n' { r := l.cur() b.WriteRune(r) l.advance() if r == '\\' { if !l.atEnd() && l.cur() != '\n' { b.WriteRune(l.cur()) l.advance() } continue } if r == '"' { return l.make(token.String, start, b.String()) } } // Unterminated string: return what we have rather than failing. return l.make(token.String, start, b.String()) } // runeLit consumes a single-quoted rune literal such as 'a' or '\n'. func (l *Lexer) runeLit(start token.Position) token.Token { var b strings.Builder b.WriteRune('\'') l.advance() // opening quote for !l.atEnd() && l.cur() != '\n' { r := l.cur() b.WriteRune(r) l.advance() if r == '\\' { if !l.atEnd() && l.cur() != '\n' { b.WriteRune(l.cur()) l.advance() } continue } if r == '\'' { return l.make(token.Rune, start, b.String()) } } return l.make(token.Rune, start, b.String()) } // ident consumes an identifier: letters, digits, '_', '.', and the middle dot. func (l *Lexer) ident(start token.Position) token.Token { var b strings.Builder for isIdentChar(l.cur()) { b.WriteRune(l.cur()) l.advance() } return l.make(token.Ident, start, b.String()) } // number consumes an integer or floating-point literal. The sign is never // part of the literal; it is scanned separately as a Minus or Plus token. func (l *Lexer) number(start token.Position) token.Token { var b strings.Builder // Base prefixes. if l.cur() == '0' && (l.peek(1) == 'x' || l.peek(1) == 'X') { b.WriteRune(l.cur()) l.advance() b.WriteRune(l.cur()) l.advance() for isHexDigit(l.cur()) { b.WriteRune(l.cur()) l.advance() } return l.make(token.Number, start, b.String()) } if l.cur() == '0' && (l.peek(1) == 'b' || l.peek(1) == 'B') { b.WriteRune(l.cur()) l.advance() b.WriteRune(l.cur()) l.advance() for l.cur() == '0' || l.cur() == '1' { b.WriteRune(l.cur()) l.advance() } return l.make(token.Number, start, b.String()) } if l.cur() == '0' && (l.peek(1) == 'o' || l.peek(1) == 'O') { b.WriteRune(l.cur()) l.advance() b.WriteRune(l.cur()) l.advance() for l.cur() >= '0' && l.cur() <= '7' { b.WriteRune(l.cur()) l.advance() } return l.make(token.Number, start, b.String()) } // Decimal, possibly fractional and/or with an exponent. for isDigit(l.cur()) { b.WriteRune(l.cur()) l.advance() } if l.cur() == '.' && isDigit(l.peek(1)) { b.WriteRune(l.cur()) l.advance() for isDigit(l.cur()) { b.WriteRune(l.cur()) l.advance() } } if l.cur() == 'e' || l.cur() == 'E' { b.WriteRune(l.cur()) l.advance() if l.cur() == '+' || l.cur() == '-' { b.WriteRune(l.cur()) l.advance() } for isDigit(l.cur()) { b.WriteRune(l.cur()) l.advance() } } return l.make(token.Number, start, b.String()) } // punct consumes a single punctuation or operator token, handling the // multi-character operators <<, >> and ->. func (l *Lexer) punct(start token.Position) token.Token { r := l.cur() switch r { case '(': l.advance() return l.make(token.LParen, start, "(") case ')': l.advance() return l.make(token.RParen, start, ")") case ',': l.advance() return l.make(token.Comma, start, ",") case '+': l.advance() return l.make(token.Plus, start, "+") case '-': if l.peek(1) == '>' { l.advance() l.advance() return l.make(token.Arrow, start, "->") } l.advance() return l.make(token.Minus, start, "-") case '*': l.advance() return l.make(token.Star, start, "*") case ':': l.advance() return l.make(token.Colon, start, ":") case '$': l.advance() return l.make(token.Dollar, start, "$") case '<': if l.peek(1) == '<' { l.advance() l.advance() return l.make(token.LShift, start, "<<") } l.advance() return l.make(token.LAngle, start, "<") case '>': if l.peek(1) == '>' { l.advance() l.advance() return l.make(token.RShift, start, ">>") } l.advance() return l.make(token.RAngle, start, ">") case '@': l.advance() return l.make(token.At, start, "@") case '#': l.advance() return l.make(token.Hash, start, "#") case '|': l.advance() return l.make(token.Pipe, start, "|") case ';': l.advance() return l.make(token.Semicolon, start, ";") case '&': l.advance() return l.make(token.Ampersand, start, "&") case '~': l.advance() return l.make(token.Tilde, start, "~") default: // Unknown rune: emit it as Illegal and move on. l.advance() return l.make(token.Illegal, start, string(r)) } } func isDigit(r rune) bool { return r >= '0' && r <= '9' } func isHexDigit(r rune) bool { return isDigit(r) || (r >= 'a' && r <= 'f') || (r >= 'A' && r <= 'F') } func isIdentStart(r rune) bool { return r == '_' || r == middleDot || unicode.IsLetter(r) } func isIdentChar(r rune) bool { return isIdentStart(r) || isDigit(r) || r == '.' }