Files
gasm-sdk/lexer/lexer_test.go
T

233 lines
7.8 KiB
Go
Raw Permalink Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lexer
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// kinds tokenizes src and returns the kind sequence, dropping Newline/EOF.
func kinds(src string) []token.Kind {
var out []token.Kind
for _, t := range Tokenize(src) {
if t.Kind == token.Newline || t.Kind == token.EOF {
continue
}
out = append(out, t.Kind)
}
return out
}
// texts tokenizes src and returns the literal text of each significant token.
func texts(src string) []string {
var out []string
for _, t := range Tokenize(src) {
if t.Kind == token.Newline || t.Kind == token.EOF {
continue
}
out = append(out, t.Text)
}
return out
}
func eq[T comparable](t *testing.T, got, want []T) {
t.Helper()
if len(got) != len(want) {
t.Fatalf("length mismatch:\n got %v\n want %v", got, want)
}
for i := range got {
if got[i] != want[i] {
t.Fatalf("index %d:\n got %v\n want %v", i, got, want)
}
}
}
func TestTextDirective(t *testing.T) {
eq(t, texts("TEXT ·analyzeO1RangeAVX2(SB), NOSPLIT, $0-65"),
[]string{"TEXT", "·analyzeO1RangeAVX2", "(", "SB", ")", ",", "NOSPLIT", ",", "$", "0", "-", "65"})
}
func TestDataAndGlobl(t *testing.T) {
eq(t, texts("GLOBL ·idx16(SB), RODATA, $64"),
[]string{"GLOBL", "·idx16", "(", "SB", ")", ",", "RODATA", ",", "$", "64"})
eq(t, texts("DATA ·idx16+0(SB)/4, $1"),
[]string{"DATA", "·idx16", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "1"})
}
func TestStaticSymbol(t *testing.T) {
// mask24<> is a file-local symbol; <> must lex as two angle tokens.
eq(t, texts("GLOBL mask24<>(SB), RODATA, $16"),
[]string{"GLOBL", "mask24", "<", ">", "(", "SB", ")", ",", "RODATA", ",", "$", "16"})
}
func TestNegativeImmediate(t *testing.T) {
eq(t, texts("ANDQ $-8, R10"),
[]string{"ANDQ", "$", "-", "8", ",", "R10"})
}
func TestHexImmediate(t *testing.T) {
eq(t, texts("DATA mask24<>+0(SB)/4, $0x80020100"),
[]string{"DATA", "mask24", "<", ">", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "0x80020100"})
}
func TestMemoryAddressing(t *testing.T) {
eq(t, texts("LEAQ (SI)(BX*4), R9"),
[]string{"LEAQ", "(", "SI", ")", "(", "BX", "*", "4", ")", ",", "R9"})
eq(t, texts("VMOVDQU32 Z0, 4(SI)(AX*1)"),
[]string{"VMOVDQU32", "Z0", ",", "4", "(", "SI", ")", "(", "AX", "*", "1", ")"})
}
func TestLabelAndComment(t *testing.T) {
eq(t, kinds("vec1:\n\tJMP vec1 // loop"),
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
}
func TestLineCommentTrailingWhitespace(t *testing.T) {
// A trailing run of CR, spaces and tabs is line-ending whitespace, not
// comment content. The token text must not depend on what follows the
// comment: before the trim covered only a CR directly before the token's
// end, "// loop\r " kept the CR while "// loop\r\n" dropped it, and the
// formatter re-lexed its own output to a shorter comment.
eq(t, texts("// loop\r"), []string{"// loop"})
eq(t, texts("// loop\r "), []string{"// loop"})
eq(t, texts("// loop \r\t\nMOVQ AX, BX"), []string{"// loop", "MOVQ", "AX", ",", "BX"})
// A CR inside the comment is content and stays.
eq(t, texts("// loops\rall"), []string{"// loops\rall"})
}
func TestAVX512Mnemonics(t *testing.T) {
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
eq(t, texts("KTESTW K1, K1"),
[]string{"KTESTW", "K1", ",", "K1"})
}
func TestArm64Shifts(t *testing.T) {
eq(t, texts("ADD R0<<2, R1, R2"),
[]string{"ADD", "R0", "<<", "2", ",", "R1", ",", "R2"})
eq(t, texts("MOVD R3->4, R5"),
[]string{"MOVD", "R3", "->", "4", ",", "R5"})
}
func TestInclude(t *testing.T) {
eq(t, texts(`#include "textflag.h"`),
[]string{"#", "include", `"textflag.h"`})
}
func TestPositions(t *testing.T) {
toks := Tokenize("MOVQ AX, BX\nRET")
// Find RET and check it landed on line 2.
var ret token.Token
for _, tok := range toks {
if tok.Text == "RET" {
ret = tok
}
}
if ret.Pos.Line != 2 || ret.Pos.Column != 1 {
t.Fatalf("RET position = %v, want 2:1", ret.Pos)
}
}
func TestIllegalNeverPanics(t *testing.T) {
// A stray backtick and NUL-ish garbage must not crash the scanner.
toks := Tokenize("MOVQ ` , \x01 AX")
if len(toks) == 0 {
t.Fatal("expected tokens")
}
}
func TestBlockComment(t *testing.T) {
eq(t, kinds("MOVQ /* inline */ AX"),
[]token.Kind{token.Ident, token.Comment, token.Ident})
// An unterminated block comment is tolerated.
toks := Tokenize("MOVQ /* never closed")
if toks[len(toks)-2].Kind != token.Comment {
t.Errorf("expected a comment token, got %v", toks)
}
}
func TestRuneLiteral(t *testing.T) {
eq(t, texts("MOVL $'a', AX"),
[]string{"MOVL", "$", "'a'", ",", "AX"})
}
func TestFloatAndBases(t *testing.T) {
eq(t, texts("$1.5"), []string{"$", "1.5"})
eq(t, texts("$0b1010"), []string{"$", "0b1010"})
eq(t, texts("$0o755"), []string{"$", "0o755"})
eq(t, texts("$1e3"), []string{"$", "1e3"})
}
func TestOperatorVariants(t *testing.T) {
eq(t, texts("R0>>2"), []string{"R0", ">>", "2"})
eq(t, texts("@>"), []string{"@", ">"})
eq(t, texts("a/b"), []string{"a", "/", "b"})
}
// TestPipeFlags covers the '|' that joins TEXT/GLOBL flag lists: it must scan
// as a token of its own so the formatter can preserve the bars the Go
// toolchain requires.
func TestPipeFlags(t *testing.T) {
eq(t, texts("TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0"),
[]string{"TEXT", "·f", "(", "SB", ")", ",", "NOSPLIT", "|", "NOFRAME", "|", "DUPOK", ",", "$", "0"})
}
// TestNulIsIllegal pins the difference between the end of input and a real
// NUL rune: the NUL must surface as an Illegal token and scanning must
// continue past it, so nothing after it is silently dropped.
func TestNulIsIllegal(t *testing.T) {
eq(t, texts("MOVQ \x00 AX"), []string{"MOVQ", "\x00", "AX"})
}
func TestDivisionSlashInIdentifiers(t *testing.T) {
// U+2215 DIVISION SLASH is an identifier character, the way the
// toolchain's tokenizer treats it: the package path of a symbol is
// written with it (internal∕runtime∕atomic·Xchg) and must lex as one
// name. The ordinary slash (U+002F) stays punctuation.
eq(t, texts("CALL internal∕runtime∕atomic·Xchg(SB)"),
[]string{"CALL", "internal∕runtime∕atomic·Xchg", "(", "SB", ")"})
eq(t, texts("MOVQ sync∕atomic·Align(SB), AX"),
[]string{"MOVQ", "sync∕atomic·Align", "(", "SB", ")", ",", "AX"})
// It may also begin a name, like any letter of the toolchain's rule.
eq(t, kinds("∕x"), []token.Kind{token.Ident})
}
// TestOffsetsAroundInvalidByte pins Position.Offset against the original
// bytes: an invalid UTF-8 byte decodes to RuneError but advances the offset
// table by exactly one byte, so every later position stays a true byte
// offset. Columns count runes, so the invalid byte occupies one column like
// any other character.
func TestOffsetsAroundInvalidByte(t *testing.T) {
// bytes: 'A'=0, ' '=1, 0xff=2, ' '=3, 'B'=4, '\n'=5, 'C'=6.
toks := Tokenize("A \xff B\nC")
want := []struct {
text string
off int
}{
{"A", 0}, {"\uFFFD", 2}, {"B", 4}, {"\n", 5}, {"C", 6},
}
if len(toks) != len(want)+1 || toks[len(toks)-1].Kind != token.EOF {
t.Fatalf("tokens = %v, want %v plus EOF", toks, want)
}
for i, w := range want {
if toks[i].Text != w.text || toks[i].Pos.Offset != w.off {
t.Errorf("token %d = %q@%d, want %q@%d", i, toks[i].Text, toks[i].Pos.Offset, w.text, w.off)
}
}
if got := toks[len(toks)-1].Pos.Offset; got != 7 {
t.Errorf("EOF offset = %d, want 7 (source length)", got)
}
if toks[1].Pos.Line != 1 || toks[1].Pos.Column != 3 {
t.Errorf("invalid byte position = %v, want 1:3", toks[1].Pos)
}
if toks[3].Kind != token.Newline || toks[3].Pos.Line != 1 || toks[3].Pos.Column != 6 {
t.Errorf("newline token = %v, want 1:6", toks[3])
}
if toks[4].Pos.Line != 2 || toks[4].Pos.Column != 1 {
t.Errorf("C position = %v, want 2:1", toks[4].Pos)
}
}