Files
gasm-sdk/lexer/lexer_test.go
T
2026-09-26 11:08:43 +02:00

233 lines
7.8 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lexer
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/token"
)
// kinds tokenizes src and returns the kind sequence, dropping Newline/EOF.
func kinds(src string) []token.Kind {
var out []token.Kind
for _, t := range Tokenize(src) {
if t.Kind == token.Newline || t.Kind == token.EOF {
continue
}
out = append(out, t.Kind)
}
return out
}
// texts tokenizes src and returns the literal text of each significant token.
func texts(src string) []string {
var out []string
for _, t := range Tokenize(src) {
if t.Kind == token.Newline || t.Kind == token.EOF {
continue
}
out = append(out, t.Text)
}
return out
}
func eq[T comparable](t *testing.T, got, want []T) {
t.Helper()
if len(got) != len(want) {
t.Fatalf("length mismatch:\n got %v\n want %v", got, want)
}
for i := range got {
if got[i] != want[i] {
t.Fatalf("index %d:\n got %v\n want %v", i, got, want)
}
}
}
func TestTextDirective(t *testing.T) {
eq(t, texts("TEXT ·analyzeO1RangeAVX2(SB), NOSPLIT, $0-65"),
[]string{"TEXT", "·analyzeO1RangeAVX2", "(", "SB", ")", ",", "NOSPLIT", ",", "$", "0", "-", "65"})
}
func TestDataAndGlobl(t *testing.T) {
eq(t, texts("GLOBL ·idx16(SB), RODATA, $64"),
[]string{"GLOBL", "·idx16", "(", "SB", ")", ",", "RODATA", ",", "$", "64"})
eq(t, texts("DATA ·idx16+0(SB)/4, $1"),
[]string{"DATA", "·idx16", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "1"})
}
func TestStaticSymbol(t *testing.T) {
// mask24<> is a file-local symbol; <> must lex as two angle tokens.
eq(t, texts("GLOBL mask24<>(SB), RODATA, $16"),
[]string{"GLOBL", "mask24", "<", ">", "(", "SB", ")", ",", "RODATA", ",", "$", "16"})
}
func TestNegativeImmediate(t *testing.T) {
eq(t, texts("ANDQ $-8, R10"),
[]string{"ANDQ", "$", "-", "8", ",", "R10"})
}
func TestHexImmediate(t *testing.T) {
eq(t, texts("DATA mask24<>+0(SB)/4, $0x80020100"),
[]string{"DATA", "mask24", "<", ">", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "0x80020100"})
}
func TestMemoryAddressing(t *testing.T) {
eq(t, texts("LEAQ (SI)(BX*4), R9"),
[]string{"LEAQ", "(", "SI", ")", "(", "BX", "*", "4", ")", ",", "R9"})
eq(t, texts("VMOVDQU32 Z0, 4(SI)(AX*1)"),
[]string{"VMOVDQU32", "Z0", ",", "4", "(", "SI", ")", "(", "AX", "*", "1", ")"})
}
func TestLabelAndComment(t *testing.T) {
eq(t, kinds("vec1:\n\tJMP vec1 // loop"),
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
}
func TestLineCommentTrailingWhitespace(t *testing.T) {
// A trailing run of CR, spaces and tabs is line-ending whitespace, not
// comment content. The token text must not depend on what follows the
// comment: before the trim covered only a CR directly before the token's
// end, "// loop\r " kept the CR while "// loop\r\n" dropped it, and the
// formatter re-lexed its own output to a shorter comment.
eq(t, texts("// loop\r"), []string{"// loop"})
eq(t, texts("// loop\r "), []string{"// loop"})
eq(t, texts("// loop \r\t\nMOVQ AX, BX"), []string{"// loop", "MOVQ", "AX", ",", "BX"})
// A CR inside the comment is content and stays.
eq(t, texts("// loops\rall"), []string{"// loops\rall"})
}
func TestAVX512Mnemonics(t *testing.T) {
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
eq(t, texts("KTESTW K1, K1"),
[]string{"KTESTW", "K1", ",", "K1"})
}
func TestArm64Shifts(t *testing.T) {
eq(t, texts("ADD R0<<2, R1, R2"),
[]string{"ADD", "R0", "<<", "2", ",", "R1", ",", "R2"})
eq(t, texts("MOVD R3->4, R5"),
[]string{"MOVD", "R3", "->", "4", ",", "R5"})
}
func TestInclude(t *testing.T) {
eq(t, texts(`#include "textflag.h"`),
[]string{"#", "include", `"textflag.h"`})
}
func TestPositions(t *testing.T) {
toks := Tokenize("MOVQ AX, BX\nRET")
// Find RET and check it landed on line 2.
var ret token.Token
for _, tok := range toks {
if tok.Text == "RET" {
ret = tok
}
}
if ret.Pos.Line != 2 || ret.Pos.Column != 1 {
t.Fatalf("RET position = %v, want 2:1", ret.Pos)
}
}
func TestIllegalNeverPanics(t *testing.T) {
// A stray backtick and NUL-ish garbage must not crash the scanner.
toks := Tokenize("MOVQ ` , \x01 AX")
if len(toks) == 0 {
t.Fatal("expected tokens")
}
}
func TestBlockComment(t *testing.T) {
eq(t, kinds("MOVQ /* inline */ AX"),
[]token.Kind{token.Ident, token.Comment, token.Ident})
// An unterminated block comment is tolerated.
toks := Tokenize("MOVQ /* never closed")
if toks[len(toks)-2].Kind != token.Comment {
t.Errorf("expected a comment token, got %v", toks)
}
}
func TestRuneLiteral(t *testing.T) {
eq(t, texts("MOVL $'a', AX"),
[]string{"MOVL", "$", "'a'", ",", "AX"})
}
func TestFloatAndBases(t *testing.T) {
eq(t, texts("$1.5"), []string{"$", "1.5"})
eq(t, texts("$0b1010"), []string{"$", "0b1010"})
eq(t, texts("$0o755"), []string{"$", "0o755"})
eq(t, texts("$1e3"), []string{"$", "1e3"})
}
func TestOperatorVariants(t *testing.T) {
eq(t, texts("R0>>2"), []string{"R0", ">>", "2"})
eq(t, texts("@>"), []string{"@", ">"})
eq(t, texts("a/b"), []string{"a", "/", "b"})
}
// TestPipeFlags covers the '|' that joins TEXT/GLOBL flag lists: it must scan
// as a token of its own so the formatter can preserve the bars the Go
// toolchain requires.
func TestPipeFlags(t *testing.T) {
eq(t, texts("TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0"),
[]string{"TEXT", "·f", "(", "SB", ")", ",", "NOSPLIT", "|", "NOFRAME", "|", "DUPOK", ",", "$", "0"})
}
// TestNulIsIllegal pins the difference between the end of input and a real
// NUL rune: the NUL must surface as an Illegal token and scanning must
// continue past it, so nothing after it is silently dropped.
func TestNulIsIllegal(t *testing.T) {
eq(t, texts("MOVQ \x00 AX"), []string{"MOVQ", "\x00", "AX"})
}
func TestDivisionSlashInIdentifiers(t *testing.T) {
// U+2215 DIVISION SLASH is an identifier character, the way the
// toolchain's tokenizer treats it: the package path of a symbol is
// written with it (internal∕runtime∕atomic·Xchg) and must lex as one
// name. The ordinary slash (U+002F) stays punctuation.
eq(t, texts("CALL internal∕runtime∕atomic·Xchg(SB)"),
[]string{"CALL", "internal∕runtime∕atomic·Xchg", "(", "SB", ")"})
eq(t, texts("MOVQ sync∕atomic·Align(SB), AX"),
[]string{"MOVQ", "sync∕atomic·Align", "(", "SB", ")", ",", "AX"})
// It may also begin a name, like any letter of the toolchain's rule.
eq(t, kinds("∕x"), []token.Kind{token.Ident})
}
// TestOffsetsAroundInvalidByte pins Position.Offset against the original
// bytes: an invalid UTF-8 byte decodes to RuneError but advances the offset
// table by exactly one byte, so every later position stays a true byte
// offset. Columns count runes, so the invalid byte occupies one column like
// any other character.
func TestOffsetsAroundInvalidByte(t *testing.T) {
// bytes: 'A'=0, ' '=1, 0xff=2, ' '=3, 'B'=4, '\n'=5, 'C'=6.
toks := Tokenize("A \xff B\nC")
want := []struct {
text string
off int
}{
{"A", 0}, {"\uFFFD", 2}, {"B", 4}, {"\n", 5}, {"C", 6},
}
if len(toks) != len(want)+1 || toks[len(toks)-1].Kind != token.EOF {
t.Fatalf("tokens = %v, want %v plus EOF", toks, want)
}
for i, w := range want {
if toks[i].Text != w.text || toks[i].Pos.Offset != w.off {
t.Errorf("token %d = %q@%d, want %q@%d", i, toks[i].Text, toks[i].Pos.Offset, w.text, w.off)
}
}
if got := toks[len(toks)-1].Pos.Offset; got != 7 {
t.Errorf("EOF offset = %d, want 7 (source length)", got)
}
if toks[1].Pos.Line != 1 || toks[1].Pos.Column != 3 {
t.Errorf("invalid byte position = %v, want 1:3", toks[1].Pos)
}
if toks[3].Kind != token.Newline || toks[3].Pos.Line != 1 || toks[3].Pos.Column != 6 {
t.Errorf("newline token = %v, want 1:6", toks[3])
}
if toks[4].Pos.Line != 2 || toks[4].Pos.Column != 1 {
t.Errorf("C position = %v, want 2:1", toks[4].Pos)
}
}