233 lines
7.8 KiB
Go
233 lines
7.8 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||
// SPDX-License-Identifier: BSD-3-Clause
|
||
|
||
package lexer
|
||
|
||
import (
|
||
"testing"
|
||
|
||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||
)
|
||
|
||
// kinds tokenizes src and returns the kind sequence, dropping Newline/EOF.
|
||
func kinds(src string) []token.Kind {
|
||
var out []token.Kind
|
||
for _, t := range Tokenize(src) {
|
||
if t.Kind == token.Newline || t.Kind == token.EOF {
|
||
continue
|
||
}
|
||
out = append(out, t.Kind)
|
||
}
|
||
return out
|
||
}
|
||
|
||
// texts tokenizes src and returns the literal text of each significant token.
|
||
func texts(src string) []string {
|
||
var out []string
|
||
for _, t := range Tokenize(src) {
|
||
if t.Kind == token.Newline || t.Kind == token.EOF {
|
||
continue
|
||
}
|
||
out = append(out, t.Text)
|
||
}
|
||
return out
|
||
}
|
||
|
||
func eq[T comparable](t *testing.T, got, want []T) {
|
||
t.Helper()
|
||
if len(got) != len(want) {
|
||
t.Fatalf("length mismatch:\n got %v\n want %v", got, want)
|
||
}
|
||
for i := range got {
|
||
if got[i] != want[i] {
|
||
t.Fatalf("index %d:\n got %v\n want %v", i, got, want)
|
||
}
|
||
}
|
||
}
|
||
|
||
func TestTextDirective(t *testing.T) {
|
||
eq(t, texts("TEXT ·analyzeO1RangeAVX2(SB), NOSPLIT, $0-65"),
|
||
[]string{"TEXT", "·analyzeO1RangeAVX2", "(", "SB", ")", ",", "NOSPLIT", ",", "$", "0", "-", "65"})
|
||
}
|
||
|
||
func TestDataAndGlobl(t *testing.T) {
|
||
eq(t, texts("GLOBL ·idx16(SB), RODATA, $64"),
|
||
[]string{"GLOBL", "·idx16", "(", "SB", ")", ",", "RODATA", ",", "$", "64"})
|
||
eq(t, texts("DATA ·idx16+0(SB)/4, $1"),
|
||
[]string{"DATA", "·idx16", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "1"})
|
||
}
|
||
|
||
func TestStaticSymbol(t *testing.T) {
|
||
// mask24<> is a file-local symbol; <> must lex as two angle tokens.
|
||
eq(t, texts("GLOBL mask24<>(SB), RODATA, $16"),
|
||
[]string{"GLOBL", "mask24", "<", ">", "(", "SB", ")", ",", "RODATA", ",", "$", "16"})
|
||
}
|
||
|
||
func TestNegativeImmediate(t *testing.T) {
|
||
eq(t, texts("ANDQ $-8, R10"),
|
||
[]string{"ANDQ", "$", "-", "8", ",", "R10"})
|
||
}
|
||
|
||
func TestHexImmediate(t *testing.T) {
|
||
eq(t, texts("DATA mask24<>+0(SB)/4, $0x80020100"),
|
||
[]string{"DATA", "mask24", "<", ">", "+", "0", "(", "SB", ")", "/", "4", ",", "$", "0x80020100"})
|
||
}
|
||
|
||
func TestMemoryAddressing(t *testing.T) {
|
||
eq(t, texts("LEAQ (SI)(BX*4), R9"),
|
||
[]string{"LEAQ", "(", "SI", ")", "(", "BX", "*", "4", ")", ",", "R9"})
|
||
eq(t, texts("VMOVDQU32 Z0, 4(SI)(AX*1)"),
|
||
[]string{"VMOVDQU32", "Z0", ",", "4", "(", "SI", ")", "(", "AX", "*", "1", ")"})
|
||
}
|
||
|
||
func TestLabelAndComment(t *testing.T) {
|
||
eq(t, kinds("vec1:\n\tJMP vec1 // loop"),
|
||
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
|
||
}
|
||
|
||
func TestLineCommentTrailingWhitespace(t *testing.T) {
|
||
// A trailing run of CR, spaces and tabs is line-ending whitespace, not
|
||
// comment content. The token text must not depend on what follows the
|
||
// comment: before the trim covered only a CR directly before the token's
|
||
// end, "// loop\r " kept the CR while "// loop\r\n" dropped it, and the
|
||
// formatter re-lexed its own output to a shorter comment.
|
||
eq(t, texts("// loop\r"), []string{"// loop"})
|
||
eq(t, texts("// loop\r "), []string{"// loop"})
|
||
eq(t, texts("// loop \r\t\nMOVQ AX, BX"), []string{"// loop", "MOVQ", "AX", ",", "BX"})
|
||
// A CR inside the comment is content and stays.
|
||
eq(t, texts("// loops\rall"), []string{"// loops\rall"})
|
||
}
|
||
|
||
func TestAVX512Mnemonics(t *testing.T) {
|
||
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
|
||
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
|
||
eq(t, texts("KTESTW K1, K1"),
|
||
[]string{"KTESTW", "K1", ",", "K1"})
|
||
}
|
||
|
||
func TestArm64Shifts(t *testing.T) {
|
||
eq(t, texts("ADD R0<<2, R1, R2"),
|
||
[]string{"ADD", "R0", "<<", "2", ",", "R1", ",", "R2"})
|
||
eq(t, texts("MOVD R3->4, R5"),
|
||
[]string{"MOVD", "R3", "->", "4", ",", "R5"})
|
||
}
|
||
|
||
func TestInclude(t *testing.T) {
|
||
eq(t, texts(`#include "textflag.h"`),
|
||
[]string{"#", "include", `"textflag.h"`})
|
||
}
|
||
|
||
func TestPositions(t *testing.T) {
|
||
toks := Tokenize("MOVQ AX, BX\nRET")
|
||
// Find RET and check it landed on line 2.
|
||
var ret token.Token
|
||
for _, tok := range toks {
|
||
if tok.Text == "RET" {
|
||
ret = tok
|
||
}
|
||
}
|
||
if ret.Pos.Line != 2 || ret.Pos.Column != 1 {
|
||
t.Fatalf("RET position = %v, want 2:1", ret.Pos)
|
||
}
|
||
}
|
||
|
||
func TestIllegalNeverPanics(t *testing.T) {
|
||
// A stray backtick and NUL-ish garbage must not crash the scanner.
|
||
toks := Tokenize("MOVQ ` , \x01 AX")
|
||
if len(toks) == 0 {
|
||
t.Fatal("expected tokens")
|
||
}
|
||
}
|
||
|
||
func TestBlockComment(t *testing.T) {
|
||
eq(t, kinds("MOVQ /* inline */ AX"),
|
||
[]token.Kind{token.Ident, token.Comment, token.Ident})
|
||
// An unterminated block comment is tolerated.
|
||
toks := Tokenize("MOVQ /* never closed")
|
||
if toks[len(toks)-2].Kind != token.Comment {
|
||
t.Errorf("expected a comment token, got %v", toks)
|
||
}
|
||
}
|
||
|
||
func TestRuneLiteral(t *testing.T) {
|
||
eq(t, texts("MOVL $'a', AX"),
|
||
[]string{"MOVL", "$", "'a'", ",", "AX"})
|
||
}
|
||
|
||
func TestFloatAndBases(t *testing.T) {
|
||
eq(t, texts("$1.5"), []string{"$", "1.5"})
|
||
eq(t, texts("$0b1010"), []string{"$", "0b1010"})
|
||
eq(t, texts("$0o755"), []string{"$", "0o755"})
|
||
eq(t, texts("$1e3"), []string{"$", "1e3"})
|
||
}
|
||
|
||
func TestOperatorVariants(t *testing.T) {
|
||
eq(t, texts("R0>>2"), []string{"R0", ">>", "2"})
|
||
eq(t, texts("@>"), []string{"@", ">"})
|
||
eq(t, texts("a/b"), []string{"a", "/", "b"})
|
||
}
|
||
|
||
// TestPipeFlags covers the '|' that joins TEXT/GLOBL flag lists: it must scan
|
||
// as a token of its own so the formatter can preserve the bars the Go
|
||
// toolchain requires.
|
||
func TestPipeFlags(t *testing.T) {
|
||
eq(t, texts("TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0"),
|
||
[]string{"TEXT", "·f", "(", "SB", ")", ",", "NOSPLIT", "|", "NOFRAME", "|", "DUPOK", ",", "$", "0"})
|
||
}
|
||
|
||
// TestNulIsIllegal pins the difference between the end of input and a real
|
||
// NUL rune: the NUL must surface as an Illegal token and scanning must
|
||
// continue past it, so nothing after it is silently dropped.
|
||
func TestNulIsIllegal(t *testing.T) {
|
||
eq(t, texts("MOVQ \x00 AX"), []string{"MOVQ", "\x00", "AX"})
|
||
}
|
||
|
||
func TestDivisionSlashInIdentifiers(t *testing.T) {
|
||
// U+2215 DIVISION SLASH is an identifier character, the way the
|
||
// toolchain's tokenizer treats it: the package path of a symbol is
|
||
// written with it (internal∕runtime∕atomic·Xchg) and must lex as one
|
||
// name. The ordinary slash (U+002F) stays punctuation.
|
||
eq(t, texts("CALL internal∕runtime∕atomic·Xchg(SB)"),
|
||
[]string{"CALL", "internal∕runtime∕atomic·Xchg", "(", "SB", ")"})
|
||
eq(t, texts("MOVQ sync∕atomic·Align(SB), AX"),
|
||
[]string{"MOVQ", "sync∕atomic·Align", "(", "SB", ")", ",", "AX"})
|
||
// It may also begin a name, like any letter of the toolchain's rule.
|
||
eq(t, kinds("∕x"), []token.Kind{token.Ident})
|
||
}
|
||
|
||
// TestOffsetsAroundInvalidByte pins Position.Offset against the original
|
||
// bytes: an invalid UTF-8 byte decodes to RuneError but advances the offset
|
||
// table by exactly one byte, so every later position stays a true byte
|
||
// offset. Columns count runes, so the invalid byte occupies one column like
|
||
// any other character.
|
||
func TestOffsetsAroundInvalidByte(t *testing.T) {
|
||
// bytes: 'A'=0, ' '=1, 0xff=2, ' '=3, 'B'=4, '\n'=5, 'C'=6.
|
||
toks := Tokenize("A \xff B\nC")
|
||
want := []struct {
|
||
text string
|
||
off int
|
||
}{
|
||
{"A", 0}, {"\uFFFD", 2}, {"B", 4}, {"\n", 5}, {"C", 6},
|
||
}
|
||
if len(toks) != len(want)+1 || toks[len(toks)-1].Kind != token.EOF {
|
||
t.Fatalf("tokens = %v, want %v plus EOF", toks, want)
|
||
}
|
||
for i, w := range want {
|
||
if toks[i].Text != w.text || toks[i].Pos.Offset != w.off {
|
||
t.Errorf("token %d = %q@%d, want %q@%d", i, toks[i].Text, toks[i].Pos.Offset, w.text, w.off)
|
||
}
|
||
}
|
||
if got := toks[len(toks)-1].Pos.Offset; got != 7 {
|
||
t.Errorf("EOF offset = %d, want 7 (source length)", got)
|
||
}
|
||
if toks[1].Pos.Line != 1 || toks[1].Pos.Column != 3 {
|
||
t.Errorf("invalid byte position = %v, want 1:3", toks[1].Pos)
|
||
}
|
||
if toks[3].Kind != token.Newline || toks[3].Pos.Line != 1 || toks[3].Pos.Column != 6 {
|
||
t.Errorf("newline token = %v, want 1:6", toks[3])
|
||
}
|
||
if toks[4].Pos.Line != 2 || toks[4].Pos.Column != 1 {
|
||
t.Errorf("C position = %v, want 2:1", toks[4].Pos)
|
||
}
|
||
}
|