fix(lexer): tokenise the flag separator and handle NUL and invalid UTF-8
Assisted-by: GLM 5.3
This commit is contained in:
@@ -6,6 +6,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
@@ -27,6 +28,12 @@ func TestStackGuardBytes(t *testing.T) {
|
|||||||
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
|
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
|
||||||
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
|
||||||
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
|
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
|
||||||
|
// Class 2 with a body long enough that the underflow JB relaxes to
|
||||||
|
// rel32: its displacement must span the real 6-byte JB, else the
|
||||||
|
// branch lands 4 bytes past the morestack block, inside the CALL
|
||||||
|
// displacement field.
|
||||||
|
{"leafbiglong", "TEXT \u00b7leafbiglong(SB), $8192-0\n" + strings.Repeat("\tMOVQ AX, BX\n", 40) + "\tRET\n",
|
||||||
|
"644c8b3425000000004989e44981ec881f00000f82960000004d3b66100f868c000000554889e54881ec00200000" + strings.Repeat("4889c3", 40) + "4881c4002000005dc3e800000000e947ffffff"},
|
||||||
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
|
||||||
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
|
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
|
||||||
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
|
||||||
|
|||||||
+44
-18
@@ -30,14 +30,22 @@ type Lexer struct {
|
|||||||
|
|
||||||
// New returns a Lexer over src.
|
// New returns a Lexer over src.
|
||||||
func New(src string) *Lexer {
|
func New(src string) *Lexer {
|
||||||
runes := []rune(src)
|
// Decode over the raw bytes rather than converting with []rune(src): a
|
||||||
off := make([]int, len(runes)+1)
|
// lone invalid byte converts to U+FFFD, whose RuneLen is three, and the
|
||||||
b := 0
|
// offset table would then count three bytes where the source has one,
|
||||||
for i, r := range runes {
|
// inflating every later Position.Offset against the original source.
|
||||||
off[i] = b
|
// Decoding advances by the true byte width (one for an invalid byte)
|
||||||
b += utf8.RuneLen(r)
|
// while the rune stream still carries RuneError, so token text keeps the
|
||||||
|
// replacement character.
|
||||||
|
runes := make([]rune, 0, len(src))
|
||||||
|
off := make([]int, 0, len(src)+1)
|
||||||
|
for b := 0; b < len(src); {
|
||||||
|
r, size := utf8.DecodeRuneInString(src[b:])
|
||||||
|
runes = append(runes, r)
|
||||||
|
off = append(off, b)
|
||||||
|
b += size
|
||||||
}
|
}
|
||||||
off[len(runes)] = b
|
off = append(off, len(src))
|
||||||
return &Lexer{src: runes, off: off, line: 1, col: 1}
|
return &Lexer{src: runes, off: off, line: 1, col: 1}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -55,9 +63,16 @@ func Tokenize(src string) []token.Token {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// cur returns the current rune, or 0 at end of input.
|
// atEnd reports whether the scanner sits past the last rune. Only the index
|
||||||
|
// decides: a literal NUL rune in the source is a real character, not the end
|
||||||
|
// of input, even though cur() returns 0 for both.
|
||||||
|
func (l *Lexer) atEnd() bool { return l.i >= len(l.src) }
|
||||||
|
|
||||||
|
// cur returns the current rune, or 0 at end of input. A real NUL rune in the
|
||||||
|
// source is indistinguishable here; callers that must tell them apart use
|
||||||
|
// atEnd.
|
||||||
func (l *Lexer) cur() rune {
|
func (l *Lexer) cur() rune {
|
||||||
if l.i >= len(l.src) {
|
if l.atEnd() {
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
return l.src[l.i]
|
return l.src[l.i]
|
||||||
@@ -128,9 +143,15 @@ func (l *Lexer) Next() token.Token {
|
|||||||
r := l.cur()
|
r := l.cur()
|
||||||
|
|
||||||
switch {
|
switch {
|
||||||
case r == 0:
|
case l.atEnd():
|
||||||
return l.make(token.EOF, start, "")
|
return l.make(token.EOF, start, "")
|
||||||
|
|
||||||
|
case r == 0:
|
||||||
|
// A real NUL rune (atEnd is false): fall through to punct, which
|
||||||
|
// emits it as an Illegal token and advances, so nothing after it
|
||||||
|
// is silently dropped.
|
||||||
|
return l.punct(start)
|
||||||
|
|
||||||
case r == '\n':
|
case r == '\n':
|
||||||
l.advance()
|
l.advance()
|
||||||
return l.make(token.Newline, start, "\n")
|
return l.make(token.Newline, start, "\n")
|
||||||
@@ -164,14 +185,16 @@ func (l *Lexer) Next() token.Token {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// lineComment consumes a // comment up to, but not including, the newline.
|
// lineComment consumes a // comment up to, but not including, the newline. A
|
||||||
|
// trailing \r is part of a CRLF line ending rather than comment content:
|
||||||
|
// dropping it keeps the formatter's output uniformly LF-terminated.
|
||||||
func (l *Lexer) lineComment(start token.Position) token.Token {
|
func (l *Lexer) lineComment(start token.Position) token.Token {
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
for l.cur() != 0 && l.cur() != '\n' {
|
for !l.atEnd() && l.cur() != '\n' {
|
||||||
b.WriteRune(l.cur())
|
b.WriteRune(l.cur())
|
||||||
l.advance()
|
l.advance()
|
||||||
}
|
}
|
||||||
return l.make(token.Comment, start, b.String())
|
return l.make(token.Comment, start, strings.TrimSuffix(b.String(), "\r"))
|
||||||
}
|
}
|
||||||
|
|
||||||
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
|
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
|
||||||
@@ -181,7 +204,7 @@ func (l *Lexer) blockComment(start token.Position) token.Token {
|
|||||||
l.advance()
|
l.advance()
|
||||||
b.WriteRune(l.cur()) // '*'
|
b.WriteRune(l.cur()) // '*'
|
||||||
l.advance()
|
l.advance()
|
||||||
for l.cur() != 0 {
|
for !l.atEnd() {
|
||||||
if l.cur() == '*' && l.peek(1) == '/' {
|
if l.cur() == '*' && l.peek(1) == '/' {
|
||||||
b.WriteString("*/")
|
b.WriteString("*/")
|
||||||
l.advance()
|
l.advance()
|
||||||
@@ -199,12 +222,12 @@ func (l *Lexer) string(start token.Position) token.Token {
|
|||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
b.WriteRune('"')
|
b.WriteRune('"')
|
||||||
l.advance() // opening quote
|
l.advance() // opening quote
|
||||||
for l.cur() != 0 && l.cur() != '\n' {
|
for !l.atEnd() && l.cur() != '\n' {
|
||||||
r := l.cur()
|
r := l.cur()
|
||||||
b.WriteRune(r)
|
b.WriteRune(r)
|
||||||
l.advance()
|
l.advance()
|
||||||
if r == '\\' {
|
if r == '\\' {
|
||||||
if l.cur() != 0 && l.cur() != '\n' {
|
if !l.atEnd() && l.cur() != '\n' {
|
||||||
b.WriteRune(l.cur())
|
b.WriteRune(l.cur())
|
||||||
l.advance()
|
l.advance()
|
||||||
}
|
}
|
||||||
@@ -223,12 +246,12 @@ func (l *Lexer) runeLit(start token.Position) token.Token {
|
|||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
b.WriteRune('\'')
|
b.WriteRune('\'')
|
||||||
l.advance() // opening quote
|
l.advance() // opening quote
|
||||||
for l.cur() != 0 && l.cur() != '\n' {
|
for !l.atEnd() && l.cur() != '\n' {
|
||||||
r := l.cur()
|
r := l.cur()
|
||||||
b.WriteRune(r)
|
b.WriteRune(r)
|
||||||
l.advance()
|
l.advance()
|
||||||
if r == '\\' {
|
if r == '\\' {
|
||||||
if l.cur() != 0 && l.cur() != '\n' {
|
if !l.atEnd() && l.cur() != '\n' {
|
||||||
b.WriteRune(l.cur())
|
b.WriteRune(l.cur())
|
||||||
l.advance()
|
l.advance()
|
||||||
}
|
}
|
||||||
@@ -373,6 +396,9 @@ func (l *Lexer) punct(start token.Position) token.Token {
|
|||||||
case '#':
|
case '#':
|
||||||
l.advance()
|
l.advance()
|
||||||
return l.make(token.Hash, start, "#")
|
return l.make(token.Hash, start, "#")
|
||||||
|
case '|':
|
||||||
|
l.advance()
|
||||||
|
return l.make(token.Pipe, start, "|")
|
||||||
default:
|
default:
|
||||||
// Unknown rune: emit it as Illegal and move on.
|
// Unknown rune: emit it as Illegal and move on.
|
||||||
l.advance()
|
l.advance()
|
||||||
|
|||||||
@@ -153,3 +153,54 @@ func TestOperatorVariants(t *testing.T) {
|
|||||||
eq(t, texts("@>"), []string{"@", ">"})
|
eq(t, texts("@>"), []string{"@", ">"})
|
||||||
eq(t, texts("a/b"), []string{"a", "/", "b"})
|
eq(t, texts("a/b"), []string{"a", "/", "b"})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestPipeFlags covers the '|' that joins TEXT/GLOBL flag lists: it must scan
|
||||||
|
// as a token of its own so the formatter can preserve the bars the Go
|
||||||
|
// toolchain requires.
|
||||||
|
func TestPipeFlags(t *testing.T) {
|
||||||
|
eq(t, texts("TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0"),
|
||||||
|
[]string{"TEXT", "·f", "(", "SB", ")", ",", "NOSPLIT", "|", "NOFRAME", "|", "DUPOK", ",", "$", "0"})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestNulIsIllegal pins the difference between the end of input and a real
|
||||||
|
// NUL rune: the NUL must surface as an Illegal token and scanning must
|
||||||
|
// continue past it, so nothing after it is silently dropped.
|
||||||
|
func TestNulIsIllegal(t *testing.T) {
|
||||||
|
eq(t, texts("MOVQ \x00 AX"), []string{"MOVQ", "\x00", "AX"})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestOffsetsAroundInvalidByte pins Position.Offset against the original
|
||||||
|
// bytes: an invalid UTF-8 byte decodes to RuneError but advances the offset
|
||||||
|
// table by exactly one byte, so every later position stays a true byte
|
||||||
|
// offset. Columns count runes, so the invalid byte occupies one column like
|
||||||
|
// any other character.
|
||||||
|
func TestOffsetsAroundInvalidByte(t *testing.T) {
|
||||||
|
// bytes: 'A'=0, ' '=1, 0xff=2, ' '=3, 'B'=4, '\n'=5, 'C'=6.
|
||||||
|
toks := Tokenize("A \xff B\nC")
|
||||||
|
want := []struct {
|
||||||
|
text string
|
||||||
|
off int
|
||||||
|
}{
|
||||||
|
{"A", 0}, {"\uFFFD", 2}, {"B", 4}, {"\n", 5}, {"C", 6},
|
||||||
|
}
|
||||||
|
if len(toks) != len(want)+1 || toks[len(toks)-1].Kind != token.EOF {
|
||||||
|
t.Fatalf("tokens = %v, want %v plus EOF", toks, want)
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
if toks[i].Text != w.text || toks[i].Pos.Offset != w.off {
|
||||||
|
t.Errorf("token %d = %q@%d, want %q@%d", i, toks[i].Text, toks[i].Pos.Offset, w.text, w.off)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if got := toks[len(toks)-1].Pos.Offset; got != 7 {
|
||||||
|
t.Errorf("EOF offset = %d, want 7 (source length)", got)
|
||||||
|
}
|
||||||
|
if toks[1].Pos.Line != 1 || toks[1].Pos.Column != 3 {
|
||||||
|
t.Errorf("invalid byte position = %v, want 1:3", toks[1].Pos)
|
||||||
|
}
|
||||||
|
if toks[3].Kind != token.Newline || toks[3].Pos.Line != 1 || toks[3].Pos.Column != 6 {
|
||||||
|
t.Errorf("newline token = %v, want 1:6", toks[3])
|
||||||
|
}
|
||||||
|
if toks[4].Pos.Line != 2 || toks[4].Pos.Column != 1 {
|
||||||
|
t.Errorf("C position = %v, want 2:1", toks[4].Pos)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -42,6 +42,7 @@ const (
|
|||||||
Arrow // ->
|
Arrow // ->
|
||||||
At // @
|
At // @
|
||||||
Hash // #
|
Hash // #
|
||||||
|
Pipe // |
|
||||||
)
|
)
|
||||||
|
|
||||||
var kindNames = map[Kind]string{
|
var kindNames = map[Kind]string{
|
||||||
@@ -69,6 +70,7 @@ var kindNames = map[Kind]string{
|
|||||||
Arrow: "->",
|
Arrow: "->",
|
||||||
At: "@",
|
At: "@",
|
||||||
Hash: "#",
|
Hash: "#",
|
||||||
|
Pipe: "|",
|
||||||
}
|
}
|
||||||
|
|
||||||
// String returns a human-readable name for the kind.
|
// String returns a human-readable name for the kind.
|
||||||
|
|||||||
@@ -13,6 +13,7 @@ func TestKindString(t *testing.T) {
|
|||||||
LParen: "(",
|
LParen: "(",
|
||||||
LShift: "<<",
|
LShift: "<<",
|
||||||
Arrow: "->",
|
Arrow: "->",
|
||||||
|
Pipe: "|",
|
||||||
Illegal: "ILLEGAL",
|
Illegal: "ILLEGAL",
|
||||||
}
|
}
|
||||||
for k, want := range cases {
|
for k, want := range cases {
|
||||||
|
|||||||
Reference in New Issue
Block a user