diff --git a/asm/guard_test.go b/asm/guard_test.go index 3109d9e..f84fb5d 100644 --- a/asm/guard_test.go +++ b/asm/guard_test.go @@ -6,6 +6,7 @@ package asm import ( "bytes" "encoding/hex" + "strings" "testing" "sourcedock.dev/petrbalvin/gasm-devkit/ast" @@ -27,6 +28,12 @@ func TestStackGuardBytes(t *testing.T) { "644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"}, {"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n", "644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"}, + // Class 2 with a body long enough that the underflow JB relaxes to + // rel32: its displacement must span the real 6-byte JB, else the + // branch lands 4 bytes past the morestack block, inside the CALL + // displacement field. + {"leafbiglong", "TEXT \u00b7leafbiglong(SB), $8192-0\n" + strings.Repeat("\tMOVQ AX, BX\n", 40) + "\tRET\n", + "644c8b3425000000004989e44981ec881f00000f82960000004d3b66100f868c000000554889e54881ec00200000" + strings.Repeat("4889c3", 40) + "4881c4002000005dc3e800000000e947ffffff"}, {"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n", "644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"}, {"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n", diff --git a/lexer/lexer.go b/lexer/lexer.go index 4a55990..e6b5f2a 100644 --- a/lexer/lexer.go +++ b/lexer/lexer.go @@ -30,14 +30,22 @@ type Lexer struct { // New returns a Lexer over src. func New(src string) *Lexer { - runes := []rune(src) - off := make([]int, len(runes)+1) - b := 0 - for i, r := range runes { - off[i] = b - b += utf8.RuneLen(r) + // Decode over the raw bytes rather than converting with []rune(src): a + // lone invalid byte converts to U+FFFD, whose RuneLen is three, and the + // offset table would then count three bytes where the source has one, + // inflating every later Position.Offset against the original source. + // Decoding advances by the true byte width (one for an invalid byte) + // while the rune stream still carries RuneError, so token text keeps the + // replacement character. + runes := make([]rune, 0, len(src)) + off := make([]int, 0, len(src)+1) + for b := 0; b < len(src); { + r, size := utf8.DecodeRuneInString(src[b:]) + runes = append(runes, r) + off = append(off, b) + b += size } - off[len(runes)] = b + off = append(off, len(src)) return &Lexer{src: runes, off: off, line: 1, col: 1} } @@ -55,9 +63,16 @@ func Tokenize(src string) []token.Token { } } -// cur returns the current rune, or 0 at end of input. +// atEnd reports whether the scanner sits past the last rune. Only the index +// decides: a literal NUL rune in the source is a real character, not the end +// of input, even though cur() returns 0 for both. +func (l *Lexer) atEnd() bool { return l.i >= len(l.src) } + +// cur returns the current rune, or 0 at end of input. A real NUL rune in the +// source is indistinguishable here; callers that must tell them apart use +// atEnd. func (l *Lexer) cur() rune { - if l.i >= len(l.src) { + if l.atEnd() { return 0 } return l.src[l.i] @@ -128,9 +143,15 @@ func (l *Lexer) Next() token.Token { r := l.cur() switch { - case r == 0: + case l.atEnd(): return l.make(token.EOF, start, "") + case r == 0: + // A real NUL rune (atEnd is false): fall through to punct, which + // emits it as an Illegal token and advances, so nothing after it + // is silently dropped. + return l.punct(start) + case r == '\n': l.advance() return l.make(token.Newline, start, "\n") @@ -164,14 +185,16 @@ func (l *Lexer) Next() token.Token { } } -// lineComment consumes a // comment up to, but not including, the newline. +// lineComment consumes a // comment up to, but not including, the newline. A +// trailing \r is part of a CRLF line ending rather than comment content: +// dropping it keeps the formatter's output uniformly LF-terminated. func (l *Lexer) lineComment(start token.Position) token.Token { var b strings.Builder - for l.cur() != 0 && l.cur() != '\n' { + for !l.atEnd() && l.cur() != '\n' { b.WriteRune(l.cur()) l.advance() } - return l.make(token.Comment, start, b.String()) + return l.make(token.Comment, start, strings.TrimSuffix(b.String(), "\r")) } // blockComment consumes a /* ... */ comment, tolerating an unterminated one. @@ -181,7 +204,7 @@ func (l *Lexer) blockComment(start token.Position) token.Token { l.advance() b.WriteRune(l.cur()) // '*' l.advance() - for l.cur() != 0 { + for !l.atEnd() { if l.cur() == '*' && l.peek(1) == '/' { b.WriteString("*/") l.advance() @@ -199,12 +222,12 @@ func (l *Lexer) string(start token.Position) token.Token { var b strings.Builder b.WriteRune('"') l.advance() // opening quote - for l.cur() != 0 && l.cur() != '\n' { + for !l.atEnd() && l.cur() != '\n' { r := l.cur() b.WriteRune(r) l.advance() if r == '\\' { - if l.cur() != 0 && l.cur() != '\n' { + if !l.atEnd() && l.cur() != '\n' { b.WriteRune(l.cur()) l.advance() } @@ -223,12 +246,12 @@ func (l *Lexer) runeLit(start token.Position) token.Token { var b strings.Builder b.WriteRune('\'') l.advance() // opening quote - for l.cur() != 0 && l.cur() != '\n' { + for !l.atEnd() && l.cur() != '\n' { r := l.cur() b.WriteRune(r) l.advance() if r == '\\' { - if l.cur() != 0 && l.cur() != '\n' { + if !l.atEnd() && l.cur() != '\n' { b.WriteRune(l.cur()) l.advance() } @@ -373,6 +396,9 @@ func (l *Lexer) punct(start token.Position) token.Token { case '#': l.advance() return l.make(token.Hash, start, "#") + case '|': + l.advance() + return l.make(token.Pipe, start, "|") default: // Unknown rune: emit it as Illegal and move on. l.advance() diff --git a/lexer/lexer_test.go b/lexer/lexer_test.go index e5ae6c8..8aa7928 100644 --- a/lexer/lexer_test.go +++ b/lexer/lexer_test.go @@ -153,3 +153,54 @@ func TestOperatorVariants(t *testing.T) { eq(t, texts("@>"), []string{"@", ">"}) eq(t, texts("a/b"), []string{"a", "/", "b"}) } + +// TestPipeFlags covers the '|' that joins TEXT/GLOBL flag lists: it must scan +// as a token of its own so the formatter can preserve the bars the Go +// toolchain requires. +func TestPipeFlags(t *testing.T) { + eq(t, texts("TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0"), + []string{"TEXT", "·f", "(", "SB", ")", ",", "NOSPLIT", "|", "NOFRAME", "|", "DUPOK", ",", "$", "0"}) +} + +// TestNulIsIllegal pins the difference between the end of input and a real +// NUL rune: the NUL must surface as an Illegal token and scanning must +// continue past it, so nothing after it is silently dropped. +func TestNulIsIllegal(t *testing.T) { + eq(t, texts("MOVQ \x00 AX"), []string{"MOVQ", "\x00", "AX"}) +} + +// TestOffsetsAroundInvalidByte pins Position.Offset against the original +// bytes: an invalid UTF-8 byte decodes to RuneError but advances the offset +// table by exactly one byte, so every later position stays a true byte +// offset. Columns count runes, so the invalid byte occupies one column like +// any other character. +func TestOffsetsAroundInvalidByte(t *testing.T) { + // bytes: 'A'=0, ' '=1, 0xff=2, ' '=3, 'B'=4, '\n'=5, 'C'=6. + toks := Tokenize("A \xff B\nC") + want := []struct { + text string + off int + }{ + {"A", 0}, {"\uFFFD", 2}, {"B", 4}, {"\n", 5}, {"C", 6}, + } + if len(toks) != len(want)+1 || toks[len(toks)-1].Kind != token.EOF { + t.Fatalf("tokens = %v, want %v plus EOF", toks, want) + } + for i, w := range want { + if toks[i].Text != w.text || toks[i].Pos.Offset != w.off { + t.Errorf("token %d = %q@%d, want %q@%d", i, toks[i].Text, toks[i].Pos.Offset, w.text, w.off) + } + } + if got := toks[len(toks)-1].Pos.Offset; got != 7 { + t.Errorf("EOF offset = %d, want 7 (source length)", got) + } + if toks[1].Pos.Line != 1 || toks[1].Pos.Column != 3 { + t.Errorf("invalid byte position = %v, want 1:3", toks[1].Pos) + } + if toks[3].Kind != token.Newline || toks[3].Pos.Line != 1 || toks[3].Pos.Column != 6 { + t.Errorf("newline token = %v, want 1:6", toks[3]) + } + if toks[4].Pos.Line != 2 || toks[4].Pos.Column != 1 { + t.Errorf("C position = %v, want 2:1", toks[4].Pos) + } +} diff --git a/token/token.go b/token/token.go index 98bb933..5de01c5 100644 --- a/token/token.go +++ b/token/token.go @@ -42,6 +42,7 @@ const ( Arrow // -> At // @ Hash // # + Pipe // | ) var kindNames = map[Kind]string{ @@ -69,6 +70,7 @@ var kindNames = map[Kind]string{ Arrow: "->", At: "@", Hash: "#", + Pipe: "|", } // String returns a human-readable name for the kind. diff --git a/token/token_test.go b/token/token_test.go index a40b6f0..4430d15 100644 --- a/token/token_test.go +++ b/token/token_test.go @@ -13,6 +13,7 @@ func TestKindString(t *testing.T) { LParen: "(", LShift: "<<", Arrow: "->", + Pipe: "|", Illegal: "ILLEGAL", } for k, want := range cases {