fix(lsp): parse-error survival, symbol ranges and UTF-16 positions

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-19 23:49:19 +02:00
parent eb8b0cd316
commit b3908fc43d
4 changed files with 547 additions and 57 deletions
+180 -54
View File
@@ -4,14 +4,17 @@
package lsp
import (
"cmp"
"fmt"
"os"
"path/filepath"
"regexp"
"runtime"
"sort"
"slices"
"strings"
"unicode"
"unicode/utf16"
"unicode/utf8"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -67,7 +70,7 @@ func (s *Server) completion(p completionParams) []CompletionItem {
items = append(items, CompletionItem{Label: name, Kind: ciModule, Detail: "local label"})
}
}
sort.Slice(items, func(i, j int) bool { return items[i].Label < items[j].Label })
slices.SortFunc(items, func(a, b CompletionItem) int { return cmp.Compare(a.Label, b.Label) })
return items
}
@@ -94,7 +97,7 @@ func (s *Server) hover(p hoverParams) *Hover {
}
return &Hover{
Contents: markupContent{Kind: "markdown", Value: md},
Range: rng,
Range: clientRange(text, rng),
}
}
@@ -112,7 +115,7 @@ func (s *Server) openASTs() []openAST {
for uri := range s.docs {
uris = append(uris, uri)
}
sort.Strings(uris)
slices.Sort(uris)
out := make([]openAST, 0, len(uris))
for _, uri := range uris {
if f, _ := parser.Parse(uriPath(uri), s.docs[uri]); f != nil {
@@ -143,11 +146,8 @@ func (s *Server) definition(p definitionParams) []Location {
if lbl, ok := stmt.(*ast.Label); ok {
if lbl.Name.Text == name || lbl.Name.Text == word {
return []Location{{
URI: p.TextDocument.URI,
Range: Range{
Start: Position{Line: lbl.Name.Pos.Line - 1, Character: lbl.Name.Pos.Column - 1},
End: Position{Line: lbl.Name.Pos.Line - 1, Character: lbl.Name.Pos.Column - 1 + len(word)},
},
URI: p.TextDocument.URI,
Range: clientRange(text, tokenRange(lbl.Name)),
}}
}
}
@@ -160,11 +160,11 @@ func (s *Server) definition(p definitionParams) []Location {
for _, of := range s.openASTs() {
for _, d := range of.file.Decls {
t, ok := d.(*ast.Text)
if !ok || t.Name == nil {
if !ok || !realSymbol(t.Name) {
continue
}
if t.Name.Name == name {
return []Location{{URI: of.uri, Range: symRange(t.Name)}}
return []Location{{URI: of.uri, Range: clientRange(s.docs[of.uri], symRange(t.Name))}}
}
}
}
@@ -190,18 +190,18 @@ func (s *Server) references(p referenceParams) []Location {
sameDoc := of.uri == uri
for _, d := range of.file.Decls {
t, ok := d.(*ast.Text)
if !ok || t.Name == nil {
if !ok || !realSymbol(t.Name) {
continue
}
// Include the definition if requested.
if p.Context.IncludeDeclaration && t.Name.Name == name {
out = append(out, Location{URI: of.uri, Range: symRange(t.Name)})
out = append(out, Location{URI: of.uri, Range: clientRange(s.docs[of.uri], symRange(t.Name))})
}
for _, stmt := range t.Body {
switch st := stmt.(type) {
case *ast.Label:
if sameDoc && st.Name.Text == name {
out = append(out, Location{URI: of.uri, Range: tokenRange(st.Name)})
out = append(out, Location{URI: of.uri, Range: clientRange(s.docs[of.uri], tokenRange(st.Name))})
}
case *ast.Instr:
for _, op := range st.Operands {
@@ -211,12 +211,15 @@ func (s *Server) references(p referenceParams) []Location {
if !sameDoc && op.Addr.Sym.Pseudo != "SB" {
continue
}
// The range covers the operand's verbatim
// identifier, `·`/package prefix included, so a
// rename replaces the whole spelling.
out = append(out, Location{
URI: of.uri,
Range: Range{
Range: clientRange(s.docs[of.uri], Range{
Start: Position{Line: op.Pos.Line - 1, Character: op.Pos.Column - 1},
End: Position{Line: op.Pos.Line - 1, Character: op.Pos.Column - 1 + runeLen(name)},
},
End: Position{Line: op.Pos.Line - 1, Character: op.Pos.Column - 1 + symIdentLen(op.Addr.Sym)},
}),
})
}
}
@@ -265,7 +268,7 @@ func (s *Server) documentFormatting(p documentFormattingParams) []TextEdit {
endLine := len(lines) - 1
endChar := 0
if endLine >= 0 {
endChar = len([]rune(lines[endLine]))
endChar = utf16Len(lines[endLine])
}
return []TextEdit{{
Range: Range{Start: Position{Line: 0, Character: 0}, End: Position{Line: endLine, Character: endChar}},
@@ -289,9 +292,10 @@ func (s *Server) inlayHints(p inlayHintParams) []InlayHint {
// Hint after the args size: show frame size.
if t.Frame != nil && t.Frame.Imm.HasVal && t.Args != nil && t.Args.Imm.HasVal {
// Place hint after the args operand using its raw text length.
col := t.Args.Pos.Column - 1 + len(t.Args.Raw)
line := t.Args.Pos.Line - 1
col := t.Args.Pos.Column - 1 + runeLen(t.Args.Raw)
out = append(out, InlayHint{
Position: Position{Line: t.Args.Pos.Line - 1, Character: col},
Position: Position{Line: line, Character: utf16Column(lineAt(text, line), col)},
Label: fmt.Sprintf(" frame=%d", t.Frame.Imm.Val),
Kind: inlayHintTypeParameter,
Tooltip: fmt.Sprintf("local frame size: %d bytes", t.Frame.Imm.Val),
@@ -341,7 +345,7 @@ func (s *Server) codeActions(p codeActionParams) []CodeAction {
newLines = append(newLines, lines[insertLine:]...)
newText := strings.Join(newLines, "\n")
endLine := len(lines) - 1
endChar := len([]rune(lines[endLine]))
endChar := utf16Len(lines[endLine])
actions = append(actions, CodeAction{
Title: "Add RET to " + t.Name.Name,
Kind: "quickfix",
@@ -367,7 +371,7 @@ func (s *Server) codeActions(p codeActionParams) []CodeAction {
newLines = append(newLines, lines[i+1:]...)
newText := strings.Join(newLines, "\n")
endLine := len(lines) - 1
endChar := len([]rune(lines[endLine]))
endChar := utf16Len(lines[endLine])
actions = append(actions, CodeAction{
Title: "Remove unused label \"" + label + "\"",
Kind: "quickfix",
@@ -458,6 +462,10 @@ func (s *Server) documentHighlights(p documentHighlightParams) []DocumentHighlig
if word == "" {
return nil
}
// Symbol names are stored without the middle dot, while the word under
// the cursor keeps it; trim once and compare the trimmed form, the way
// references and definition do.
name := strings.TrimPrefix(word, "\u00B7")
f, errs := parser.Parse(uriPath(p.TextDocument.URI), text)
if f == nil || len(errs) > 0 {
return nil
@@ -469,29 +477,29 @@ func (s *Server) documentHighlights(p documentHighlightParams) []DocumentHighlig
continue
}
// Highlight the definition.
if t.Name.Name == word {
if t.Name != nil && t.Name.Name == name {
out = append(out, DocumentHighlight{
Range: symRange(t.Name),
Range: clientRange(text, symRange(t.Name)),
Kind: highlightWrite,
})
}
for _, stmt := range t.Body {
switch st := stmt.(type) {
case *ast.Label:
if st.Name.Text == word {
if st.Name.Text == name {
out = append(out, DocumentHighlight{
Range: tokenRange(st.Name),
Range: clientRange(text, tokenRange(st.Name)),
Kind: highlightWrite,
})
}
case *ast.Instr:
for _, op := range st.Operands {
if op.Addr.Sym != nil && op.Addr.Sym.Name == word {
if op.Addr.Sym != nil && op.Addr.Sym.Name == name {
out = append(out, DocumentHighlight{
Range: Range{
Range: clientRange(text, Range{
Start: Position{Line: op.Pos.Line - 1, Character: op.Pos.Column - 1},
End: Position{Line: op.Pos.Line - 1, Character: op.Pos.Column - 1 + runeLen(word)},
},
End: Position{Line: op.Pos.Line - 1, Character: op.Pos.Column - 1 + symIdentLen(op.Addr.Sym)},
}),
Kind: highlightRead,
})
}
@@ -517,11 +525,17 @@ func (s *Server) workspaceSymbols(p workspaceSymbolParams) []WorkspaceSymbol {
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Text:
// A malformed TEXT line is kept in the tree under a "?"
// placeholder; it is not a symbol, and skipping it keeps
// the answer serving the rest of a mid-edit buffer.
if !realSymbol(dd.Name) {
continue
}
if strings.Contains(strings.ToLower(dd.Name.Name), query) {
out = append(out, WorkspaceSymbol{
Name: dd.Name.Name,
Kind: symFunction,
Location: Location{URI: uri, Range: symRange(dd.Name)},
Location: Location{URI: uri, Range: clientRange(text, symRange(dd.Name))},
})
}
case *ast.Globl:
@@ -529,7 +543,7 @@ func (s *Server) workspaceSymbols(p workspaceSymbolParams) []WorkspaceSymbol {
out = append(out, WorkspaceSymbol{
Name: dd.Name.Name,
Kind: symConstant,
Location: Location{URI: uri, Range: symRange(dd.Name)},
Location: Location{URI: uri, Range: clientRange(text, symRange(dd.Name))},
})
}
case *ast.Data:
@@ -537,7 +551,7 @@ func (s *Server) workspaceSymbols(p workspaceSymbolParams) []WorkspaceSymbol {
out = append(out, WorkspaceSymbol{
Name: dd.Name.Name,
Kind: symConstant,
Location: Location{URI: uri, Range: symRange(dd.Name)},
Location: Location{URI: uri, Range: clientRange(text, symRange(dd.Name))},
})
}
}
@@ -557,33 +571,44 @@ func (s *Server) documentSymbols(p documentSymbolParams) []DocumentSymbol {
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Text:
// As in workspaceSymbols: a "?" placeholder is not a symbol,
// and the remaining declarations are still listed.
if !realSymbol(dd.Name) {
continue
}
sym := DocumentSymbol{
Name: dd.Name.Name,
Detail: "TEXT " + strings.Join(dd.Flags, " "),
Kind: symFunction,
Range: textRange(dd),
SelectionRange: symRange(dd.Name),
Range: clientRange(text, textRange(dd)),
SelectionRange: clientRange(text, symRange(dd.Name)),
}
for _, st := range dd.Body {
if l, ok := st.(*ast.Label); ok {
sym.Children = append(sym.Children, DocumentSymbol{
Name: l.Name.Text,
Kind: symVariable,
Range: tokenRange(l.Name),
SelectionRange: tokenRange(l.Name),
Range: clientRange(text, tokenRange(l.Name)),
SelectionRange: clientRange(text, tokenRange(l.Name)),
})
}
}
out = append(out, sym)
case *ast.Globl:
if dd.Name == nil {
continue
}
out = append(out, DocumentSymbol{
Name: dd.Name.Name, Detail: "GLOBL", Kind: symConstant,
Range: symRange(dd.Name), SelectionRange: symRange(dd.Name),
Range: clientRange(text, symRange(dd.Name)), SelectionRange: clientRange(text, symRange(dd.Name)),
})
case *ast.Data:
if dd.Name == nil {
continue
}
out = append(out, DocumentSymbol{
Name: dd.Name.Name, Detail: "DATA", Kind: symConstant,
Range: symRange(dd.Name), SelectionRange: symRange(dd.Name),
Range: clientRange(text, symRange(dd.Name)), SelectionRange: clientRange(text, symRange(dd.Name)),
})
}
}
@@ -607,10 +632,17 @@ func (s *Server) semanticTokens(p semanticTokensParams) SemanticTokens {
toks := lexer.Tokenize(text)
lines := groupLines(toks)
srcLines := strings.Split(text, "\n")
var encoded []semTok
for _, line := range lines {
encoded = append(encoded, classifyLine(line, a, labels)...)
for _, st := range classifyLine(line, a, labels) {
// Columns cross the protocol boundary in UTF-16 code units.
if st.line >= 0 && st.line < len(srcLines) {
st.char = utf16Column(srcLines[st.line], st.char)
}
encoded = append(encoded, st)
}
}
return SemanticTokens{Data: deltaEncode(encoded)}
@@ -649,14 +681,15 @@ func classifyLine(line []token.Token, a *arch.Table, labels map[string]bool) []s
typ = classifyIdent(i, first, t.Text, a, labels, isDirective, isLabel, isInstr, &mnemonicDone)
case token.Colon, token.Comma, token.LParen, token.RParen,
token.Plus, token.Minus, token.Star, token.Slash, token.Dollar,
token.LAngle, token.RAngle, token.LShift, token.RShift, token.Arrow, token.At:
token.LAngle, token.RAngle, token.LShift, token.RShift, token.Arrow,
token.At, token.Pipe:
typ = stOperator
}
if typ >= 0 {
out = append(out, semTok{
line: t.Pos.Line - 1,
char: t.Pos.Column - 1,
length: runeLen(t.Text),
length: utf16Len(t.Text),
typ: typ,
})
}
@@ -714,14 +747,16 @@ func deltaEncode(toks []semTok) []int {
// --- shared helpers ---------------------------------------------------------
// wordAt extracts the identifier surrounding pos and its range.
// wordAt extracts the identifier surrounding pos and its range. The
// incoming character offset is UTF-16 code units, as LSP defines it, and is
// converted to the rune index the scanning works in.
func wordAt(text string, pos Position) (string, Range) {
lines := strings.Split(text, "\n")
if pos.Line < 0 || pos.Line >= len(lines) {
return "", Range{}
}
runes := []rune(lines[pos.Line])
col := pos.Character
col := runeColumn(lines[pos.Line], pos.Character)
if col < 0 || col > len(runes) {
return "", Range{}
}
@@ -802,14 +837,102 @@ func firstSignificant(line []token.Token) int {
func runeLen(s string) int { return len([]rune(s)) }
// symRange builds a range covering a symbol from its position and raw text.
// utf16Len returns the length of s in UTF-16 code units, the unit LSP
// positions count: an astral rune (an emoji in a comment) is two of them.
func utf16Len(s string) int {
n := 0
for _, r := range s {
n += utf16.RuneLen(r)
}
return n
}
// utf16Column converts a rune-based column on line into a UTF-16 code-unit
// offset. Rune columns are the lexer's convention; code units are the
// protocol's, and the two diverge once an astral rune precedes the column.
func utf16Column(line string, col int) int {
if col <= 0 {
return 0
}
units := 0
seen := 0
for _, r := range line {
if seen >= col {
break
}
units += utf16.RuneLen(r)
seen++
}
return units
}
// runeColumn converts a UTF-16 code-unit offset on line into a rune column,
// the inverse of utf16Column, applied to positions arriving from the client.
func runeColumn(line string, units int) int {
if units <= 0 {
return 0
}
col := 0
u := 0
for _, r := range line {
if u >= units {
break
}
u += utf16.RuneLen(r)
col++
}
return col
}
// lineAt returns the n-th zero-based line of text, or "" when out of range.
func lineAt(text string, n int) string {
if n < 0 {
return ""
}
lines := strings.Split(text, "\n")
if n >= len(lines) {
return ""
}
return lines[n]
}
// clientRange re-encodes a rune-based range (the columns the lexer, parser
// and the helpers above produce) in the UTF-16 code units LSP mandates.
// Every range leaving the server passes through here.
func clientRange(text string, r Range) Range {
return Range{
Start: Position{Line: r.Start.Line, Character: utf16Column(lineAt(text, r.Start.Line), r.Start.Character)},
End: Position{Line: r.End.Line, Character: utf16Column(lineAt(text, r.End.Line), r.End.Character)},
}
}
// symIdentLen returns the rune length of a symbol's identifier as written:
// the verbatim spelling up to the ABI marker, offset or pseudo-register
// group, so `pkg·name<ABIInternal>(SB)` counts the package prefix and the
// middle dot. A range built from it covers the whole token a rename
// replaces; the stripped Name alone would stop one character short.
func symIdentLen(sym *ast.Symbol) int {
if i := strings.IndexAny(sym.Raw, "<+-("); i >= 0 {
return runeLen(sym.Raw[:i])
}
return runeLen(sym.Raw)
}
// symRange builds a range covering a symbol as written, prefix included.
func symRange(sym *ast.Symbol) Range {
start := Position{Line: sym.Pos.Line - 1, Character: sym.Pos.Column - 1}
end := start
end.Character += runeLen(sym.Name)
end.Character += symIdentLen(sym)
return Range{Start: start, End: end}
}
// realSymbol reports whether sym names something a client can act on. The
// parser keeps a malformed TEXT in the tree under a "?" placeholder so the
// rest of the buffer stays servable; that placeholder is not a symbol.
func realSymbol(sym *ast.Symbol) bool {
return sym != nil && sym.Name != "?"
}
// tokenRange builds a range covering one token.
func tokenRange(t token.Token) Range {
return Range{
@@ -852,7 +975,7 @@ func (s *Server) diagnosticsFor(uri string) []Diagnostic {
pos = pe.Pos
}
out = append(out, Diagnostic{
Range: toRange(pos.Line, pos.Column, token.Position{}),
Range: clientRange(text, toRange(pos.Line, pos.Column, token.Position{})),
Severity: sevError,
Code: "syntax",
Source: "gasm",
@@ -861,7 +984,7 @@ func (s *Server) diagnosticsFor(uri string) []Diagnostic {
}
for _, d := range diags {
out = append(out, Diagnostic{
Range: toRange(d.Pos.Line, d.Pos.Column, d.End),
Range: clientRange(text, toRange(d.Pos.Line, d.Pos.Column, d.End)),
Severity: lintSeverity(d.Severity),
Code: d.Code,
Source: "gasm",
@@ -886,7 +1009,9 @@ func (s *Server) documentLinks(uri string) []DocumentLink {
goroot := runtime.GOROOT()
var out []DocumentLink
for i, line := range strings.Split(text, "\n") {
lineNo := -1
for line := range strings.SplitSeq(text, "\n") {
lineNo++
m := includeRe.FindStringSubmatch(line)
if m == nil {
continue
@@ -895,12 +1020,13 @@ func (s *Server) documentLinks(uri string) []DocumentLink {
if target == "" {
continue
}
start := strings.Index(line, "\"")
// strings.Index is a byte offset; columns are runes.
start := utf8.RuneCountInString(line[:strings.Index(line, "\"")])
out = append(out, DocumentLink{
Range: Range{
Start: Position{Line: i, Character: start},
End: Position{Line: i, Character: start + len(m[1]) + 2},
},
Range: clientRange(text, Range{
Start: Position{Line: lineNo, Character: start},
End: Position{Line: lineNo, Character: start + runeLen(m[1]) + 2},
}),
Target: "file://" + target,
})
}