254 lines
5.8 KiB
Go
254 lines
5.8 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|||
|
|
// SPDX-License-Identifier: MIT
|
||
|
|
|
||
|
|
package markdown
|
||
|
|
|
||
|
|
import (
|
||
|
|
"bytes"
|
||
|
|
"strings"
|
||
|
|
)
|
||
|
|
|
||
|
|
// extractReferences strips leading link reference definitions from a
|
||
|
|
// closed paragraph and records them in the document. The first definition
|
||
|
|
// of a label wins. What remains of the paragraph keeps a single trailing
|
||
|
|
// newline removed.
|
||
|
|
func (p *parser) extractReferences(n *Node) {
|
||
|
|
for {
|
||
|
|
def, consumed, ok := parseReference(n.content)
|
||
|
|
if !ok || consumed <= 0 {
|
||
|
|
break
|
||
|
|
}
|
||
|
|
if _, exists := p.doc.refs[def.label]; !exists {
|
||
|
|
p.doc.refs[def.label] = def.reference
|
||
|
|
}
|
||
|
|
n.content = n.content[consumed:]
|
||
|
|
}
|
||
|
|
if bytes.HasSuffix(n.content, []byte("\n")) {
|
||
|
|
n.content = n.content[:len(n.content)-1]
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
// refDef is a parsed link reference definition: the normalised label under
|
||
|
|
// which it is recorded and the reference it defines.
|
||
|
|
type refDef struct {
|
||
|
|
label string
|
||
|
|
reference
|
||
|
|
}
|
||
|
|
|
||
|
|
// parseReference parses one link reference definition from the start of
|
||
|
|
// the paragraph content, possibly spanning lines. It returns the
|
||
|
|
// definition, the number of bytes consumed and whether a definition was
|
||
|
|
// there at all.
|
||
|
|
func parseReference(c []byte) (refDef, int, bool) {
|
||
|
|
// The label: brackets around one to 999 characters, no unescaped
|
||
|
|
// bracket inside.
|
||
|
|
if len(c) == 0 || c[0] != '[' {
|
||
|
|
return refDef{}, 0, false
|
||
|
|
}
|
||
|
|
i := 1
|
||
|
|
labelEnd := -1
|
||
|
|
for i < len(c) {
|
||
|
|
ch := c[i]
|
||
|
|
if ch == '\\' && i+1 < len(c) {
|
||
|
|
i += 2
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
if ch == ']' {
|
||
|
|
labelEnd = i
|
||
|
|
break
|
||
|
|
}
|
||
|
|
i++
|
||
|
|
}
|
||
|
|
if labelEnd < 0 {
|
||
|
|
return refDef{}, 0, false
|
||
|
|
}
|
||
|
|
label := c[1:labelEnd]
|
||
|
|
if len(label) < 1 || len(label) > 999 || len(bytes.TrimSpace(label)) == 0 || !validLabel(label) {
|
||
|
|
return refDef{}, 0, false
|
||
|
|
}
|
||
|
|
i = labelEnd + 1
|
||
|
|
if i >= len(c) || c[i] != ':' {
|
||
|
|
return refDef{}, 0, false
|
||
|
|
}
|
||
|
|
i++
|
||
|
|
|
||
|
|
// Up to one line ending may sit between the colon and the destination.
|
||
|
|
i, _, ok := skipSpaceOneNewline(c, i)
|
||
|
|
if !ok {
|
||
|
|
return refDef{}, 0, false
|
||
|
|
}
|
||
|
|
dest, n, ok := scanDestination(c[i:])
|
||
|
|
if !ok {
|
||
|
|
return refDef{}, 0, false
|
||
|
|
}
|
||
|
|
i += n
|
||
|
|
|
||
|
|
// A definition without a title needs the rest of the destination's
|
||
|
|
// line to be blank.
|
||
|
|
titlelessEnd := -1
|
||
|
|
if lineEnd := bytes.IndexByte(c[i:], '\n'); lineEnd < 0 {
|
||
|
|
if allSpaceTab(c[i:]) {
|
||
|
|
titlelessEnd = len(c)
|
||
|
|
}
|
||
|
|
} else if allSpaceTab(c[i : i+lineEnd]) {
|
||
|
|
titlelessEnd = i + lineEnd + 1
|
||
|
|
}
|
||
|
|
|
||
|
|
// A title, when present, sits after at least one character of
|
||
|
|
// whitespace, with at most one line ending between it and the
|
||
|
|
// destination, and nothing but whitespace may follow it.
|
||
|
|
if j, skipped, ok := skipSpaceOneNewline(c, i); ok && skipped > 0 && j < len(c) && (c[j] == '"' || c[j] == '\'' || c[j] == '(') {
|
||
|
|
open := c[j]
|
||
|
|
closer := open
|
||
|
|
if open == '(' {
|
||
|
|
closer = ')'
|
||
|
|
}
|
||
|
|
k := j + 1
|
||
|
|
for k < len(c) {
|
||
|
|
ch := c[k]
|
||
|
|
if ch == '\\' && k+1 < len(c) {
|
||
|
|
k += 2
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
if open == '(' && ch == '(' {
|
||
|
|
break
|
||
|
|
}
|
||
|
|
if ch == closer {
|
||
|
|
after := k + 1
|
||
|
|
for after < len(c) && isSpaceTab(c[after]) {
|
||
|
|
after++
|
||
|
|
}
|
||
|
|
if after >= len(c) {
|
||
|
|
return def(label, dest, c[j+1:k], len(c))
|
||
|
|
}
|
||
|
|
if c[after] == '\n' {
|
||
|
|
return def(label, dest, c[j+1:k], after+1)
|
||
|
|
}
|
||
|
|
break
|
||
|
|
}
|
||
|
|
k++
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
if titlelessEnd < 0 {
|
||
|
|
return refDef{}, 0, false
|
||
|
|
}
|
||
|
|
return def(label, dest, nil, titlelessEnd)
|
||
|
|
}
|
||
|
|
|
||
|
|
// def builds the result of a parsed definition.
|
||
|
|
func def(label, dest, title []byte, consumed int) (refDef, int, bool) {
|
||
|
|
r := reference{destination: unescapeText(string(dest))}
|
||
|
|
if title != nil {
|
||
|
|
r.title = unescapeText(string(title))
|
||
|
|
r.hasTitle = true
|
||
|
|
}
|
||
|
|
return refDef{label: normaliseLabel(string(label)), reference: r}, consumed, true
|
||
|
|
}
|
||
|
|
|
||
|
|
// skipSpaceOneNewline skips spaces, tabs and at most one newline, stopping
|
||
|
|
// at the first other character or the end. It reports how much it skipped
|
||
|
|
// and fails on a second line ending.
|
||
|
|
func skipSpaceOneNewline(c []byte, i int) (int, int, bool) {
|
||
|
|
start := i
|
||
|
|
newlines := 0
|
||
|
|
for i < len(c) {
|
||
|
|
switch c[i] {
|
||
|
|
case ' ', '\t':
|
||
|
|
i++
|
||
|
|
case '\n':
|
||
|
|
newlines++
|
||
|
|
if newlines > 1 {
|
||
|
|
return i, i - start, false
|
||
|
|
}
|
||
|
|
i++
|
||
|
|
default:
|
||
|
|
return i, i - start, true
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return i, i - start, true
|
||
|
|
}
|
||
|
|
|
||
|
|
// scanDestination parses a link destination: a run in angle brackets with
|
||
|
|
// no line ending inside, or a bare run without whitespace in which
|
||
|
|
// parentheses stay balanced.
|
||
|
|
func scanDestination(c []byte) ([]byte, int, bool) {
|
||
|
|
if len(c) > 0 && c[0] == '<' {
|
||
|
|
i := 1
|
||
|
|
for i < len(c) {
|
||
|
|
ch := c[i]
|
||
|
|
if ch == '\\' && i+1 < len(c) {
|
||
|
|
i += 2
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
if ch == '>' {
|
||
|
|
return c[1:i], i + 1, true
|
||
|
|
}
|
||
|
|
if ch == '<' || ch == '\n' {
|
||
|
|
return nil, 0, false
|
||
|
|
}
|
||
|
|
i++
|
||
|
|
}
|
||
|
|
return nil, 0, false
|
||
|
|
}
|
||
|
|
i := 0
|
||
|
|
depth := 0
|
||
|
|
for i < len(c) {
|
||
|
|
ch := c[i]
|
||
|
|
if ch == '\\' && i+1 < len(c) {
|
||
|
|
i += 2
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
if ch == '(' {
|
||
|
|
depth++
|
||
|
|
i++
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
if ch == ')' {
|
||
|
|
if depth == 0 {
|
||
|
|
break
|
||
|
|
}
|
||
|
|
depth--
|
||
|
|
i++
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
if isSpaceTab(ch) || ch == '\n' {
|
||
|
|
break
|
||
|
|
}
|
||
|
|
i++
|
||
|
|
}
|
||
|
|
if depth != 0 || i == 0 {
|
||
|
|
return nil, 0, false
|
||
|
|
}
|
||
|
|
return c[:i], i, true
|
||
|
|
}
|
||
|
|
|
||
|
|
// validLabel reports whether the raw label text carries no unescaped open
|
||
|
|
// bracket, which a link label may not contain.
|
||
|
|
func validLabel(label []byte) bool {
|
||
|
|
for i := 0; i < len(label); i++ {
|
||
|
|
switch label[i] {
|
||
|
|
case '\\':
|
||
|
|
i++
|
||
|
|
case '[':
|
||
|
|
return false
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return true
|
||
|
|
}
|
||
|
|
|
||
|
|
// labelFolder carries the case fold pairs the standard library's lower
|
||
|
|
// casing does not perform, which the Unicode case fold CommonMark names
|
||
|
|
// does.
|
||
|
|
var labelFolder = strings.NewReplacer(
|
||
|
|
"ß", "ss", "ff", "ff", "fi", "fi", "fl", "fl",
|
||
|
|
"ffi", "ffi", "ffl", "ffl", "ſt", "st", "st", "st",
|
||
|
|
)
|
||
|
|
|
||
|
|
// normaliseLabel brings a link label to the form definitions and uses are
|
||
|
|
// compared under: surrounding and repeated whitespace collapsed to single
|
||
|
|
// spaces, then case folded.
|
||
|
|
func normaliseLabel(s string) string {
|
||
|
|
return labelFolder.Replace(strings.ToLower(strings.Join(strings.Fields(s), " ")))
|
||
|
|
}
|