Files

254 lines
5.8 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package markdown
import (
"bytes"
"strings"
)
// extractReferences strips leading link reference definitions from a
// closed paragraph and records them in the document. The first definition
// of a label wins. What remains of the paragraph keeps a single trailing
// newline removed.
func (p *parser) extractReferences(n *Node) {
for {
def, consumed, ok := parseReference(n.content)
if !ok || consumed <= 0 {
break
}
if _, exists := p.doc.refs[def.label]; !exists {
p.doc.refs[def.label] = def.reference
}
n.content = n.content[consumed:]
}
if bytes.HasSuffix(n.content, []byte("\n")) {
n.content = n.content[:len(n.content)-1]
}
}
// refDef is a parsed link reference definition: the normalised label under
// which it is recorded and the reference it defines.
type refDef struct {
label string
reference
}
// parseReference parses one link reference definition from the start of
// the paragraph content, possibly spanning lines. It returns the
// definition, the number of bytes consumed and whether a definition was
// there at all.
func parseReference(c []byte) (refDef, int, bool) {
// The label: brackets around one to 999 characters, no unescaped
// bracket inside.
if len(c) == 0 || c[0] != '[' {
return refDef{}, 0, false
}
i := 1
labelEnd := -1
for i < len(c) {
ch := c[i]
if ch == '\\' && i+1 < len(c) {
i += 2
continue
}
if ch == ']' {
labelEnd = i
break
}
i++
}
if labelEnd < 0 {
return refDef{}, 0, false
}
label := c[1:labelEnd]
if len(label) < 1 || len(label) > 999 || len(bytes.TrimSpace(label)) == 0 || !validLabel(label) {
return refDef{}, 0, false
}
i = labelEnd + 1
if i >= len(c) || c[i] != ':' {
return refDef{}, 0, false
}
i++
// Up to one line ending may sit between the colon and the destination.
i, _, ok := skipSpaceOneNewline(c, i)
if !ok {
return refDef{}, 0, false
}
dest, n, ok := scanDestination(c[i:])
if !ok {
return refDef{}, 0, false
}
i += n
// A definition without a title needs the rest of the destination's
// line to be blank.
titlelessEnd := -1
if lineEnd := bytes.IndexByte(c[i:], '\n'); lineEnd < 0 {
if allSpaceTab(c[i:]) {
titlelessEnd = len(c)
}
} else if allSpaceTab(c[i : i+lineEnd]) {
titlelessEnd = i + lineEnd + 1
}
// A title, when present, sits after at least one character of
// whitespace, with at most one line ending between it and the
// destination, and nothing but whitespace may follow it.
if j, skipped, ok := skipSpaceOneNewline(c, i); ok && skipped > 0 && j < len(c) && (c[j] == '"' || c[j] == '\'' || c[j] == '(') {
open := c[j]
closer := open
if open == '(' {
closer = ')'
}
k := j + 1
for k < len(c) {
ch := c[k]
if ch == '\\' && k+1 < len(c) {
k += 2
continue
}
if open == '(' && ch == '(' {
break
}
if ch == closer {
after := k + 1
for after < len(c) && isSpaceTab(c[after]) {
after++
}
if after >= len(c) {
return def(label, dest, c[j+1:k], len(c))
}
if c[after] == '\n' {
return def(label, dest, c[j+1:k], after+1)
}
break
}
k++
}
}
if titlelessEnd < 0 {
return refDef{}, 0, false
}
return def(label, dest, nil, titlelessEnd)
}
// def builds the result of a parsed definition.
func def(label, dest, title []byte, consumed int) (refDef, int, bool) {
r := reference{destination: unescapeText(string(dest))}
if title != nil {
r.title = unescapeText(string(title))
r.hasTitle = true
}
return refDef{label: normaliseLabel(string(label)), reference: r}, consumed, true
}
// skipSpaceOneNewline skips spaces, tabs and at most one newline, stopping
// at the first other character or the end. It reports how much it skipped
// and fails on a second line ending.
func skipSpaceOneNewline(c []byte, i int) (int, int, bool) {
start := i
newlines := 0
for i < len(c) {
switch c[i] {
case ' ', '\t':
i++
case '\n':
newlines++
if newlines > 1 {
return i, i - start, false
}
i++
default:
return i, i - start, true
}
}
return i, i - start, true
}
// scanDestination parses a link destination: a run in angle brackets with
// no line ending inside, or a bare run without whitespace in which
// parentheses stay balanced.
func scanDestination(c []byte) ([]byte, int, bool) {
if len(c) > 0 && c[0] == '<' {
i := 1
for i < len(c) {
ch := c[i]
if ch == '\\' && i+1 < len(c) {
i += 2
continue
}
if ch == '>' {
return c[1:i], i + 1, true
}
if ch == '<' || ch == '\n' {
return nil, 0, false
}
i++
}
return nil, 0, false
}
i := 0
depth := 0
for i < len(c) {
ch := c[i]
if ch == '\\' && i+1 < len(c) {
i += 2
continue
}
if ch == '(' {
depth++
i++
continue
}
if ch == ')' {
if depth == 0 {
break
}
depth--
i++
continue
}
if isSpaceTab(ch) || ch == '\n' {
break
}
i++
}
if depth != 0 || i == 0 {
return nil, 0, false
}
return c[:i], i, true
}
// validLabel reports whether the raw label text carries no unescaped open
// bracket, which a link label may not contain.
func validLabel(label []byte) bool {
for i := 0; i < len(label); i++ {
switch label[i] {
case '\\':
i++
case '[':
return false
}
}
return true
}
// labelFolder carries the case fold pairs the standard library's lower
// casing does not perform, which the Unicode case fold CommonMark names
// does.
var labelFolder = strings.NewReplacer(
"ß", "ss", "ff", "ff", "fi", "fi", "fl", "fl",
"ffi", "ffi", "ffl", "ffl", "ſt", "st", "st", "st",
)
// normaliseLabel brings a link label to the form definitions and uses are
// compared under: surrounding and repeated whitespace collapsed to single
// spaces, then case folded.
func normaliseLabel(s string) string {
return labelFolder.Replace(strings.ToLower(strings.Join(strings.Fields(s), " ")))
}