// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: MIT package markdown import ( "bytes" "strings" ) // extractReferences strips leading link reference definitions from a // closed paragraph and records them in the document. The first definition // of a label wins. What remains of the paragraph keeps a single trailing // newline removed. func (p *parser) extractReferences(n *Node) { for { def, consumed, ok := parseReference(n.content) if !ok || consumed <= 0 { break } if _, exists := p.doc.refs[def.label]; !exists { p.doc.refs[def.label] = def.reference } n.content = n.content[consumed:] } if bytes.HasSuffix(n.content, []byte("\n")) { n.content = n.content[:len(n.content)-1] } } // refDef is a parsed link reference definition: the normalised label under // which it is recorded and the reference it defines. type refDef struct { label string reference } // parseReference parses one link reference definition from the start of // the paragraph content, possibly spanning lines. It returns the // definition, the number of bytes consumed and whether a definition was // there at all. func parseReference(c []byte) (refDef, int, bool) { // The label: brackets around one to 999 characters, no unescaped // bracket inside. if len(c) == 0 || c[0] != '[' { return refDef{}, 0, false } i := 1 labelEnd := -1 for i < len(c) { ch := c[i] if ch == '\\' && i+1 < len(c) { i += 2 continue } if ch == ']' { labelEnd = i break } i++ } if labelEnd < 0 { return refDef{}, 0, false } label := c[1:labelEnd] if len(label) < 1 || len(label) > 999 || len(bytes.TrimSpace(label)) == 0 || !validLabel(label) { return refDef{}, 0, false } i = labelEnd + 1 if i >= len(c) || c[i] != ':' { return refDef{}, 0, false } i++ // Up to one line ending may sit between the colon and the destination. i, _, ok := skipSpaceOneNewline(c, i) if !ok { return refDef{}, 0, false } dest, n, ok := scanDestination(c[i:]) if !ok { return refDef{}, 0, false } i += n // A definition without a title needs the rest of the destination's // line to be blank. titlelessEnd := -1 if lineEnd := bytes.IndexByte(c[i:], '\n'); lineEnd < 0 { if allSpaceTab(c[i:]) { titlelessEnd = len(c) } } else if allSpaceTab(c[i : i+lineEnd]) { titlelessEnd = i + lineEnd + 1 } // A title, when present, sits after at least one character of // whitespace, with at most one line ending between it and the // destination, and nothing but whitespace may follow it. if j, skipped, ok := skipSpaceOneNewline(c, i); ok && skipped > 0 && j < len(c) && (c[j] == '"' || c[j] == '\'' || c[j] == '(') { open := c[j] closer := open if open == '(' { closer = ')' } k := j + 1 for k < len(c) { ch := c[k] if ch == '\\' && k+1 < len(c) { k += 2 continue } if open == '(' && ch == '(' { break } if ch == closer { after := k + 1 for after < len(c) && isSpaceTab(c[after]) { after++ } if after >= len(c) { return def(label, dest, c[j+1:k], len(c)) } if c[after] == '\n' { return def(label, dest, c[j+1:k], after+1) } break } k++ } } if titlelessEnd < 0 { return refDef{}, 0, false } return def(label, dest, nil, titlelessEnd) } // def builds the result of a parsed definition. func def(label, dest, title []byte, consumed int) (refDef, int, bool) { r := reference{destination: unescapeText(string(dest))} if title != nil { r.title = unescapeText(string(title)) r.hasTitle = true } return refDef{label: normaliseLabel(string(label)), reference: r}, consumed, true } // skipSpaceOneNewline skips spaces, tabs and at most one newline, stopping // at the first other character or the end. It reports how much it skipped // and fails on a second line ending. func skipSpaceOneNewline(c []byte, i int) (int, int, bool) { start := i newlines := 0 for i < len(c) { switch c[i] { case ' ', '\t': i++ case '\n': newlines++ if newlines > 1 { return i, i - start, false } i++ default: return i, i - start, true } } return i, i - start, true } // scanDestination parses a link destination: a run in angle brackets with // no line ending inside, or a bare run without whitespace in which // parentheses stay balanced. func scanDestination(c []byte) ([]byte, int, bool) { if len(c) > 0 && c[0] == '<' { i := 1 for i < len(c) { ch := c[i] if ch == '\\' && i+1 < len(c) { i += 2 continue } if ch == '>' { return c[1:i], i + 1, true } if ch == '<' || ch == '\n' { return nil, 0, false } i++ } return nil, 0, false } i := 0 depth := 0 for i < len(c) { ch := c[i] if ch == '\\' && i+1 < len(c) { i += 2 continue } if ch == '(' { depth++ i++ continue } if ch == ')' { if depth == 0 { break } depth-- i++ continue } if isSpaceTab(ch) || ch == '\n' { break } i++ } if depth != 0 || i == 0 { return nil, 0, false } return c[:i], i, true } // validLabel reports whether the raw label text carries no unescaped open // bracket, which a link label may not contain. func validLabel(label []byte) bool { for i := 0; i < len(label); i++ { switch label[i] { case '\\': i++ case '[': return false } } return true } // labelFolder carries the case fold pairs the standard library's lower // casing does not perform, which the Unicode case fold CommonMark names // does. var labelFolder = strings.NewReplacer( "ß", "ss", "ff", "ff", "fi", "fi", "fl", "fl", "ffi", "ffi", "ffl", "ffl", "ſt", "st", "st", "st", ) // normaliseLabel brings a link label to the form definitions and uses are // compared under: surrounding and repeated whitespace collapsed to single // spaces, then case folded. func normaliseLabel(s string) string { return labelFolder.Replace(strings.ToLower(strings.Join(strings.Fields(s), " "))) }