Files
volumen/internal/markdown/markdown.go
T
petrbalvin f8ed33df83
Test / test (push) Successful in 7m5s
Release / gates (push) Successful in 7m28s
Release / build (amd64, freebsd) (push) Successful in 2m52s
Release / build (amd64, linux) (push) Successful in 2m46s
Release / build (arm64, freebsd) (push) Successful in 2m22s
Release / build (arm64, linux) (push) Successful in 2m38s
Release / build (loong64, linux) (push) Successful in 2m7s
Release / build (riscv64, linux) (push) Successful in 2m17s
Release / release (push) Successful in 1m0s
Initial commit
Assisted-by: GLM 5.3
2026-09-29 10:03:32 +02:00

342 lines
12 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: PolyForm-Noncommercial-1.0.0
// Package markdown renders Markdown posts to sanitised HTML.
//
// scriptorium renders the body (CommonMark with the GFM extensions,
// footnotes and definition lists). Mathematics and Mermaid diagrams, which
// scriptorium renders only when a consumer asks, are recognised here: a
// pre-render scan lifts $$…$$ and $…$ runs out of the source into
// placeholders and splices the MathML back after rendering, and fenced
// mermaid blocks are replaced by their SVG. Images titled with a caption
// are wrapped in <figure>/<figcaption>, headings gain id attributes and a
// table of contents, and bluemonday strips everything outside a narrow
// tag/attribute allowlist.
package markdown
import (
"fmt"
stdhtml "html"
"regexp"
"strconv"
"strings"
"sync"
"github.com/microcosm-cc/bluemonday"
"sourcedock.dev/petrbalvin/scriptorium"
)
// MaxBodyLength caps the Markdown source size (1 MiB).
const MaxBodyLength = 1_048_576
// MaxQuoteDepth bounds how deeply one line may nest blockquote markers.
// Measured: rendering cost grows superlinearly with depth (100 000
// levels take seconds, and a 1 MiB body of nothing but markers could
// reach hundreds of thousands), while legitimate prose never approaches
// the bound. It turns the worst case from an unbounded CPU burn into a
// rejection.
const MaxQuoteDepth = 100
var (
policyOnce sync.Once
sanitizer *bluemonday.Policy
)
// tocHeading is one entry of the rendered table of contents.
type tocHeading struct {
level int
id string
text string
}
func getPolicy() *bluemonday.Policy {
policyOnce.Do(func() {
sanitizer = bluemonday.NewPolicy()
sanitizer.AllowElements(
"a", "abbr", "blockquote", "br", "caption", "code", "del",
"dd", "div", "dl", "dt", "em", "figcaption", "figure",
"h1", "h2", "h3", "h4", "h5", "h6", "hr", "img", "input",
"li", "ol", "p", "pre", "section", "span", "strong", "sub", "sup",
"table", "tbody", "td", "th", "thead", "tr", "ul",
// SVG, the output of the Mermaid diagram renderer.
"svg", "line", "path", "polygon", "rect", "text",
// MathML Core, the output of the mathematics renderer.
"math", "mi", "mn", "mo", "ms", "mtext", "mspace", "mrow",
"mfrac", "msqrt", "mroot", "msub", "msup", "msubsup",
"munder", "mover", "munderover", "merror", "mpadded",
"mphantom", "mstyle", "mtable", "mtr", "mtd",
)
// MathML leaves most elements bare: an <mi> carries no attribute
// at all, and without this the policy admits only elements that
// do.
sanitizer.AllowNoAttrs().OnElements(
"math", "mi", "mn", "mo", "ms", "mtext", "mspace", "mrow",
"mfrac", "msqrt", "mroot", "msub", "msup", "msubsup",
"munder", "mover", "munderover", "merror", "mpadded",
"mphantom", "mstyle", "mtable", "mtr", "mtd",
)
sanitizer.AllowAttrs("href", "title", "class", "id", "aria-label", "data-footnote-ref").OnElements("a")
sanitizer.AllowAttrs("src", "alt", "title", "width", "height", "loading").OnElements("img")
sanitizer.AllowAttrs("class").OnElements("div", "span", "code", "pre", "figure", "figcaption", "section", "li", "sup")
sanitizer.AllowAttrs("id").OnElements("h1", "h2", "h3", "h4", "h5", "h6", "section", "li")
sanitizer.AllowAttrs("align").OnElements("th", "td")
sanitizer.AllowAttrs("type", "checked", "disabled").OnElements("input")
sanitizer.AllowAttrs("display", "xmlns").OnElements("math")
sanitizer.AllowAttrs(
"mathvariant", "stretchy", "accent", "accentunder", "separator",
"form", "fence", "lspace", "rspace", "mathcolor",
).OnElements("mi", "mn", "mo", "ms", "mtext")
sanitizer.AllowAttrs("width").OnElements("mspace")
sanitizer.AllowAttrs("displaystyle", "scriptlevel", "mathcolor").OnElements("mstyle")
sanitizer.AllowAttrs("columnalign").OnElements("mtable")
sanitizer.AllowAttrs("data-footnotes").OnElements("section")
// The attributes of the diagram SVG: geometry and paint, the
// shapes the renderer draws and the styles a classDef or a
// linkStyle statement asks for.
sanitizer.AllowAttrs("xmlns", "viewBox", "width", "height", "font-family").OnElements("svg")
sanitizer.AllowAttrs("x", "y", "width", "height", "rx").OnElements("rect")
sanitizer.AllowAttrs("x1", "y1", "x2", "y2").OnElements("line")
sanitizer.AllowAttrs("d").OnElements("path")
sanitizer.AllowAttrs("x", "y", "text-anchor").OnElements("text")
sanitizer.AllowAttrs("points").OnElements("polygon")
sanitizer.AllowAttrs(
"fill", "stroke", "stroke-width", "stroke-dasharray",
"font-size", "opacity", "style",
).OnElements("svg", "line", "path", "polygon", "rect", "text")
sanitizer.AllowURLSchemes("http", "https", "mailto")
sanitizer.AllowRelativeURLs(true)
})
return sanitizer
}
// Render renders Markdown to sanitised HTML.
func Render(text string) (string, error) {
out, _, err := RenderWithTOC(text)
return out, err
}
// RenderWithTOC renders Markdown to sanitised HTML and returns
// (html, toc_html). toc_html is the sanitised table-of-contents markup,
// or an empty string when the body has no headings.
func RenderWithTOC(src string) (string, string, error) {
if src == "" {
return "", "", nil
}
if len(src) > MaxBodyLength {
return "", "", fmt.Errorf("body exceeds %d bytes", MaxBodyLength)
}
if quoteDepthTooDeep(src) {
return "", "", fmt.Errorf("body nests blockquotes deeper than %d levels", MaxQuoteDepth)
}
rewritten, spans := extractMath(src)
html := string(scriptorium.Render([]byte(rewritten)))
html = spliceMath(html, spans)
html, toc := addHeadingIDs(html)
raw := unwrapFigureParagraphs(wrapFigures(html))
out := sanitize(raw)
// The diagrams are drawn after sanitisation: the SVG is scriptorium's
// own output, not authored markup, and the HTML policy's parser
// rewrites the case-sensitive viewBox attribute into a form no browser
// reads. A diagram the library refuses keeps its code block, which the
// sanitiser above has already cleaned like every other one.
out = renderDiagrams(out)
return out, sanitize(toc), nil
}
// quoteDepthTooDeep reports whether any line nests blockquote markers
// beyond MaxQuoteDepth. The markers may be written with or without
// spaces between them, so both spellings are counted.
func quoteDepthTooDeep(src string) bool {
for line := range strings.SplitSeq(src, "\n") {
rest := strings.TrimLeft(line, " \t")
depth := 0
for strings.HasPrefix(rest, ">") {
depth++
if depth > MaxQuoteDepth {
return true
}
rest = strings.TrimLeft(rest[1:], " \t")
}
}
return false
}
var figureParaRe = regexp.MustCompile(`(?s)<p>(<figure>.*?</figure>)</p>`)
// unwrapFigureParagraphs lifts figures out of their wrapping paragraph
// (<p> cannot contain <figure>).
func unwrapFigureParagraphs(in string) string {
return figureParaRe.ReplaceAllString(in, "<p></p>$1<p></p>")
}
func sanitize(in string) string {
if in == "" {
return in
}
out := getPolicy().Sanitize(in)
return addLinkRel(out)
}
// linkTagRe matches anchor start tags in sanitised output.
var linkTagRe = regexp.MustCompile(`<a\s[^>]*>`)
// addLinkRel forces rel="noopener noreferrer" onto every link, matching
// the shape consumers expect.
func addLinkRel(in string) string {
return linkTagRe.ReplaceAllStringFunc(in, func(tag string) string {
if strings.Contains(tag, "rel=") {
return tag
}
return `<a rel="noopener noreferrer" ` + strings.TrimPrefix(tag, "<a ")
})
}
var (
imgTitleRe = regexp.MustCompile(`(?is)<img\s[^>]*\stitle=(?:"[^"]*"|'[^']*')[^>]*/?>`)
titleAttrRe = regexp.MustCompile(`(?is)\s+title=(?:"[^"]*"|'[^']*')`)
titleValRe = regexp.MustCompile(`(?is)^<img\s[^>]*\stitle=(?:"([^"]*)"|'([^']*)')`)
)
// wrapFigures wraps every <img title="…"> in a <figure> with a
// <figcaption>, on the raw pre-sanitisation output so the title
// attribute is still present.
func wrapFigures(in string) string {
return imgTitleRe.ReplaceAllStringFunc(in, func(imgTag string) string {
title := ""
if m := titleValRe.FindStringSubmatch(imgTag); m != nil {
title = m[1]
if title == "" {
title = m[2]
}
}
// The captured attribute value is HTML-escaped (the renderer
// escapes what it writes), so it is unescaped first: escaping it
// again would show "&amp;amp;" for a title containing "&".
// Unescaping leaves a raw-HTML title the author wrote verbatim,
// and the re-escape puts both forms back in canonical shape.
caption := stdhtml.EscapeString(stdhtml.UnescapeString(title))
cleanImg := titleAttrRe.ReplaceAllString(imgTag, "")
return "<figure>" +
cleanImg +
`<div class="fig-info" aria-hidden="true">i</div>` +
"<figcaption>" + caption + "</figcaption>" +
"</figure>"
})
}
// headingTagRe matches the headings the renderer writes: scriptorium
// emits them without attributes, so the pass below is the only source of
// their id attributes. RE2 has no backreference, so the func below
// checks that the opening and closing levels agree.
var headingTagRe = regexp.MustCompile(`(?s)<h([1-6])>(.*?)</h([1-6])>`)
// innerTagRe strips the inline markup of a heading, leaving its text.
var innerTagRe = regexp.MustCompile(`(?s)<[^>]*>`)
// addHeadingIDs gives every heading an id attribute derived from its
// text and returns the table of contents built from the same walk.
func addHeadingIDs(in string) (string, string) {
used := make(map[string]int)
var headings []tocHeading
out := headingTagRe.ReplaceAllStringFunc(in, func(m string) string {
sub := headingTagRe.FindStringSubmatch(m)
level, inner, closeLevel := sub[1], sub[2], sub[3]
if level != closeLevel {
return m
}
label := strings.TrimSpace(stdhtml.UnescapeString(innerTagRe.ReplaceAllString(inner, "")))
if label == "" {
return m
}
id := headingSlug(label, used)
levelNum, _ := strconv.Atoi(level)
headings = append(headings, tocHeading{level: levelNum, id: id, text: label})
return "<h" + level + ` id="` + id + `">` + inner + "</h" + level + ">"
})
return out, renderTOC(headings)
}
// headingSlug turns a heading label into a unique id by the rule the
// previous renderer established, so the anchors the published pages
// already carry keep resolving: ASCII letters and digits, lowercased;
// spaces, dashes and underscores as dashes; every other byte, the
// diacritics of Czech prose included, dropped. Collisions are told
// apart by a numeric suffix.
func headingSlug(label string, used map[string]int) string {
var b strings.Builder
for i := 0; i < len(label); i++ {
c := label[i]
switch {
case c >= 0x80:
// A multi-byte rune is dropped whole.
continue
case 'A' <= c && c <= 'Z':
b.WriteByte(c + 'a' - 'A')
case 'a' <= c && c <= 'z' || '0' <= c && c <= '9':
b.WriteByte(c)
case c == ' ' || c == '\t' || c == '-' || c == '_':
b.WriteByte('-')
}
}
id := b.String()
if id == "" {
id = "heading"
}
if n, ok := used[id]; ok {
used[id] = n + 1
id = id + "-" + strconv.Itoa(n)
}
used[id] = 1
return id
}
// renderTOC renders the headings as a table of contents in the shape the
// API documents:
// <div class="toc"><ul><li><a href="#id">Title</a></li></ul></div>.
// The wrapper is emitted even with an empty list, so a consumer can rely
// on its presence.
func renderTOC(headings []tocHeading) string {
var b strings.Builder
b.WriteString(`<div class="toc">` + "\n")
if len(headings) == 0 {
b.WriteString("<ul></ul>\n")
} else {
b.WriteString(renderTOCList(headings))
}
b.WriteString("</div>\n")
return b.String()
}
// renderTOCList renders the headings as nested lists. A run of deeper
// headings becomes a sub-list of the heading above it, and a heading
// that returns to a shallower level closes the lists it left behind and
// continues as a sibling, so a body that opens with a second-level
// heading and later uses a first-level one keeps both in the list.
func renderTOCList(headings []tocHeading) string {
var b strings.Builder
b.WriteString("<ul>\n")
for i := 0; i < len(headings); i++ {
h := headings[i]
b.WriteString(`<li><a href="#` + stdhtml.EscapeString(h.id) + `">` +
stdhtml.EscapeString(h.text) + "</a>")
if j := deeperRun(headings, i+1, h.level); j > i+1 {
b.WriteString("\n")
b.WriteString(renderTOCList(headings[i+1 : j]))
i = j - 1
}
b.WriteString("</li>\n")
}
b.WriteString("</ul>\n")
return b.String()
}
// deeperRun returns the end index of the contiguous run of headings
// deeper than level, starting at start.
func deeperRun(headings []tocHeading, start, level int) int {
end := start
for end < len(headings) && headings[end].level > level {
end++
}
return end
}