342 lines
12 KiB
Go
342 lines
12 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|||
|
|
// SPDX-License-Identifier: PolyForm-Noncommercial-1.0.0
|
||
|
|
|
||
|
|
// Package markdown renders Markdown posts to sanitised HTML.
|
||
|
|
//
|
||
|
|
// scriptorium renders the body (CommonMark with the GFM extensions,
|
||
|
|
// footnotes and definition lists). Mathematics and Mermaid diagrams, which
|
||
|
|
// scriptorium renders only when a consumer asks, are recognised here: a
|
||
|
|
// pre-render scan lifts $$…$$ and $…$ runs out of the source into
|
||
|
|
// placeholders and splices the MathML back after rendering, and fenced
|
||
|
|
// mermaid blocks are replaced by their SVG. Images titled with a caption
|
||
|
|
// are wrapped in <figure>/<figcaption>, headings gain id attributes and a
|
||
|
|
// table of contents, and bluemonday strips everything outside a narrow
|
||
|
|
// tag/attribute allowlist.
|
||
|
|
package markdown
|
||
|
|
|
||
|
|
import (
|
||
|
|
"fmt"
|
||
|
|
stdhtml "html"
|
||
|
|
"regexp"
|
||
|
|
"strconv"
|
||
|
|
"strings"
|
||
|
|
"sync"
|
||
|
|
|
||
|
|
"github.com/microcosm-cc/bluemonday"
|
||
|
|
"sourcedock.dev/petrbalvin/scriptorium"
|
||
|
|
)
|
||
|
|
|
||
|
|
// MaxBodyLength caps the Markdown source size (1 MiB).
|
||
|
|
const MaxBodyLength = 1_048_576
|
||
|
|
|
||
|
|
// MaxQuoteDepth bounds how deeply one line may nest blockquote markers.
|
||
|
|
// Measured: rendering cost grows superlinearly with depth (100 000
|
||
|
|
// levels take seconds, and a 1 MiB body of nothing but markers could
|
||
|
|
// reach hundreds of thousands), while legitimate prose never approaches
|
||
|
|
// the bound. It turns the worst case from an unbounded CPU burn into a
|
||
|
|
// rejection.
|
||
|
|
const MaxQuoteDepth = 100
|
||
|
|
|
||
|
|
var (
|
||
|
|
policyOnce sync.Once
|
||
|
|
sanitizer *bluemonday.Policy
|
||
|
|
)
|
||
|
|
|
||
|
|
// tocHeading is one entry of the rendered table of contents.
|
||
|
|
type tocHeading struct {
|
||
|
|
level int
|
||
|
|
id string
|
||
|
|
text string
|
||
|
|
}
|
||
|
|
|
||
|
|
func getPolicy() *bluemonday.Policy {
|
||
|
|
policyOnce.Do(func() {
|
||
|
|
sanitizer = bluemonday.NewPolicy()
|
||
|
|
sanitizer.AllowElements(
|
||
|
|
"a", "abbr", "blockquote", "br", "caption", "code", "del",
|
||
|
|
"dd", "div", "dl", "dt", "em", "figcaption", "figure",
|
||
|
|
"h1", "h2", "h3", "h4", "h5", "h6", "hr", "img", "input",
|
||
|
|
"li", "ol", "p", "pre", "section", "span", "strong", "sub", "sup",
|
||
|
|
"table", "tbody", "td", "th", "thead", "tr", "ul",
|
||
|
|
// SVG, the output of the Mermaid diagram renderer.
|
||
|
|
"svg", "line", "path", "polygon", "rect", "text",
|
||
|
|
// MathML Core, the output of the mathematics renderer.
|
||
|
|
"math", "mi", "mn", "mo", "ms", "mtext", "mspace", "mrow",
|
||
|
|
"mfrac", "msqrt", "mroot", "msub", "msup", "msubsup",
|
||
|
|
"munder", "mover", "munderover", "merror", "mpadded",
|
||
|
|
"mphantom", "mstyle", "mtable", "mtr", "mtd",
|
||
|
|
)
|
||
|
|
// MathML leaves most elements bare: an <mi> carries no attribute
|
||
|
|
// at all, and without this the policy admits only elements that
|
||
|
|
// do.
|
||
|
|
sanitizer.AllowNoAttrs().OnElements(
|
||
|
|
"math", "mi", "mn", "mo", "ms", "mtext", "mspace", "mrow",
|
||
|
|
"mfrac", "msqrt", "mroot", "msub", "msup", "msubsup",
|
||
|
|
"munder", "mover", "munderover", "merror", "mpadded",
|
||
|
|
"mphantom", "mstyle", "mtable", "mtr", "mtd",
|
||
|
|
)
|
||
|
|
sanitizer.AllowAttrs("href", "title", "class", "id", "aria-label", "data-footnote-ref").OnElements("a")
|
||
|
|
sanitizer.AllowAttrs("src", "alt", "title", "width", "height", "loading").OnElements("img")
|
||
|
|
sanitizer.AllowAttrs("class").OnElements("div", "span", "code", "pre", "figure", "figcaption", "section", "li", "sup")
|
||
|
|
sanitizer.AllowAttrs("id").OnElements("h1", "h2", "h3", "h4", "h5", "h6", "section", "li")
|
||
|
|
sanitizer.AllowAttrs("align").OnElements("th", "td")
|
||
|
|
sanitizer.AllowAttrs("type", "checked", "disabled").OnElements("input")
|
||
|
|
sanitizer.AllowAttrs("display", "xmlns").OnElements("math")
|
||
|
|
sanitizer.AllowAttrs(
|
||
|
|
"mathvariant", "stretchy", "accent", "accentunder", "separator",
|
||
|
|
"form", "fence", "lspace", "rspace", "mathcolor",
|
||
|
|
).OnElements("mi", "mn", "mo", "ms", "mtext")
|
||
|
|
sanitizer.AllowAttrs("width").OnElements("mspace")
|
||
|
|
sanitizer.AllowAttrs("displaystyle", "scriptlevel", "mathcolor").OnElements("mstyle")
|
||
|
|
sanitizer.AllowAttrs("columnalign").OnElements("mtable")
|
||
|
|
sanitizer.AllowAttrs("data-footnotes").OnElements("section")
|
||
|
|
// The attributes of the diagram SVG: geometry and paint, the
|
||
|
|
// shapes the renderer draws and the styles a classDef or a
|
||
|
|
// linkStyle statement asks for.
|
||
|
|
sanitizer.AllowAttrs("xmlns", "viewBox", "width", "height", "font-family").OnElements("svg")
|
||
|
|
sanitizer.AllowAttrs("x", "y", "width", "height", "rx").OnElements("rect")
|
||
|
|
sanitizer.AllowAttrs("x1", "y1", "x2", "y2").OnElements("line")
|
||
|
|
sanitizer.AllowAttrs("d").OnElements("path")
|
||
|
|
sanitizer.AllowAttrs("x", "y", "text-anchor").OnElements("text")
|
||
|
|
sanitizer.AllowAttrs("points").OnElements("polygon")
|
||
|
|
sanitizer.AllowAttrs(
|
||
|
|
"fill", "stroke", "stroke-width", "stroke-dasharray",
|
||
|
|
"font-size", "opacity", "style",
|
||
|
|
).OnElements("svg", "line", "path", "polygon", "rect", "text")
|
||
|
|
sanitizer.AllowURLSchemes("http", "https", "mailto")
|
||
|
|
sanitizer.AllowRelativeURLs(true)
|
||
|
|
})
|
||
|
|
return sanitizer
|
||
|
|
}
|
||
|
|
|
||
|
|
// Render renders Markdown to sanitised HTML.
|
||
|
|
func Render(text string) (string, error) {
|
||
|
|
out, _, err := RenderWithTOC(text)
|
||
|
|
return out, err
|
||
|
|
}
|
||
|
|
|
||
|
|
// RenderWithTOC renders Markdown to sanitised HTML and returns
|
||
|
|
// (html, toc_html). toc_html is the sanitised table-of-contents markup,
|
||
|
|
// or an empty string when the body has no headings.
|
||
|
|
func RenderWithTOC(src string) (string, string, error) {
|
||
|
|
if src == "" {
|
||
|
|
return "", "", nil
|
||
|
|
}
|
||
|
|
if len(src) > MaxBodyLength {
|
||
|
|
return "", "", fmt.Errorf("body exceeds %d bytes", MaxBodyLength)
|
||
|
|
}
|
||
|
|
if quoteDepthTooDeep(src) {
|
||
|
|
return "", "", fmt.Errorf("body nests blockquotes deeper than %d levels", MaxQuoteDepth)
|
||
|
|
}
|
||
|
|
rewritten, spans := extractMath(src)
|
||
|
|
html := string(scriptorium.Render([]byte(rewritten)))
|
||
|
|
html = spliceMath(html, spans)
|
||
|
|
html, toc := addHeadingIDs(html)
|
||
|
|
raw := unwrapFigureParagraphs(wrapFigures(html))
|
||
|
|
out := sanitize(raw)
|
||
|
|
// The diagrams are drawn after sanitisation: the SVG is scriptorium's
|
||
|
|
// own output, not authored markup, and the HTML policy's parser
|
||
|
|
// rewrites the case-sensitive viewBox attribute into a form no browser
|
||
|
|
// reads. A diagram the library refuses keeps its code block, which the
|
||
|
|
// sanitiser above has already cleaned like every other one.
|
||
|
|
out = renderDiagrams(out)
|
||
|
|
return out, sanitize(toc), nil
|
||
|
|
}
|
||
|
|
|
||
|
|
// quoteDepthTooDeep reports whether any line nests blockquote markers
|
||
|
|
// beyond MaxQuoteDepth. The markers may be written with or without
|
||
|
|
// spaces between them, so both spellings are counted.
|
||
|
|
func quoteDepthTooDeep(src string) bool {
|
||
|
|
for line := range strings.SplitSeq(src, "\n") {
|
||
|
|
rest := strings.TrimLeft(line, " \t")
|
||
|
|
depth := 0
|
||
|
|
for strings.HasPrefix(rest, ">") {
|
||
|
|
depth++
|
||
|
|
if depth > MaxQuoteDepth {
|
||
|
|
return true
|
||
|
|
}
|
||
|
|
rest = strings.TrimLeft(rest[1:], " \t")
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return false
|
||
|
|
}
|
||
|
|
|
||
|
|
var figureParaRe = regexp.MustCompile(`(?s)<p>(<figure>.*?</figure>)</p>`)
|
||
|
|
|
||
|
|
// unwrapFigureParagraphs lifts figures out of their wrapping paragraph
|
||
|
|
// (<p> cannot contain <figure>).
|
||
|
|
func unwrapFigureParagraphs(in string) string {
|
||
|
|
return figureParaRe.ReplaceAllString(in, "<p></p>$1<p></p>")
|
||
|
|
}
|
||
|
|
|
||
|
|
func sanitize(in string) string {
|
||
|
|
if in == "" {
|
||
|
|
return in
|
||
|
|
}
|
||
|
|
out := getPolicy().Sanitize(in)
|
||
|
|
return addLinkRel(out)
|
||
|
|
}
|
||
|
|
|
||
|
|
// linkTagRe matches anchor start tags in sanitised output.
|
||
|
|
var linkTagRe = regexp.MustCompile(`<a\s[^>]*>`)
|
||
|
|
|
||
|
|
// addLinkRel forces rel="noopener noreferrer" onto every link, matching
|
||
|
|
// the shape consumers expect.
|
||
|
|
func addLinkRel(in string) string {
|
||
|
|
return linkTagRe.ReplaceAllStringFunc(in, func(tag string) string {
|
||
|
|
if strings.Contains(tag, "rel=") {
|
||
|
|
return tag
|
||
|
|
}
|
||
|
|
return `<a rel="noopener noreferrer" ` + strings.TrimPrefix(tag, "<a ")
|
||
|
|
})
|
||
|
|
}
|
||
|
|
|
||
|
|
var (
|
||
|
|
imgTitleRe = regexp.MustCompile(`(?is)<img\s[^>]*\stitle=(?:"[^"]*"|'[^']*')[^>]*/?>`)
|
||
|
|
titleAttrRe = regexp.MustCompile(`(?is)\s+title=(?:"[^"]*"|'[^']*')`)
|
||
|
|
titleValRe = regexp.MustCompile(`(?is)^<img\s[^>]*\stitle=(?:"([^"]*)"|'([^']*)')`)
|
||
|
|
)
|
||
|
|
|
||
|
|
// wrapFigures wraps every <img title="…"> in a <figure> with a
|
||
|
|
// <figcaption>, on the raw pre-sanitisation output so the title
|
||
|
|
// attribute is still present.
|
||
|
|
func wrapFigures(in string) string {
|
||
|
|
return imgTitleRe.ReplaceAllStringFunc(in, func(imgTag string) string {
|
||
|
|
title := ""
|
||
|
|
if m := titleValRe.FindStringSubmatch(imgTag); m != nil {
|
||
|
|
title = m[1]
|
||
|
|
if title == "" {
|
||
|
|
title = m[2]
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// The captured attribute value is HTML-escaped (the renderer
|
||
|
|
// escapes what it writes), so it is unescaped first: escaping it
|
||
|
|
// again would show "&amp;" for a title containing "&".
|
||
|
|
// Unescaping leaves a raw-HTML title the author wrote verbatim,
|
||
|
|
// and the re-escape puts both forms back in canonical shape.
|
||
|
|
caption := stdhtml.EscapeString(stdhtml.UnescapeString(title))
|
||
|
|
cleanImg := titleAttrRe.ReplaceAllString(imgTag, "")
|
||
|
|
return "<figure>" +
|
||
|
|
cleanImg +
|
||
|
|
`<div class="fig-info" aria-hidden="true">i</div>` +
|
||
|
|
"<figcaption>" + caption + "</figcaption>" +
|
||
|
|
"</figure>"
|
||
|
|
})
|
||
|
|
}
|
||
|
|
|
||
|
|
// headingTagRe matches the headings the renderer writes: scriptorium
|
||
|
|
// emits them without attributes, so the pass below is the only source of
|
||
|
|
// their id attributes. RE2 has no backreference, so the func below
|
||
|
|
// checks that the opening and closing levels agree.
|
||
|
|
var headingTagRe = regexp.MustCompile(`(?s)<h([1-6])>(.*?)</h([1-6])>`)
|
||
|
|
|
||
|
|
// innerTagRe strips the inline markup of a heading, leaving its text.
|
||
|
|
var innerTagRe = regexp.MustCompile(`(?s)<[^>]*>`)
|
||
|
|
|
||
|
|
// addHeadingIDs gives every heading an id attribute derived from its
|
||
|
|
// text and returns the table of contents built from the same walk.
|
||
|
|
func addHeadingIDs(in string) (string, string) {
|
||
|
|
used := make(map[string]int)
|
||
|
|
var headings []tocHeading
|
||
|
|
out := headingTagRe.ReplaceAllStringFunc(in, func(m string) string {
|
||
|
|
sub := headingTagRe.FindStringSubmatch(m)
|
||
|
|
level, inner, closeLevel := sub[1], sub[2], sub[3]
|
||
|
|
if level != closeLevel {
|
||
|
|
return m
|
||
|
|
}
|
||
|
|
label := strings.TrimSpace(stdhtml.UnescapeString(innerTagRe.ReplaceAllString(inner, "")))
|
||
|
|
if label == "" {
|
||
|
|
return m
|
||
|
|
}
|
||
|
|
id := headingSlug(label, used)
|
||
|
|
levelNum, _ := strconv.Atoi(level)
|
||
|
|
headings = append(headings, tocHeading{level: levelNum, id: id, text: label})
|
||
|
|
return "<h" + level + ` id="` + id + `">` + inner + "</h" + level + ">"
|
||
|
|
})
|
||
|
|
return out, renderTOC(headings)
|
||
|
|
}
|
||
|
|
|
||
|
|
// headingSlug turns a heading label into a unique id by the rule the
|
||
|
|
// previous renderer established, so the anchors the published pages
|
||
|
|
// already carry keep resolving: ASCII letters and digits, lowercased;
|
||
|
|
// spaces, dashes and underscores as dashes; every other byte, the
|
||
|
|
// diacritics of Czech prose included, dropped. Collisions are told
|
||
|
|
// apart by a numeric suffix.
|
||
|
|
func headingSlug(label string, used map[string]int) string {
|
||
|
|
var b strings.Builder
|
||
|
|
for i := 0; i < len(label); i++ {
|
||
|
|
c := label[i]
|
||
|
|
switch {
|
||
|
|
case c >= 0x80:
|
||
|
|
// A multi-byte rune is dropped whole.
|
||
|
|
continue
|
||
|
|
case 'A' <= c && c <= 'Z':
|
||
|
|
b.WriteByte(c + 'a' - 'A')
|
||
|
|
case 'a' <= c && c <= 'z' || '0' <= c && c <= '9':
|
||
|
|
b.WriteByte(c)
|
||
|
|
case c == ' ' || c == '\t' || c == '-' || c == '_':
|
||
|
|
b.WriteByte('-')
|
||
|
|
}
|
||
|
|
}
|
||
|
|
id := b.String()
|
||
|
|
if id == "" {
|
||
|
|
id = "heading"
|
||
|
|
}
|
||
|
|
if n, ok := used[id]; ok {
|
||
|
|
used[id] = n + 1
|
||
|
|
id = id + "-" + strconv.Itoa(n)
|
||
|
|
}
|
||
|
|
used[id] = 1
|
||
|
|
return id
|
||
|
|
}
|
||
|
|
|
||
|
|
// renderTOC renders the headings as a table of contents in the shape the
|
||
|
|
// API documents:
|
||
|
|
// <div class="toc"><ul><li><a href="#id">Title</a></li></ul></div>.
|
||
|
|
// The wrapper is emitted even with an empty list, so a consumer can rely
|
||
|
|
// on its presence.
|
||
|
|
func renderTOC(headings []tocHeading) string {
|
||
|
|
var b strings.Builder
|
||
|
|
b.WriteString(`<div class="toc">` + "\n")
|
||
|
|
if len(headings) == 0 {
|
||
|
|
b.WriteString("<ul></ul>\n")
|
||
|
|
} else {
|
||
|
|
b.WriteString(renderTOCList(headings))
|
||
|
|
}
|
||
|
|
b.WriteString("</div>\n")
|
||
|
|
return b.String()
|
||
|
|
}
|
||
|
|
|
||
|
|
// renderTOCList renders the headings as nested lists. A run of deeper
|
||
|
|
// headings becomes a sub-list of the heading above it, and a heading
|
||
|
|
// that returns to a shallower level closes the lists it left behind and
|
||
|
|
// continues as a sibling, so a body that opens with a second-level
|
||
|
|
// heading and later uses a first-level one keeps both in the list.
|
||
|
|
func renderTOCList(headings []tocHeading) string {
|
||
|
|
var b strings.Builder
|
||
|
|
b.WriteString("<ul>\n")
|
||
|
|
for i := 0; i < len(headings); i++ {
|
||
|
|
h := headings[i]
|
||
|
|
b.WriteString(`<li><a href="#` + stdhtml.EscapeString(h.id) + `">` +
|
||
|
|
stdhtml.EscapeString(h.text) + "</a>")
|
||
|
|
if j := deeperRun(headings, i+1, h.level); j > i+1 {
|
||
|
|
b.WriteString("\n")
|
||
|
|
b.WriteString(renderTOCList(headings[i+1 : j]))
|
||
|
|
i = j - 1
|
||
|
|
}
|
||
|
|
b.WriteString("</li>\n")
|
||
|
|
}
|
||
|
|
b.WriteString("</ul>\n")
|
||
|
|
return b.String()
|
||
|
|
}
|
||
|
|
|
||
|
|
// deeperRun returns the end index of the contiguous run of headings
|
||
|
|
// deeper than level, starting at start.
|
||
|
|
func deeperRun(headings []tocHeading, start, level int) int {
|
||
|
|
end := start
|
||
|
|
for end < len(headings) && headings[end].level > level {
|
||
|
|
end++
|
||
|
|
}
|
||
|
|
return end
|
||
|
|
}
|